Files
atc-agents/api/agent_activity.py
T
mo 9008fbd512 feat: Authentik login + switchable GPU prod target
Add OIDC auth for Command Center and runtime GPU endpoint selection
pointed at atc-gpu-prod (10.0.10.106), matching what is currently deployed.
2026-07-21 23:20:24 +00:00

207 lines
9.7 KiB
Python

"""Continuous autonomous activity for the field agents that don't drive a
data-movement loop of their own (Lakehouse Ops, Hadoop Ranger, Infra Sentinel,
Network Watcher).
Every tick the loop runs ONE real, lightweight probe for the next agent in the
rotation and streams the exact command + result into that agent's terminal, so
the operator can always see what each agent is doing in the background instead
of an idle prompt. All probes hit live endpoints (Trino, WebHDFS JMX, YARN,
Dockhand, VLAN hosts) and are individually guarded so a single failure never
breaks the loop.
"""
from __future__ import annotations
import asyncio
import os
import time
from typing import Any
import httpx
TRINO_URL = os.getenv("TRINO_URL", "http://10.0.21.50:8089").rstrip("/")
TRINO_USER = os.getenv("TRINO_USER", "mo")
HDFS_NN_URL = os.getenv("HDFS_NN_URL", "http://10.0.21.61:9870").rstrip("/")
YARN_URL = os.getenv("YARN_URL", "http://10.0.21.62:8088").rstrip("/")
DOCKHAND_URL = os.getenv("DOCKHAND_URL", "http://10.0.21.45:8082").rstrip("/")
DOCKHAND_API_TOKEN = os.getenv("DOCKHAND_API_TOKEN", "")
OBJECTSCALE_URL = os.getenv("OBJECTSCALE_URL", os.getenv("S3_ENDPOINT", "http://10.0.20.111:9020")).rstrip("/")
TICK_SECONDS = float(os.getenv("AGENT_ACTIVITY_TICK_SECONDS", "8"))
# VLAN data paths the Network Watcher keeps an eye on.
_NET_TARGETS = [
("Kafka UI", "http://10.0.21.36:9000"),
("Trino", f"{TRINO_URL}/v1/info"),
("HDFS NameNode", f"{HDFS_NN_URL}/dfshealth.html"),
("Airflow", os.getenv("AIRFLOW_URL", "http://10.0.21.55:8080")),
("ObjectScale S3", OBJECTSCALE_URL),
]
async def _term(agent_id: str, text: str, level: str = "info", phase: str = "ops") -> None:
try:
from agent_terminal import terminal_log
await terminal_log(agent_id, text, level=level, phase=phase)
except Exception:
pass
async def _trino(sql: str, timeout: float = 8.0) -> list[list[Any]]:
rows: list[list[Any]] = []
async with httpx.AsyncClient(timeout=timeout) as client:
d = (await client.post(f"{TRINO_URL}/v1/statement", content=sql.encode(),
headers={"X-Trino-User": TRINO_USER})).json()
for _ in range(40):
if d.get("error"):
raise RuntimeError(d["error"].get("message", "trino error"))
rows += d.get("data") or []
nxt = d.get("nextUri")
if not nxt:
break
d = (await client.get(nxt)).json()
return rows
# ── per-agent probes ─────────────────────────────────────────────────────────
async def _lakehouse_ops() -> None:
aid = "lakehouse-ops"
try:
async with httpx.AsyncClient(timeout=6.0) as client:
info = (await client.get(f"{TRINO_URL}/v1/info")).json()
ver = info.get("nodeVersion", {}).get("version", "?")
up = info.get("uptime", "?")
await _term(aid, f"$ curl -s {TRINO_URL}/v1/info # Trino coordinator health", level="cmd", phase="trino")
await _term(aid, f" ← Trino {ver} · uptime {up} · serving federated queries", level="ok", phase="trino")
except Exception as exc:
await _term(aid, f" ✗ Trino unreachable: {str(exc)[:100]}", level="err", phase="trino")
try:
cats = await _trino("SHOW CATALOGS")
names = ", ".join(sorted(c[0] for c in cats))
await _term(aid, "$ trino --execute 'SHOW CATALOGS'", level="cmd", phase="trino")
await _term(aid, f" ← {len(cats)} catalogs federated: {names}", level="ok", phase="trino")
except Exception as exc:
await _term(aid, f" ✗ SHOW CATALOGS failed: {str(exc)[:100]}", level="err", phase="trino")
try:
sql = "SELECT count(*) FROM iceberg.curated_masked.sales_orders_masked"
rows = await _trino(sql, timeout=12.0)
n = int(rows[0][0]) if rows else 0
await _term(aid, f"$ trino --execute '{sql}' # curated Iceberg lakehouse", level="cmd", phase="iceberg")
await _term(aid, f" ← {n:,} masked rows in iceberg.curated_masked (PII-safe layer)", level="ok", phase="iceberg")
except Exception as exc:
await _term(aid, f" ✗ Iceberg count failed: {str(exc)[:100]}", level="err", phase="iceberg")
def _g(d: dict, *keys) -> Any:
for k in keys:
if k in d:
return d[k]
return None
async def _hadoop_ranger() -> None:
aid = "hadoop-ranger"
try:
async with httpx.AsyncClient(timeout=6.0) as client:
fs = (await client.get(f"{HDFS_NN_URL}/jmx?qry=Hadoop:service=NameNode,name=FSNamesystem")).json()
state = (await client.get(f"{HDFS_NN_URL}/jmx?qry=Hadoop:service=NameNode,name=FSNamesystemState")).json()
fsb = (fs.get("beans") or [{}])[0]
stb = (state.get("beans") or [{}])[0]
cap_total = float(_g(fsb, "CapacityTotalGB") or 0)
cap_used = float(_g(fsb, "CapacityUsedGB") or 0)
blocks = int(_g(fsb, "BlocksTotal", "TotalBlocks") or _g(stb, "BlocksTotal") or 0)
live = int(_g(stb, "NumLiveDataNodes") or 0)
dead = int(_g(stb, "NumDeadDataNodes") or 0)
pct = (100.0 * cap_used / cap_total) if cap_total else 0.0
await _term(aid, f"$ curl -s {HDFS_NN_URL}/jmx?qry=...FSNamesystemState # NameNode health", level="cmd", phase="hdfs")
await _term(aid, f" ← live datanodes={live} dead={dead} · blocks={blocks:,} · "
f"used {cap_used:.1f}/{cap_total:.1f} GB ({pct:.0f}%)", level="ok", phase="hdfs")
except Exception as exc:
await _term(aid, f" ✗ NameNode JMX unreachable: {str(exc)[:100]}", level="err", phase="hdfs")
try:
async with httpx.AsyncClient(timeout=6.0) as client:
m = (await client.get(f"{YARN_URL}/ws/v1/cluster/metrics")).json().get("clusterMetrics", {})
await _term(aid, f"$ yarn application -list # ResourceManager {YARN_URL}", level="cmd", phase="yarn")
await _term(aid, f" ← apps running={m.get('appsRunning', 0)} pending={m.get('appsPending', 0)} · "
f"available {round(m.get('availableMB', 0) / 1024, 1)} GB / "
f"{m.get('totalNodes', 0)} nodes", level="ok", phase="yarn")
except Exception as exc:
await _term(aid, f" ⚠ YARN RM not responding ({str(exc)[:60]}) — HDFS storage layer still healthy", level="warn", phase="yarn")
async def _infra_sentinel() -> None:
aid = "infra-sentinel"
await _term(aid, f"$ dockhand ps --all-envs # container inventory via {DOCKHAND_URL}", level="cmd", phase="docker")
try:
from dockhand_envs import DOCKHAND_ENVS
envs = DOCKHAND_ENVS
except Exception:
envs = {"docker01": 1, "docker02": 2, "lakehouse": 9, "airflow": 10, "db02": 5}
total = running = 0
reached = 0
try:
async with httpx.AsyncClient(timeout=4.0) as client:
async def _one(name: str, eid: int):
try:
hdr = {"Authorization": f"Bearer {DOCKHAND_API_TOKEN}"} if DOCKHAND_API_TOKEN else {}
r = await client.get(f"{DOCKHAND_URL}/api/containers", params={"env": eid}, headers=hdr)
if r.status_code >= 400:
return None
d = r.json()
return d if isinstance(d, list) else d.get("containers", [])
except Exception:
return None
results = await asyncio.gather(*[_one(n, e) for n, e in envs.items()])
for conts in results:
if conts is None:
continue
reached += 1
total += len(conts)
running += sum(1 for c in conts
if str(c.get("state", c.get("status", ""))).lower().startswith(("run", "up")))
if reached:
await _term(aid, f" ← {running}/{total} containers up across {reached} Dockhand environments", level="ok", phase="docker")
else:
await _term(aid, " ✗ Dockhand returned no environments", level="warn", phase="docker")
except Exception as exc:
await _term(aid, f" ✗ Dockhand unreachable: {str(exc)[:100]}", level="err", phase="docker")
try:
load = os.getloadavg()
await _term(aid, "$ cat /proc/loadavg # command-center host", level="cmd", phase="host")
await _term(aid, f" ← load avg {load[0]:.2f} {load[1]:.2f} {load[2]:.2f} (1/5/15m)", level="ok", phase="host")
except Exception:
pass
async def _network_watcher() -> None:
aid = "network-watcher"
await _term(aid, "$ probe VLAN 20/21 data paths # ingress/egress reachability", level="cmd", phase="net")
async with httpx.AsyncClient(timeout=4.0, verify=False) as client:
for label, url in _NET_TARGETS:
t0 = time.time()
try:
r = await client.get(url)
ms = int((time.time() - t0) * 1000)
lvl = "ok" if r.status_code < 500 else "warn"
await _term(aid, f" → {label:<16} {url} {r.status_code} {ms}ms", level=lvl, phase="net")
except Exception as exc:
await _term(aid, f" → {label:<16} {url} DOWN ({str(exc)[:60]})", level="err", phase="net")
_ROTATION = [_lakehouse_ops, _hadoop_ranger, _infra_sentinel, _network_watcher]
async def agent_activity_loop() -> None:
"""Round-robin: run one agent's live probe per tick so each agent terminal
shows fresh real activity roughly every (len(rotation) * TICK_SECONDS)s."""
await asyncio.sleep(15) # let the platform settle
idx = 0
while True:
probe = _ROTATION[idx % len(_ROTATION)]
idx += 1
try:
await probe()
except Exception:
pass
await asyncio.sleep(max(3.0, TICK_SECONDS))