Skip to content

Commit 955e38c

Browse files
committed
merge origin/main (watchdog #272/#273) into reseller-portal
2 parents a4eb346 + e852935 commit 955e38c

1 file changed

Lines changed: 20 additions & 9 deletions

File tree

‎agents/watchdog.py‎

Lines changed: 20 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -12,11 +12,12 @@
1212

1313
# --- Discovery-surface reconciliation ---------------------------------------
1414
# Two surfaces agents discover us through: the MCP tool catalog (Smithery
15-
# listing) and the A2A Agent-Card. When we add tools/skills but forget to
16-
# re-publish, discovery goes stale SILENTLY (e.g. server exposes 44 tools while
17-
# Smithery still lists 39). This reconciles what we actually serve against each
18-
# listing and alerts on the mismatch — "did the listing keep up", not "did a run
19-
# error". The Smithery registry is queryable (registry.smithery.ai).
15+
# listing) and the A2A Agent-Card. When the origin gains/loses a tool but the
16+
# Smithery listing hasn't re-scanned, discovery goes stale SILENTLY. With Option
17+
# B (Smithery lists remote-at-origin, api.moltrust.ch/mcp), the steady state is
18+
# origin == listing → Δ0. Any Δ = the Smithery remote needs a re-scan. This
19+
# reconciles what we serve against each listing — "did the listing keep up", not
20+
# "did a run error". The Smithery registry is queryable (registry.smithery.ai).
2021
MCP_LOCAL_URL = "http://127.0.0.1:8002/mcp"
2122
SMITHERY_REGISTRY_URL = "https://registry.smithery.ai/servers/@moltrust/moltrust-mcp-server"
2223
AGENT_CARD_URL = "https://api.moltrust.ch/.well-known/agent-card.json"
@@ -166,15 +167,25 @@ def check_discovery_drift(now: datetime.datetime) -> list:
166167
"detail": "tools/list unreachable (mcp_http :8002 down?)"})
167168
else:
168169
try:
169-
sm = httpx.get(SMITHERY_REGISTRY_URL, timeout=12.0).json()
170+
# cache-bust: registry.smithery.ai sits behind Cloudflare (max-age 4h,
171+
# stale-while-revalidate 24h) — the plain URL can lag a real change by
172+
# hours (verified 2026-07-18: cached 39 vs fresh 53). Force a fresh
173+
# read so a legit re-scan doesn't trigger a day of false drift alarms.
174+
sm = httpx.get(SMITHERY_REGISTRY_URL, params={"_cb": int(now.timestamp())},
175+
headers={"Cache-Control": "no-cache", "Pragma": "no-cache"},
176+
timeout=12.0).json()
170177
listed = len(sm.get("tools") or [])
178+
# Steady state (Option B: Smithery lists remote-at-origin) is
179+
# origin == listing → Δ0 → silent. Any Δ is real drift: the origin
180+
# gained/lost a tool and the Smithery remote hasn't re-scanned.
171181
if live != listed:
172182
out.append({"surface": "MCP↔Smithery", "ok": False,
173-
"detail": f"server exposes {live} tools, Smithery lists {listed} "
174-
f"(Δ{live - listed}) — re-publish the Smithery listing"})
183+
"detail": f"origin exposes {live} tools, Smithery lists {listed} "
184+
f"(Δ{live - listed}) — Smithery remote out of sync with the "
185+
f"origin; re-scan/redeploy the Smithery listing"})
175186
else:
176187
out.append({"surface": "MCP↔Smithery", "ok": True,
177-
"detail": f"{live} tools in sync"})
188+
"detail": f"{live} tools in sync (origin == listing)"})
178189
except Exception as e:
179190
# A Smithery registry outage must not masquerade as our drift.
180191
out.append({"surface": "MCP↔Smithery", "ok": True,

0 commit comments

Comments
 (0)