feat(production): sync all modified production files to git

Includes updates across gateway, router, node-worker, memory-service, aurora-service, swapper, sofiia-console UI and node2 infrastructure: - gateway-bot: Dockerfile, http_api.py, druid/aistalk prompts, doc_service - services/router: main.py, router-config.yml, fabric_metrics, memory_retrieval, offload_client, prompt_builder - services/node-worker: worker.py, main.py, config.py, fabric_metrics - services/memory-service: Dockerfile, database.py, main.py, requirements - services/aurora-service: main.py (+399), kling.py, quality_report.py - services/swapper-service: main.py, swapper_config_node2.yaml - services/sofiia-console: static/index.html (console UI update) - config: agent_registry, crewai_agents/teams, router_agents - ops/fabric_preflight.sh: updated preflight checks - router-config.yml, docker-compose.node2.yml: infra updates - docs: NODA1-AGENT-ARCHITECTURE, fabric_contract updated Made-with: Cursor
2026-03-03 07:13:29 -08:00
parent 9aac835882
commit e9dedffa48
35 changed files with 3317 additions and 805 deletions
--- a/services/node-worker/main.py
+++ b/services/node-worker/main.py
@@ -43,7 +43,30 @@ async def prom_metrics():

@app.get("/caps")
 async def caps():
-    """Capability flags for NCS to aggregate."""
+    """Capability flags for NCS to aggregate.
+
+    Semantic vs operational separation (contract):
+    - capabilities.voice_* = semantic availability (provider configured).
+      True as long as the provider is configured, regardless of NATS state.
+      Routing decisions are based on this.
+    - runtime.nats_subscriptions.voice_* = operational (NATS sub active).
+      Used for health/telemetry only — NOT for routing.
+
+    This prevents false-negatives during reconnects / restart races.
+    """
+    import worker as _w
+    nid = config.NODE_ID.lower()
+
+    # Semantic: provider configured → capability is available
+    voice_tts_cap = config.TTS_PROVIDER != "none"
+    voice_stt_cap = config.STT_PROVIDER != "none"
+    voice_llm_cap = True  # LLM always available when node-worker is up
+
+    # Operational: actual NATS subscription state (health/telemetry only)
+    nats_voice_tts_active = f"node.{nid}.voice.tts.request" in _w._VOICE_SUBJECTS
+    nats_voice_stt_active = f"node.{nid}.voice.stt.request" in _w._VOICE_SUBJECTS
+    nats_voice_llm_active = f"node.{nid}.voice.llm.request" in _w._VOICE_SUBJECTS
+
    return {
        "node_id": config.NODE_ID,
        "capabilities": {
@@ -53,6 +76,10 @@ async def caps():
            "tts": config.TTS_PROVIDER != "none",
            "ocr": config.OCR_PROVIDER != "none",
            "image": config.IMAGE_PROVIDER != "none",
+            # Voice HA semantic capability flags (provider-based, not NATS-based)
+            "voice_tts": voice_tts_cap,
+            "voice_llm": voice_llm_cap,
+            "voice_stt": voice_stt_cap,
        },
        "providers": {
            "stt": config.STT_PROVIDER,
@@ -65,6 +92,19 @@ async def caps():
            "vision": config.DEFAULT_VISION,
        },
        "concurrency": config.MAX_CONCURRENCY,
+        "voice_concurrency": {
+            "voice_tts": config.VOICE_MAX_CONCURRENT_TTS,
+            "voice_llm": config.VOICE_MAX_CONCURRENT_LLM,
+            "voice_stt": config.VOICE_MAX_CONCURRENT_STT,
+        },
+        # Operational NATS subscription state — for health/monitoring only
+        "runtime": {
+            "nats_subscriptions": {
+                "voice_tts_active": nats_voice_tts_active,
+                "voice_stt_active": nats_voice_stt_active,
+                "voice_llm_active": nats_voice_llm_active,
+            }
+        },
    }