mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-08-23 13:02:23 +02:00
534 lines
20 KiB
Python
534 lines
20 KiB
Python
"""Service SubApp.
|
|
|
|
Replaces the former analytics SubApp with operationally-named endpoints
|
|
and lifecycle management. Responsibilities:
|
|
|
|
- Usage-summary and cost-breakdown endpoints (user-facing, for the
|
|
Settings / Usage page)
|
|
- Background heartbeat that reports operational state to the cloud
|
|
- 9Router auto-start for OpenSwarm Pro users
|
|
- Frontend event endpoint (`POST /api/service/event`)
|
|
- Periodic spool drainer for offline retry
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import json
|
|
import logging
|
|
import os
|
|
import platform
|
|
from collections import Counter
|
|
from contextlib import asynccontextmanager
|
|
from datetime import datetime
|
|
|
|
from fastapi import Body
|
|
|
|
from backend.config.Apps import SubApp
|
|
from backend.config.paths import SESSIONS_DIR
|
|
from backend.apps.service import client as svc
|
|
from backend.apps.service.version import APP_VERSION, read_app_version
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
p_pulse_task: asyncio.Task | None = None
|
|
p_drain_task: asyncio.Task | None = None
|
|
p_9r_start_task: asyncio.Task | None = None
|
|
|
|
p_last_9r_cost: float | None = None
|
|
p_last_9r_prompt_tokens: int | None = None
|
|
p_last_9r_completion_tokens: int | None = None
|
|
p_last_9r_requests: int | None = None
|
|
P_RESTART_THRESHOLD = 1.0
|
|
|
|
|
|
def p_compute_delta(current: float, last: float | None, threshold: float = P_RESTART_THRESHOLD) -> tuple[float, float]:
|
|
if last is None:
|
|
return 0.0, current
|
|
if current < last - threshold:
|
|
return current, current
|
|
if current < last:
|
|
return 0.0, last
|
|
return current - last, current
|
|
|
|
|
|
p_pulse_count = 0
|
|
p_pulse_hours: set = set()
|
|
p_pulse_delta_cost_total = 0.0
|
|
p_pulse_batch_size = 10
|
|
|
|
|
|
async def p_pulse_loop():
|
|
"""Periodic state-pulse loop. Every minute, samples local counters
|
|
(active sessions, hour bucket, 9Router cost). Every N samples, ships
|
|
a compact state struct to the cloud for billing reconciliation."""
|
|
global p_last_9r_cost, p_last_9r_prompt_tokens, p_last_9r_completion_tokens, p_last_9r_requests
|
|
global p_pulse_count, p_pulse_hours, p_pulse_delta_cost_total
|
|
|
|
while True:
|
|
await asyncio.sleep(60)
|
|
p_pulse_count += 1
|
|
try:
|
|
import datetime as p_dt
|
|
p_pulse_hours.add(p_dt.datetime.now().hour)
|
|
except Exception:
|
|
pass
|
|
|
|
cost_delta = 0.0
|
|
try:
|
|
from backend.apps.nine_router import get_usage_stats, is_running as p_9r_running
|
|
if p_9r_running():
|
|
stats = await get_usage_stats()
|
|
if stats:
|
|
cur_cost = stats.get("totalCost", 0) or 0
|
|
cur_prompt = stats.get("totalPromptTokens", 0) or 0
|
|
cur_completion = stats.get("totalCompletionTokens", 0) or 0
|
|
cur_requests = stats.get("totalRequests", 0) or 0
|
|
cost_delta, p_last_9r_cost = p_compute_delta(cur_cost, p_last_9r_cost)
|
|
prompt_delta, p_last_9r_prompt_tokens = p_compute_delta(cur_prompt, p_last_9r_prompt_tokens, threshold=1000)
|
|
completion_delta, p_last_9r_completion_tokens = p_compute_delta(cur_completion, p_last_9r_completion_tokens, threshold=1000)
|
|
requests_delta, p_last_9r_requests = p_compute_delta(cur_requests, p_last_9r_requests, threshold=10)
|
|
p_pulse_delta_cost_total += cost_delta
|
|
except Exception:
|
|
pass
|
|
|
|
if p_pulse_count >= p_pulse_batch_size:
|
|
try:
|
|
from backend.apps.agents.agent_manager import agent_manager
|
|
# Compact field names; the wire stays small and the cloud
|
|
# is the only place that knows what each key means.
|
|
svc.sync({
|
|
"a": len(agent_manager.sessions), # active sessions
|
|
"h": sorted(p_pulse_hours), # hour bucket set
|
|
"n": p_pulse_count, # samples in batch
|
|
"c": p_last_9r_cost or 0, # cumulative cost
|
|
"d1": p_pulse_delta_cost_total, # cost delta since last batch
|
|
})
|
|
except Exception:
|
|
pass
|
|
p_pulse_count = 0
|
|
p_pulse_hours = set()
|
|
p_pulse_delta_cost_total = 0.0
|
|
|
|
|
|
async def p_drain_loop():
|
|
while True:
|
|
try:
|
|
await svc.drain_spool()
|
|
except Exception:
|
|
pass
|
|
await asyncio.sleep(60)
|
|
|
|
|
|
@asynccontextmanager
|
|
async def service_lifespan():
|
|
global p_pulse_task, p_drain_task, p_9r_start_task
|
|
|
|
try:
|
|
from backend.apps.settings.settings import load_settings, save_settings
|
|
settings = load_settings()
|
|
|
|
is_first_open = settings.first_opened_at is None
|
|
if is_first_open:
|
|
settings.first_opened_at = datetime.now().isoformat()
|
|
save_settings(settings)
|
|
|
|
days_since_install = 0
|
|
if settings.first_opened_at:
|
|
try:
|
|
first = datetime.fromisoformat(settings.first_opened_at[:19])
|
|
days_since_install = (datetime.now() - first).days
|
|
except Exception:
|
|
pass
|
|
|
|
providers = []
|
|
if getattr(settings, "anthropic_api_key", None):
|
|
providers.append("anthropic")
|
|
if getattr(settings, "openai_api_key", None):
|
|
providers.append("openai")
|
|
if getattr(settings, "google_api_key", None):
|
|
providers.append("gemini")
|
|
if getattr(settings, "openrouter_api_key", None):
|
|
providers.append("openrouter")
|
|
for cp in getattr(settings, "custom_providers", []):
|
|
providers.append(cp.name)
|
|
|
|
svc.sync({
|
|
"os": platform.system(),
|
|
"platform": platform.platform(),
|
|
"provider_count": len(providers),
|
|
"providers": providers,
|
|
"is_first_open": is_first_open,
|
|
"days_since_install": days_since_install,
|
|
"app_version": APP_VERSION,
|
|
})
|
|
|
|
id_props: dict = {
|
|
"providers_configured": providers,
|
|
"provider_count": len(providers),
|
|
"app_version": APP_VERSION,
|
|
}
|
|
if getattr(settings, "user_email", None):
|
|
id_props["email"] = settings.user_email
|
|
if getattr(settings, "user_name", None):
|
|
id_props["name"] = settings.user_name
|
|
if getattr(settings, "user_use_case", None):
|
|
id_props["use_case"] = settings.user_use_case
|
|
if getattr(settings, "user_referral_source", None):
|
|
id_props["referral_source"] = settings.user_referral_source
|
|
|
|
mode = getattr(settings, "connection_mode", "own_key")
|
|
plan = getattr(settings, "openswarm_subscription_plan", None)
|
|
is_paying = mode == "openswarm-pro" and bool(
|
|
getattr(settings, "openswarm_bearer_token", None)
|
|
)
|
|
id_props["connection_mode"] = mode
|
|
id_props["plan"] = plan if is_paying else "free"
|
|
id_props["is_paying_customer"] = is_paying
|
|
if is_paying and getattr(settings, "openswarm_subscription_expires", None):
|
|
id_props["subscription_expires"] = settings.openswarm_subscription_expires
|
|
|
|
svc.sync({"identity": id_props})
|
|
|
|
# First-boot log write doubles as the token-registration trigger.
|
|
from backend.apps.service.analytics.client import get_analytics_client, track_link_email
|
|
analytics_client = get_analytics_client()
|
|
if analytics_client is not None:
|
|
try:
|
|
analytics_client.logs.write(tag="app", subtag="backend_started", data={"app_version": APP_VERSION})
|
|
except Exception:
|
|
pass
|
|
track_link_email(getattr(settings, "user_email", None))
|
|
except Exception as e:
|
|
logger.debug(f"Service startup event failed (non-critical): {e}")
|
|
|
|
try:
|
|
from backend.apps.nine_router import ensure_running as ensure_9router
|
|
# Start 9Router in the BACKGROUND instead of awaiting it here. Awaiting
|
|
# it was ~7s (up to ~18s cold) of the startup critical path, blocking the
|
|
# HTTP bind and the whole UI behind it. 9Router is only needed when the
|
|
# user sends an agent message, and the dispatch path calls ensure_running()
|
|
# itself (now serialized, so no double-spawn), so the first message waits
|
|
# for readiness lazily. This is the single biggest warm-startup win.
|
|
p_9r_start_task = asyncio.create_task(ensure_9router())
|
|
except Exception as e:
|
|
logger.debug(f"9Router auto-start skipped: {e}")
|
|
|
|
p_pulse_task = asyncio.create_task(p_pulse_loop())
|
|
p_drain_task = asyncio.create_task(p_drain_loop())
|
|
|
|
yield
|
|
|
|
if p_pulse_task:
|
|
p_pulse_task.cancel()
|
|
try:
|
|
await p_pulse_task
|
|
except asyncio.CancelledError:
|
|
pass
|
|
p_pulse_task = None
|
|
|
|
if p_drain_task:
|
|
p_drain_task.cancel()
|
|
try:
|
|
await p_drain_task
|
|
except asyncio.CancelledError:
|
|
pass
|
|
p_drain_task = None
|
|
|
|
if p_9r_start_task and not p_9r_start_task.done():
|
|
p_9r_start_task.cancel()
|
|
try:
|
|
await p_9r_start_task
|
|
except (asyncio.CancelledError, Exception):
|
|
pass
|
|
p_9r_start_task = None
|
|
|
|
try:
|
|
from backend.apps.nine_router import stop as stop_9router
|
|
stop_9router()
|
|
except Exception:
|
|
pass
|
|
|
|
# Flush before the process exits or buffered events are lost.
|
|
try:
|
|
from backend.apps.service.analytics.client import track_app_closed, shutdown_analytics
|
|
track_app_closed()
|
|
shutdown_analytics()
|
|
except Exception:
|
|
pass
|
|
|
|
logger.info("Service shut down")
|
|
|
|
|
|
service = SubApp("service", service_lifespan)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Usage endpoints (user-facing, read by the Settings / Usage page)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def p_load_all_sessions() -> list[dict]:
|
|
results = []
|
|
if not os.path.exists(SESSIONS_DIR):
|
|
return results
|
|
for fname in os.listdir(SESSIONS_DIR):
|
|
if fname.endswith(".json"):
|
|
try:
|
|
with open(os.path.join(SESSIONS_DIR, fname)) as f:
|
|
results.append(json.load(f))
|
|
except Exception:
|
|
pass
|
|
return results
|
|
|
|
|
|
@service.router.get("/usage-summary")
|
|
async def usage_summary():
|
|
from backend.apps.agents.agent_manager import agent_manager
|
|
|
|
sessions = p_load_all_sessions()
|
|
for s in agent_manager.get_all_sessions():
|
|
sessions.append(s.model_dump(mode="json"))
|
|
|
|
def p_is_real(sess: dict) -> bool:
|
|
# "Real" = actually ran. Empty draft/abandoned sessions (no assistant turn, no tokens,
|
|
# no active time) otherwise inflate the count and drag every average toward zero.
|
|
if (sess.get("agent_active_ms") or 0) > 0 or (sess.get("cost_usd") or 0) > 0:
|
|
return True
|
|
tk = sess.get("tokens") or {}
|
|
if (tk.get("input") or 0) > 0 or (tk.get("output") or 0) > 0:
|
|
return True
|
|
return any(m.get("role") == "assistant" for m in sess.get("messages", []))
|
|
|
|
sessions = [s for s in sessions if p_is_real(s)]
|
|
|
|
total_sessions = len(sessions)
|
|
total_cost = sum(s.get("cost_usd", 0) for s in sessions)
|
|
total_messages = 0
|
|
total_tool_calls = 0
|
|
total_run_seconds = 0.0
|
|
timed_sessions = 0
|
|
model_counts: Counter = Counter()
|
|
provider_counts: Counter = Counter()
|
|
tool_counts: Counter = Counter()
|
|
status_counts: Counter = Counter()
|
|
|
|
for s in sessions:
|
|
messages = s.get("messages", [])
|
|
total_messages += sum(1 for m in messages if m.get("role") in ("user", "assistant"))
|
|
model_counts[s.get("model", "unknown")] += 1
|
|
provider_counts[s.get("provider", "anthropic")] += 1
|
|
status_counts[s.get("status", "unknown")] += 1
|
|
|
|
# Tool calls: tool_latencies carries authoritative per-tool counts; older sessions only have
|
|
# the sparse tool_call messages. Per session take whichever source recorded more so we never
|
|
# undercount what's on record (and so the total never drops below the old message-only count).
|
|
lat_counts: Counter = Counter()
|
|
for tool, d in (s.get("tool_latencies") or {}).items():
|
|
cnt = (d or {}).get("count", 0) or 0
|
|
if tool and cnt:
|
|
lat_counts[tool] += cnt
|
|
msg_counts: Counter = Counter()
|
|
for m in messages:
|
|
if m.get("role") == "tool_call":
|
|
content = m.get("content", {})
|
|
name = content.get("tool") if isinstance(content, dict) else None
|
|
msg_counts[name or "tool"] += 1
|
|
chosen = lat_counts if sum(lat_counts.values()) >= sum(msg_counts.values()) else msg_counts
|
|
total_tool_calls += sum(chosen.values())
|
|
tool_counts.update(chosen)
|
|
|
|
# Run time: real agent-active time when tracked, else session wall-clock as a rough proxy.
|
|
run_s = (s.get("agent_active_ms") or 0) / 1000.0
|
|
if run_s <= 0:
|
|
created, closed = s.get("created_at"), s.get("closed_at")
|
|
if created and closed:
|
|
try:
|
|
run_s = (datetime.fromisoformat(closed[:19]) - datetime.fromisoformat(created[:19])).total_seconds()
|
|
except Exception:
|
|
run_s = 0
|
|
if run_s > 0:
|
|
total_run_seconds += run_s
|
|
timed_sessions += 1
|
|
|
|
avg_duration = total_run_seconds / timed_sessions if timed_sessions > 0 else 0
|
|
completed = status_counts.get("completed", 0)
|
|
completion_rate = completed / total_sessions if total_sessions > 0 else 0
|
|
|
|
from backend.apps.nine_router import get_usage_stats, is_running as p_9r_running
|
|
nine_router_stats = await get_usage_stats() if p_9r_running() else None
|
|
|
|
if nine_router_stats and nine_router_stats.get("totalCost", 0) > 0:
|
|
cost_source = "9router"
|
|
total_cost = nine_router_stats["totalCost"]
|
|
elif total_cost > 0:
|
|
cost_source = "sdk"
|
|
else:
|
|
cost_source = "none"
|
|
|
|
avg_cost = total_cost / total_sessions if total_sessions > 0 else 0
|
|
|
|
cost_by_model = {}
|
|
cost_by_provider = {}
|
|
total_prompt_tokens = 0
|
|
total_completion_tokens = 0
|
|
total_requests = 0
|
|
|
|
if nine_router_stats:
|
|
total_prompt_tokens = nine_router_stats.get("totalPromptTokens", 0)
|
|
total_completion_tokens = nine_router_stats.get("totalCompletionTokens", 0)
|
|
total_requests = nine_router_stats.get("totalRequests", 0)
|
|
for key, val in (nine_router_stats.get("byModel") or {}).items():
|
|
cost_by_model[key] = {
|
|
"cost": val.get("cost", 0),
|
|
"requests": val.get("count", 0),
|
|
"prompt_tokens": val.get("promptTokens", 0),
|
|
"completion_tokens": val.get("completionTokens", 0),
|
|
}
|
|
for key, val in (nine_router_stats.get("byProvider") or {}).items():
|
|
cost_by_provider[key] = {
|
|
"cost": val.get("cost", 0),
|
|
"requests": val.get("count", 0),
|
|
}
|
|
|
|
return {
|
|
"total_sessions": total_sessions,
|
|
"total_cost_usd": round(total_cost, 4),
|
|
"total_messages": total_messages,
|
|
"total_tool_calls": total_tool_calls,
|
|
"total_run_seconds": round(total_run_seconds, 1),
|
|
"avg_duration_seconds": round(avg_duration, 1),
|
|
"avg_cost_per_session": round(avg_cost, 4),
|
|
"completion_rate": round(completion_rate, 3),
|
|
"models_used": dict(model_counts.most_common(10)),
|
|
"providers_used": dict(provider_counts.most_common(10)),
|
|
"top_tools": dict(tool_counts.most_common(15)),
|
|
"status_breakdown": dict(status_counts),
|
|
"total_prompt_tokens": total_prompt_tokens,
|
|
"total_completion_tokens": total_completion_tokens,
|
|
"cost_by_model": cost_by_model,
|
|
"cost_by_provider": cost_by_provider,
|
|
"cost_source": cost_source,
|
|
"nine_router_available": nine_router_stats is not None,
|
|
"total_requests": total_requests,
|
|
}
|
|
|
|
|
|
@service.router.get("/cost-breakdown")
|
|
async def cost_breakdown(period: str = "7d"):
|
|
from backend.apps.nine_router import get_usage_stats, is_running as p_9r_running
|
|
if not p_9r_running():
|
|
return {"available": False, "by_model": {}, "by_provider": {}}
|
|
stats = await get_usage_stats(period)
|
|
if not stats:
|
|
return {"available": False, "by_model": {}, "by_provider": {}}
|
|
return {
|
|
"available": True,
|
|
"period": period,
|
|
"total_cost": stats.get("totalCost", 0),
|
|
"total_requests": stats.get("totalRequests", 0),
|
|
"total_prompt_tokens": stats.get("totalPromptTokens", 0),
|
|
"total_completion_tokens": stats.get("totalCompletionTokens", 0),
|
|
"by_model": stats.get("byModel", {}),
|
|
"by_provider": stats.get("byProvider", {}),
|
|
}
|
|
|
|
|
|
@service.router.get("/status")
|
|
async def service_status():
|
|
return {"status": "ok", "enabled": True}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Frontend event endpoints
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def p_bridge_to_analytics(item: dict) -> None:
|
|
# Boundary adapter: validate the raw report() envelope into a typed event, hand it to the analytics bridge.
|
|
from backend.apps.service.analytics.frontend_bridge import bridge_frontend_event, FrontendEvent
|
|
try:
|
|
bridge_frontend_event(FrontendEvent.model_validate(item))
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
@service.router.post("/submit")
|
|
async def post_submit(body=Body(...)):
|
|
"""Accepts three body shapes for backward compatibility:
|
|
|
|
1. Frontend `report()` shape; flat `{s, a, p, submission_id, t}`.
|
|
This is what `frontend/src/shared/serviceClient.ts:report()` sends
|
|
on every UI interaction. Pass through unchanged so the cloud sees
|
|
it as a frontend.event.
|
|
|
|
2. Legacy `{kind, payload}` shape; used by older callers that wrapped
|
|
the payload in a kind+payload envelope before submitting. Unwrap
|
|
and forward the payload.
|
|
|
|
3. Batched array; frontend collects up to 1s of events and sends them
|
|
as a single JSON array to cut N POSTs/sec down to 1. Each item is
|
|
processed exactly as if it had arrived as its own request.
|
|
|
|
Pre-fix this endpoint required shape #2 and silently rejected shape #1
|
|
with a 200 + `{ok:false}`, so every UI event from `report()` was
|
|
dropped; `frontend.event` count was 0 in production analytics.
|
|
"""
|
|
# Shape 3: batched array. Recurse per-item so single-item handling
|
|
# logic stays in one place. Returns a single ok regardless of
|
|
# individual item shape; analytics calls aren't transactional.
|
|
if isinstance(body, list):
|
|
for item in body:
|
|
if isinstance(item, dict):
|
|
if any(k in item for k in ("s", "a", "p")):
|
|
svc.sync(item)
|
|
p_bridge_to_analytics(item)
|
|
continue
|
|
kind = item.get("kind") or ""
|
|
payload = item.get("payload") or {}
|
|
if isinstance(payload, dict):
|
|
payload.setdefault("kind", kind)
|
|
svc.sync(payload)
|
|
return {"ok": True}
|
|
if not isinstance(body, dict):
|
|
return {"ok": False, "error": "JSON object or array required"}
|
|
# Shape 1: frontend `report()`; flat {s, a, p, ...}
|
|
if any(k in body for k in ("s", "a", "p")):
|
|
svc.sync(body)
|
|
p_bridge_to_analytics(body)
|
|
return {"ok": True}
|
|
# Shape 2: legacy {kind, payload}
|
|
kind = body.get("kind") or ""
|
|
payload = body.get("payload")
|
|
if kind and isinstance(payload, dict):
|
|
svc.sync(payload)
|
|
return {"ok": True}
|
|
return {"ok": False, "error": "expected {s,a,p,...} or {kind,payload}"}
|
|
|
|
|
|
@service.router.post("/event")
|
|
async def post_event(body: dict):
|
|
surface = body.get("surface") or body.get("event_type") or ""
|
|
action = body.get("action") or ""
|
|
|
|
# Legacy path: frontend sends {event_type: "foo.bar", properties: {...}}
|
|
if not action and "." in surface:
|
|
surface, action = surface.split(".", 1)
|
|
if not surface:
|
|
return {"ok": False, "error": "surface required"}
|
|
if not action:
|
|
action = "fired"
|
|
|
|
envelope = {
|
|
"s": str(surface)[:64],
|
|
"a": str(action)[:64],
|
|
"p": body.get("props") or body.get("properties") or {},
|
|
}
|
|
svc.sync(envelope)
|
|
p_bridge_to_analytics(envelope)
|
|
return {"ok": True}
|
|
|
|
|
|
@service.router.get("/spool/count")
|
|
async def spool_count():
|
|
from backend.apps.service import buffer
|
|
return {"pending": buffer.count(svc.spool_path())}
|