diff --git a/backend/apps/agents/agent_manager.py b/backend/apps/agents/agent_manager.py index 81da2584..6c1ba332 100644 --- a/backend/apps/agents/agent_manager.py +++ b/backend/apps/agents/agent_manager.py @@ -22,7 +22,7 @@ from backend.apps.agents.manager.streaming.PartialReply import PartialReply from backend.apps.agents.manager.session.SessionLifecycle import SessionLifecycle from backend.apps.agents.manager.SpawnAgentRun import SpawnAgentRun from backend.apps.agents.manager.session.SessionPersistence import SessionPersistence -from backend.apps.agents.manager.Messaging import Messaging +from backend.apps.agents.manager.Messaging import Messaging, QueuedMessage from backend.apps.agents.manager.SessionControl import SessionControl from backend.apps.agents.manager.AgentLaunch import AgentLaunch from backend.apps.agents.manager.MockAgent import MockAgent @@ -55,6 +55,8 @@ class AgentManager(SessionLifecycle, SessionPersistence, Messaging, SessionContr # Per-SESSION hook context + stderr buffer, updated in place each turn: a persistent client's hooks/stderr callback were bound at connect, so they must read stable objects, not per-turn rebuilds. self.hook_ctxs: Dict[str, HookContext] = {} self.stderr_buffers: Dict[str, List[str]] = {} + # Messages typed while a turn was live, replayed in order when it ends (see Messaging). + self.pending_messages: Dict[str, List[QueuedMessage]] = {} # Admission gate: one shared semaphore caps concurrent ROOT turns (children bypass). (Re)created per running loop by get_turn_admission so it never binds to a dead loop across a uvicorn reload or a test's asyncio.run. self.p_turn_admission_sema: Optional[asyncio.Semaphore] = None self.p_turn_admission_loop: Optional[asyncio.AbstractEventLoop] = None diff --git a/backend/apps/agents/agents.py b/backend/apps/agents/agents.py index 2b21d5cb..a0b54f8f 100644 --- a/backend/apps/agents/agents.py +++ b/backend/apps/agents/agents.py @@ -438,10 +438,11 @@ async def subscriptions_connect(body: dict, request: Request): raise HTTPException(status_code=503, detail="9Router not available. Please install Node.js.") # Reconnecting gemini-cli must wipe antigravity; registry prefers AG and a stale AG token would 400 after gemini-cli refreshes. - cascade = P_PROVIDER_CASCADE_REMOVES.get(provider, []) + from backend.apps.agents.disconnect_subscription import PROVIDER_CASCADE_REMOVES, delete_provider_connections + cascade = PROVIDER_CASCADE_REMOVES.get(provider, []) if cascade: try: - await p_delete_provider_connections(cascade) + await delete_provider_connections(cascade) except Exception: pass @@ -894,47 +895,12 @@ async def list_models(): return {"models": result, "notes": notes} -# gemini-cli and antigravity are two Google OAuth lanes; registry prefers AG, so we cascade-wipe AG when reconnecting gemini-cli to avoid stale-AG 400s. One-directional: AG operations MUST NOT cascade back. -P_PROVIDER_CASCADE_REMOVES: dict[str, list[str]] = { - "gemini-cli": ["antigravity"], -} - - -async def p_delete_provider_connections(providers: list[str]) -> int: - """Delete 9Router connections in `providers`; returns count removed, silent on 9Router unreachable.""" - import httpx - from backend.apps.nine_router import NINE_ROUTER_API, get_providers - try: - connections = await get_providers() - except Exception: - return 0 - targets = [c for c in connections if c.get("provider") in providers and c.get("id")] - removed = 0 - async with httpx.AsyncClient(timeout=10.0) as client: - for c in targets: - try: - await client.delete(f"{NINE_ROUTER_API}/providers/{c['id']}") - removed += 1 - except Exception: - pass - return removed - - @agents.router.post("/subscriptions/disconnect") async def subscriptions_disconnect(body: dict): - """Disconnect a subscription provider via 9Router; cascades-wipe Google's paired lanes.""" + """Disconnect a subscription provider's 9Router lane; reports ok only once the lane is verifiably gone.""" + from backend.apps.agents.disconnect_subscription import disconnect_subscription provider = body.get("provider", "") if not provider: raise HTTPException(status_code=400, detail="provider required") - - try: - to_remove = [provider, *P_PROVIDER_CASCADE_REMOVES.get(provider, [])] - removed = await p_delete_provider_connections(to_remove) - if removed: - from backend.apps.service.client import sync as p_sync - from backend.apps.settings.settings import load_settings - p_sync(load_settings().model_dump()) - return {"ok": True} - return {"ok": False, "error": "Connection not found"} - except Exception as e: - raise HTTPException(status_code=500, detail=str(e)) + result = await disconnect_subscription(provider) + return result.model_dump() diff --git a/backend/apps/agents/browser/browser_agent.py b/backend/apps/agents/browser/browser_agent.py index 6e5ad5eb..c63e5f02 100644 --- a/backend/apps/agents/browser/browser_agent.py +++ b/backend/apps/agents/browser/browser_agent.py @@ -53,6 +53,7 @@ from backend.apps.agents.browser.browser_loop import ( stagnation_exhausted, ) from backend.apps.agents.browser.browser_validator import adjudicate_stuck +from backend.apps.agents.browser.humanize_element_rows import humanize_element_rows from backend.apps.agents.browser.strip_lone_surrogates import strip_lone_surrogates # Single actions the model could have folded into one BrowserBatch turn; reads, waits, and the batch tools themselves don't count toward the streak. @@ -3598,8 +3599,13 @@ async def run_browser_agents( final = [] for r in results: - if isinstance(r, Exception): + # gather(return_exceptions=True) hands back a cancelled child as a bare BaseException too, and that is not a result dict either. + if not isinstance(r, dict): final.append({"summary": f"Error: {str(r)}", "action_log": [], "final_screenshot": None}) - else: - final.append(r) + continue + # Last stop before the sub-agent's own words reach a parent agent or, on the fast path, the user verbatim. + for p_key in ("summary", "error"): + if isinstance(r.get(p_key), str): + r[p_key] = humanize_element_rows(r[p_key]) + final.append(r) return final diff --git a/backend/apps/agents/browser/humanize_element_rows.py b/backend/apps/agents/browser/humanize_element_rows.py new file mode 100644 index 00000000..5c600f08 --- /dev/null +++ b/backend/apps/agents/browser/humanize_element_rows.py @@ -0,0 +1,69 @@ +"""Rewrite the browser subsystem's internal element-index rows into plain prose. + +BrowserListInteractives hands the sub-agent rows like `[3] )} + + {error && ( + + {error} + + )} ); }; diff --git a/frontend/src/app/pages/Settings/sections/subscription/SubscriptionCards.tsx b/frontend/src/app/pages/Settings/sections/subscription/SubscriptionCards.tsx index 8e5e1976..d5a6d76d 100644 --- a/frontend/src/app/pages/Settings/sections/subscription/SubscriptionCards.tsx +++ b/frontend/src/app/pages/Settings/sections/subscription/SubscriptionCards.tsx @@ -11,12 +11,24 @@ import { setSubscriptionStatus, markSubscriptionConnected, selectSubscriptionConnections, + type SubscriptionConnection, } from '@/shared/state/subscriptionsSlice'; import { API_BASE } from '@/shared/config'; import { SUBSCRIPTION_PROVIDERS } from './subscriptionProviders'; import SubscriptionCard from './SubscriptionCard'; import { runConnectFlow } from './subscriptionConnect'; +/** What POST /agents/subscriptions/disconnect answers; `ok` is the backend's verified end state, never a guess. */ +interface DisconnectResponse { + ok?: boolean; + removed?: number; + error?: string; +} + +function isProviderActive(connections: SubscriptionConnection[], providerId: string): boolean { + return connections.some((p) => p.provider === providerId && (p.isActive || p.testStatus === 'active')); +} + function friendlyConnectError(detail: string): string { const d = (detail || '').trim(); const lower = d.toLowerCase(); @@ -36,6 +48,8 @@ const SubscriptionCards: React.FC = () => { const connections = useAppSelector(selectSubscriptionConnections); const [connecting, setConnecting] = useState(null); const [disconnecting, setDisconnecting] = useState(null); + const [confirmingDisconnect, setConfirmingDisconnect] = useState(null); + const [disconnectError, setDisconnectError] = useState<{ provider: string; message: string } | null>(null); const [userCode, setUserCode] = useState(''); const [pollTimer, setPollTimer] = useState(null); const [connectError, setConnectError] = useState(null); @@ -70,15 +84,18 @@ const SubscriptionCards: React.FC = () => { return () => { cancelled = true; clearInterval(interval); }; }, [fetchStatus]); - const isConnected = (providerId: string) => - connections.some( - (p: any) => - p.provider === providerId && (p.isActive || p.testStatus === 'active'), - ); + const isConnected = (providerId: string) => isProviderActive(connections, providerId); + + // A pending confirm on a lane that died some other way (reconnect, cascade wipe) would pop back on reconnect. + useEffect(() => { + if (confirmingDisconnect && !isProviderActive(connections, confirmingDisconnect)) setConfirmingDisconnect(null); + }, [confirmingDisconnect, connections]); const handleConnect = async (providerId: string) => { if (pollTimer) { clearInterval(pollTimer); setPollTimer(null); } setConnectError(null); + setConfirmingDisconnect(null); + setDisconnectError(null); setConnecting(providerId); setUserCode(''); @@ -105,20 +122,27 @@ const SubscriptionCards: React.FC = () => { }; const handleDisconnect = async (providerId: string) => { + if (disconnecting) return; + setConfirmingDisconnect(null); + setDisconnectError(null); setDisconnecting(providerId); + // No settle delay: the backend only answers ok after re-reading 9Router, so the lane is already gone. try { - await fetch(`${API_BASE}/agents/subscriptions/disconnect`, { + const r = await fetch(`${API_BASE}/agents/subscriptions/disconnect`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ provider: providerId }), }); - } catch {} - // Wait briefly for 9Router to process, then refresh subscription status + model picker. - setTimeout(() => { - fetchStatus(); - refreshPickerModels(); - setDisconnecting(null); - }, 500); + const data = (await r.json().catch(() => ({}))) as DisconnectResponse; + if (!r.ok || !data.ok) { + setDisconnectError({ provider: providerId, message: data.error || 'Could not disconnect. Please try again.' }); + } + } catch { + setDisconnectError({ provider: providerId, message: 'Could not reach OpenSwarm. Please try again.' }); + } + await fetchStatus(); + refreshPickerModels(); + setDisconnecting(null); }; // 4s safety-net poller while connecting; clears Connecting state whenever 9Router reports the provider isActive (handles Windows postMessage failures). @@ -206,9 +230,13 @@ const SubscriptionCards: React.FC = () => { provider={p} connected={isConnected(p.id)} onConnect={() => handleConnect(p.id)} + onRequestDisconnect={() => { setDisconnectError(null); setConfirmingDisconnect(p.id); }} + onCancelDisconnect={() => setConfirmingDisconnect(null)} onDisconnect={() => handleDisconnect(p.id)} connecting={connecting === p.id} + confirmingDisconnect={confirmingDisconnect === p.id} disconnecting={disconnecting === p.id} + error={disconnectError?.provider === p.id ? disconnectError.message : undefined} userCode={connecting === p.id ? userCode : undefined} /> ))} diff --git a/frontend/src/app/pages/Skills/CommunitySkillsDialog.tsx b/frontend/src/app/pages/Skills/CommunitySkillsDialog.tsx index 22f6fc49..cb4398a5 100644 --- a/frontend/src/app/pages/Skills/CommunitySkillsDialog.tsx +++ b/frontend/src/app/pages/Skills/CommunitySkillsDialog.tsx @@ -12,6 +12,7 @@ import CircularProgress from '@mui/material/CircularProgress'; import Alert from '@mui/material/Alert'; import InputAdornment from '@mui/material/InputAdornment'; import SearchIcon from '@mui/icons-material/Search'; +import { EmptyState } from '@/app/components/feedback/Loading'; import OpenInNewIcon from '@mui/icons-material/OpenInNew'; import WarningAmberIcon from '@mui/icons-material/WarningAmber'; import { useClaudeTokens } from '@/shared/styles/ThemeContext'; @@ -132,9 +133,7 @@ const CommunitySkillsDialog: React.FC = ({ open, onClose, onInstalled }) /> {loading && } {!loading && results.length === 0 && ( - - {query.trim() ? 'No matching skills.' : 'Type to search the community registry.'} - + )} {!loading && results.map((s) => ( = ({ outputId, isAgentActive, saveLabel, onB ) : versions.length === 0 ? ( - - - - No history yet. Every time you change your app, we'll save a snapshot here so you can go back. - - + } + title="No history yet" + hint="Every time you change your app, we'll save a snapshot here so you can go back." + /> ) : ( <> diff --git a/frontend/src/app/pages/Workflows/SchedulePopover.tsx b/frontend/src/app/pages/Workflows/SchedulePopover.tsx index 48e8b6b4..5d83fc1a 100644 --- a/frontend/src/app/pages/Workflows/SchedulePopover.tsx +++ b/frontend/src/app/pages/Workflows/SchedulePopover.tsx @@ -31,6 +31,7 @@ interface Props { historyQuery: string; onHistoryQueryChange: (q: string) => void; onHistorySelect: (id: string) => void; + onHistoryContextMenu?: (e: React.MouseEvent, entry: { id: string; name: string }) => void; onNewChat: () => void; onWorkflowSelect: (id: string) => void; onExpand: () => void; @@ -50,7 +51,7 @@ interface Props { export default function SchedulePopover({ mode, onModeChange, historyResults, historyLoading, historyQuery, onHistoryQueryChange, - onHistorySelect, onNewChat, onWorkflowSelect, onExpand, + onHistorySelect, onHistoryContextMenu, onNewChat, onWorkflowSelect, onExpand, allRuns, allRunsLoading, onRunOpen, workflowTitleFor, historyScrollRef, onHistoryScroll, hideTopChrome = false, @@ -98,12 +99,9 @@ export default function SchedulePopover({ // Both Search and Schedule modes render at the same fixed dimensions so toggling chips doesn't resize the popover. Schedule sets the floor: its 7-day calendar needs ~620w x ~420h, search inherits the same. const POPOVER_W = 620; const CONTENT_H = 420; - return ( - {/* Floating mode chips. Hidden when the parent toolbar supplies its - own pill row (Image #32 / #54); kept around so the legacy callers - that surface Schedule mode still have a way in. */} + {/* Floating mode chips. Hidden when the parent toolbar supplies its own pill row; kept for legacy callers that surface Schedule mode. */} {!hideTopChrome && ( } active={mode === 'search'} onClick={() => onModeChange('search')} /> @@ -175,7 +173,7 @@ export default function SchedulePopover({ {bucket !== prevBucket && ( {bucket} )} - onHistorySelect(entry.id)} sx={{ display: 'flex', alignItems: 'center', gap: 1, px: 1, py: 0.8, cursor: 'pointer', borderRadius: `${c.radius.md}px`, '&:hover': { bgcolor: c.bg.elevated } }}> + onHistorySelect(entry.id)} onContextMenu={(e: React.MouseEvent) => onHistoryContextMenu?.(e, entry)} sx={{ display: 'flex', alignItems: 'center', gap: 1, px: 1, py: 0.8, cursor: 'pointer', borderRadius: `${c.radius.md}px`, '&:hover': { bgcolor: c.bg.elevated } }}> diff --git a/frontend/src/app/pages/Workflows/WorkflowNoticeToast.tsx b/frontend/src/app/pages/Workflows/WorkflowNoticeToast.tsx new file mode 100644 index 00000000..7c4ab5d1 --- /dev/null +++ b/frontend/src/app/pages/Workflows/WorkflowNoticeToast.tsx @@ -0,0 +1,38 @@ +// The server's own words when it refuses something the user asked for. Today that is a delete the +// cloud would not let go of; without it the card simply stays put and the click looks broken. + +import React from 'react'; +import Snackbar from '@mui/material/Snackbar'; +import Alert from '@mui/material/Alert'; +import { useClaudeTokens } from '@/shared/styles/ThemeContext'; +import { useAppDispatch, useAppSelector } from '@/shared/hooks'; +import { dismissNoticeToast } from '@/shared/state/workflowsSlice'; + +export default function WorkflowNoticeToast() { + const c = useClaudeTokens(); + const dispatch = useAppDispatch(); + const notice = useAppSelector((s) => s.workflows.noticeToast); + + return ( + { if (reason !== 'clickaway') dispatch(dismissNoticeToast()); }} + anchorOrigin={{ vertical: 'bottom', horizontal: 'left' }} + > + dispatch(dismissNoticeToast())} + sx={{ + bgcolor: c.bg.surface, + color: c.text.primary, + border: `1px solid ${c.border.medium}`, + maxWidth: 420, + }} + > + {notice} + + + ); +} diff --git a/frontend/src/app/pages/Workflows/app/CloudRunSection.tsx b/frontend/src/app/pages/Workflows/app/CloudRunSection.tsx new file mode 100644 index 00000000..13736877 --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/CloudRunSection.tsx @@ -0,0 +1,174 @@ +import React from 'react'; +import type { CSSProperties } from 'react'; +import { useAppDispatch } from '@/shared/hooks'; +import { openSettingsCard } from '@/shared/state/dashboardLayoutSlice'; +import type { Workflow } from '@/shared/state/workflowsSlice'; +import { useWC } from './uiKit'; +import type { WCPalette } from './uiKit'; +import { clockOf, relativeDayLabel } from './model'; +import { cloudAvailability, usageText } from './cloudAvailability'; +import type { CloudProbe } from './cloudAvailability'; +import type { HostedState } from './cloudApi'; +import type { CloudStatusHandle } from './useCloudStatus'; + +const CLOUD_PATH = 'M17.5 19a4.5 4.5 0 0 0 .5-8.97A6 6 0 0 0 6.2 10.5 4 4 0 0 0 6.5 19z'; + +function hostedOf(probe: CloudProbe) { + if (probe.phase !== 'answered' || probe.status.state !== 'ready') return null; + return probe.status.hosted; +} + +// The cloud's own clock, printed from the answer we just fetched; our mirrored copy only moves on a probe. +function nextCloudRunText(hosted: HostedState): string { + if (!hosted.enabled) return 'Paused in the cloud'; + if (!hosted.next_run_at) return 'No cloud run scheduled'; + const at = new Date(hosted.next_run_at); + return `Next cloud run ${relativeDayLabel(at)} at ${clockOf(at)}`; +} + +const Bullet: React.FC<{ wc: WCPalette; accent?: boolean; children: React.ReactNode }> = ({ wc, accent, children }) => ( +
+
+
+
+ {children} +
+); + +const Note: React.FC<{ wc: WCPalette; tone: 'quiet' | 'warn'; children: React.ReactNode }> = ({ wc, tone, children }) => ( +
+ {children} +
+); + +const CloudRunSection: React.FC<{ workflow: Workflow; cloud: CloudStatusHandle }> = ({ workflow, cloud }) => { + const WC = useWC(); + const dispatch = useAppDispatch(); + const availability = cloudAvailability(cloud.probe); + const hosted = hostedOf(cloud.probe); + const target = cloud.probe.phase === 'answered' ? cloud.probe.status.target : workflow.execution_target ?? 'device'; + const onCloud = target === 'cloud'; + const usage = usageText(cloud.probe, availability); + const canPickCloud = availability.kind === 'available' && !cloud.pending; + + const seg = (active: boolean, enabled: boolean): CSSProperties => ({ + flex: 1, padding: '6px 2px', borderRadius: 7, border: 'none', fontSize: 11.5, fontWeight: 600, + cursor: enabled ? 'pointer' : 'default', + background: active ? WC.paper : 'transparent', + color: active ? WC.ink : enabled ? WC.muted : WC.faint, + boxShadow: active ? WC.shadow.sm : 'none', + }); + + const link: CSSProperties = { + background: 'none', border: 'none', padding: 0, marginLeft: 4, cursor: 'pointer', + color: WC.accent, fontSize: 11.5, fontWeight: 600, textDecoration: 'underline', + }; + + return ( +
+
+ + + Runs on + +
+ + +
+
+ + {cloud.pending && Talking to the cloud…} + + {/* A refusal used to swallow every other branch, including the one holding the upgrade link, so being told you need Pro removed the way to get Pro. Carry the action through. */} + {!cloud.pending && cloud.refusal && ( + + + {cloud.refusal} + {availability.kind === 'blocked' && availability.action === 'sign_in' && ( + + )} + {availability.kind === 'blocked' && availability.action === 'plans' && ( + + )} + + + + )} + + {!cloud.pending && !cloud.refusal && availability.kind === 'checking' && ( + Checking what your account allows… + )} + + {!cloud.pending && !cloud.refusal && availability.kind === 'unknown' && ( + + + Can't reach the cloud, so we can't tell whether this can run there. + {onCloud + ? ' It stays scheduled in the cloud; nothing changed.' + : ' This workflow still runs on this device.'} + + + + )} + + {!cloud.pending && !cloud.refusal && availability.kind === 'blocked' && ( + + + {availability.reason} + {availability.action === 'sign_in' && ( + + )} + {availability.action === 'plans' && ( + + )} + + + )} + + {!cloud.pending && !cloud.refusal && availability.kind === 'available' && !onCloud && ( + Cloud runs fire on our servers, so they still happen with this app closed. + )} + + {!cloud.pending && onCloud && hosted === null && cloud.probe.phase === 'answered' && cloud.probe.status.state === 'ready' && ( + + + This is set to run in the cloud, but the cloud has no copy of it, so nothing is running it. + + + + )} + + {!cloud.pending && onCloud && hosted && !hosted.in_sync && ( + + + The cloud is still running the version you sent it. Your later edits are not up there yet. + + + + )} + + {onCloud && hosted && {nextCloudRunText(hosted)}} + {usage && {usage}} +
+ ); +}; + +export default CloudRunSection; diff --git a/frontend/src/app/pages/Workflows/app/HistoryCard.tsx b/frontend/src/app/pages/Workflows/app/HistoryCard.tsx index aafaafc3..5cc60bfd 100644 --- a/frontend/src/app/pages/Workflows/app/HistoryCard.tsx +++ b/frontend/src/app/pages/Workflows/app/HistoryCard.tsx @@ -3,36 +3,88 @@ import { useAppDispatch, useAppSelector } from '@/shared/hooks'; import { fetchRuns } from '@/shared/state/workflowsSlice'; import { openWorkflowMonitor } from '@/shared/state/dashboardLayoutSlice'; import { useWC, FONT_SERIF, statusChip, statusDot, statusLabel } from './uiKit'; +import type { WCPalette } from './uiKit'; import { toRunRow, whenText } from './model'; +import { useCloudRuns } from './useCloudRuns'; +import { toCloudHistoryRow } from './cloudRunRow'; +import type { CloudHistoryRow } from './cloudRunRow'; + +interface Entry extends CloudHistoryRow { + where: 'device' | 'cloud'; + open?: () => void; +} + +const Row: React.FC<{ entry: Entry; wc: WCPalette; now: Date }> = ({ entry, wc, now }) => ( +
+
+
+ {/* Why a run did not happen is a sentence, not a label, so let it wrap rather than ellipsing the part that answers the question. */} +
{entry.summary}
+
+ {[entry.where === 'cloud' ? 'Cloud' : null, whenText(entry.when, now), entry.durationText, entry.costText] + .filter(Boolean) + .join(' · ')} +
+
+ {entry.label} +
+); const HistoryCard: React.FC<{ workflowId: string; title: string }> = ({ workflowId, title }) => { const WC = useWC(); const dispatch = useAppDispatch(); const runs = useAppSelector((s) => s.workflows.runs[workflowId]); + const workflow = useAppSelector((s) => s.workflows.items[workflowId]); + const onCloud = workflow?.execution_target === 'cloud'; + const cloudRuns = useCloudRuns(workflowId, onCloud, workflow?.updated_at ?? ''); useEffect(() => { dispatch(fetchRuns(workflowId)); }, [workflowId, dispatch]); - const rows = (runs || []).slice(0, 8).map((r) => toRunRow(r, title)); + const local: Entry[] = (runs || []).map((r) => { + const row = toRunRow(r, title); + return { + id: row.id, + label: statusLabel(row.status), + tone: row.status, + summary: row.summary, + when: row.when, + durationText: row.durationText, + costText: '', + where: 'device', + open: () => dispatch(openWorkflowMonitor({ workflowId, runId: row.id })), + }; + }); + const remote: Entry[] = + cloudRuns.phase === 'answered' && cloudRuns.response.state === 'ready' + ? cloudRuns.response.runs.map((r) => ({ ...toCloudHistoryRow(r, title), where: 'cloud' as const })) + : []; + const rows = [...local, ...remote] + .sort((a, b) => (b.when?.getTime() ?? 0) - (a.when?.getTime() ?? 0)) + .slice(0, 8); + + // A history we could not load is not an empty history, and must never be drawn as one. + const cloudBlind = + onCloud && (cloudRuns.phase === 'checking' || (cloudRuns.phase === 'answered' && cloudRuns.response.state !== 'ready')); const now = new Date(); return (
History
- {rows.length === 0 &&
No runs yet.
} + {rows.length === 0 && !cloudBlind &&
No runs yet.
}
- {rows.map((r) => ( -
dispatch(openWorkflowMonitor({ workflowId, runId: r.id }))} title="Open this run" style={{ display: 'flex', alignItems: 'center', gap: 11, padding: '9px 0', borderBottom: `1px solid rgba(${WC.inkRGB},0.05)`, cursor: 'pointer' }}> -
-
-
{r.summary}
-
- {whenText(r.when, now)}{r.durationText ? ` · ${r.durationText}` : ''} -
-
- {statusLabel(r.status)} -
- ))} + {rows.map((entry) => )}
+ {cloudBlind && ( +
+ {cloudRuns.phase === 'checking' + ? 'Loading cloud runs…' + : 'Couldn’t load this workflow’s cloud runs, so any that ran are not shown here.'} +
+ )}
); }; diff --git a/frontend/src/app/pages/Workflows/app/LeftRail.tsx b/frontend/src/app/pages/Workflows/app/LeftRail.tsx index db1a07c5..d78f0536 100644 --- a/frontend/src/app/pages/Workflows/app/LeftRail.tsx +++ b/frontend/src/app/pages/Workflows/app/LeftRail.tsx @@ -31,8 +31,6 @@ const LeftRail: React.FC<{ nav: AppNav }> = ({ nav }) => { return workflows.filter((w) => w.title.toLowerCase().includes(q)); }, [workflows, query]); - const activeCount = workflows.filter((w) => isScheduleActive(w.schedule)).length; - const onDelete = (id: string) => { // Soft-delete: moves to Trash (recoverable), so no scary confirm. dispatch(deleteWorkflow(id)); @@ -84,7 +82,8 @@ const LeftRail: React.FC<{ nav: AppNav }> = ({ nav }) => {
Workflows - {activeCount} + {/* Counts what the list below is actually showing. It used to count only the scheduled ones, so a real workflow sat under a header that said 0. */} + {filtered.length}
diff --git a/frontend/src/app/pages/Workflows/app/RunMonitor.tsx b/frontend/src/app/pages/Workflows/app/RunMonitor.tsx index 1e4dd702..3eb79ae7 100644 --- a/frontend/src/app/pages/Workflows/app/RunMonitor.tsx +++ b/frontend/src/app/pages/Workflows/app/RunMonitor.tsx @@ -11,6 +11,7 @@ import { } from '@/shared/state/dashboardLayoutSlice'; import type { CardType } from '@/shared/state/dashboardLayoutSlice'; import WorkflowTitle from './WorkflowTitle'; +import { isScheduleActive } from '@/app/pages/Workflows/scheduleUtils'; const DRAG_THRESHOLD = 3; @@ -231,7 +232,10 @@ const RunMonitor: React.FC = ({ workflow, cardX, cardY, cardWidth, cardHe
Schedule
-
{formatSchedule(workflow.schedule)}
+ {/* A paused workflow fell through formatSchedule to "Scheduled for 9:00 AM", so this card announced a schedule while the panel beside it said Not scheduled. */} +
+ {isScheduleActive(workflow.schedule) ? formatSchedule(workflow.schedule) : 'Not scheduled'} +
{workflow.next_run_at && (
Next run {new Date(workflow.next_run_at).toLocaleString([], { weekday: 'short', month: 'short', day: 'numeric', hour: 'numeric', minute: '2-digit' })} diff --git a/frontend/src/app/pages/Workflows/app/ScheduleCard.tsx b/frontend/src/app/pages/Workflows/app/ScheduleCard.tsx index a9ab59a5..94c20975 100644 --- a/frontend/src/app/pages/Workflows/app/ScheduleCard.tsx +++ b/frontend/src/app/pages/Workflows/app/ScheduleCard.tsx @@ -7,6 +7,8 @@ import { freqOf, patchForFreq, intervalMinutes, timeInputValue, parseTimeInput, ordinal, nextRunText, type Freq, } from './model'; import { useWorkflowPatch } from './useWorkflowPatch'; +import { useCloudStatus } from './useCloudStatus'; +import CloudRunSection from './CloudRunSection'; import RepeatField from './RepeatField'; const FREQS: Array<[Freq, string]> = [['daily', 'Daily'], ['weekly', 'Weekly'], ['monthly', 'Monthly'], ['interval', 'Interval']]; @@ -15,9 +17,11 @@ const DAY_LABELS: Array<[string, number]> = [['S', 0], ['M', 1], ['T', 2], ['W', const ScheduleCard: React.FC<{ workflow: Workflow }> = ({ workflow }) => { const WC = useWC(); const patch = useWorkflowPatch(); + const cloud = useCloudStatus(workflow); const sched = workflow.schedule; const freq = freqOf(sched); const enabled = sched.enabled; + const onCloud = workflow.execution_target === 'cloud'; const patchSched = (p: Partial) => patch(workflow, { schedule: { ...sched, ...p } }); @@ -32,6 +36,11 @@ const ScheduleCard: React.FC<{ workflow: Workflow }> = ({ workflow }) => { // Turning a weekly schedule on with no days picked is "unconfigured", so the backend silently forces it back off and the switch looks dead. Seed today's weekday so the default Weekly 9am toggles on (and stays on) in one click. const toggleEnabled = () => { + // While the cloud holds the timer, this switch is the CLOUD's switch: flipping only our copy would pause nothing. + if (onCloud) { + cloud.choose('cloud', !enabled); + return; + } if (!enabled && sched.repeat_unit === 'week' && sched.on_days.length === 0) { patchSched({ enabled: true, on_days: [new Date().getDay()] }); } else { @@ -197,16 +206,21 @@ const ScheduleCard: React.FC<{ workflow: Workflow }> = ({ workflow }) => { {describeSchedule(sched)}
-
-
- Next run {nextRunText(workflow, workflow.next_run_at ? new Date(workflow.next_run_at) : null)} -
+ {/* On cloud our copy of next_run_at is a mirror that only refreshes on a probe, so the cloud section prints the time it just fetched instead. */} + {!onCloud && ( +
+
+ Next run {nextRunText(workflow, workflow.next_run_at ? new Date(workflow.next_run_at) : null)} +
+ )} {maxRuns != null && (
{sched.runs_count} of {maxRuns} run{maxRuns === 1 ? '' : 's'} done
)} + +
); }; diff --git a/frontend/src/app/pages/Workflows/app/WorkflowsAppCard.tsx b/frontend/src/app/pages/Workflows/app/WorkflowsAppCard.tsx index 5c5e5443..1c37139c 100644 --- a/frontend/src/app/pages/Workflows/app/WorkflowsAppCard.tsx +++ b/frontend/src/app/pages/Workflows/app/WorkflowsAppCard.tsx @@ -1,6 +1,6 @@ import React, { useCallback, useEffect } from 'react'; import { useAppDispatch, useAppSelector } from '@/shared/hooks'; -import { setWorkflowsHubPosition, setWorkflowsHubSize } from '@/shared/state/dashboardLayoutSlice'; +import { setWorkflowsHubPosition, setWorkflowsHubSize, WORKFLOWS_HUB_ID } from '@/shared/state/dashboardLayoutSlice'; import CanvasWindowCard from '@/app/pages/Dashboard/cards/CanvasWindowCard'; import type { CardType } from '@/shared/state/dashboardLayoutSlice'; import { useWC } from './uiKit'; @@ -34,7 +34,7 @@ const WorkflowsAppCard: React.FC = ({ }) => { const WC = useWC(); const dispatch = useAppDispatch(); - const isFullscreen = useAppSelector((s) => !!s.dashboardLayout.workflowsHub?.fullscreen); + const isMinimized = useAppSelector((s) => !!s.dashboardLayout.minimizedCards[WORKFLOWS_HUB_ID]); // Keep fonts/keyframes available while the card is mounted. useEffect(() => { ensureAssets(); }, []); @@ -48,7 +48,7 @@ const WorkflowsAppCard: React.FC = ({ return ( = ({ cardWidth={cardWidth} cardHeight={cardHeight} cardZOrder={cardZOrder} - fullscreen={isFullscreen} + minimized={isMinimized} minWidth={MIN_W} minHeight={MIN_H} background={WC.page} @@ -74,8 +74,8 @@ const WorkflowsAppCard: React.FC = ({ onCommitPosition={commitPosition} onCommitSize={commitSize} > - {({ header, onTileZone }) => ( - + {({ header, tileZone, onTileZone }) => ( + )} ); diff --git a/frontend/src/app/pages/Workflows/app/WorkflowsAppContent.tsx b/frontend/src/app/pages/Workflows/app/WorkflowsAppContent.tsx index 63f1d91c..84412c1d 100644 --- a/frontend/src/app/pages/Workflows/app/WorkflowsAppContent.tsx +++ b/frontend/src/app/pages/Workflows/app/WorkflowsAppContent.tsx @@ -1,7 +1,7 @@ import React, { useEffect, useMemo, useState } from 'react'; import EventRepeatIcon from '@mui/icons-material/EventRepeat'; import { useAppDispatch, useAppSelector } from '@/shared/hooks'; -import { clearWorkflowsAppTarget, closeWorkflowsApp, toggleWorkflowsHubFullscreen } from '@/shared/state/dashboardLayoutSlice'; +import { clearWorkflowsAppTarget, closeWorkflowsApp, toggleMinimizeCard, WORKFLOWS_HUB_ID } from '@/shared/state/dashboardLayoutSlice'; import WindowControls from '@/app/pages/Dashboard/cards/WindowControls'; import { fetchWorkflows, fetchAllRuns, fetchPausedState, fetchActiveRuns, fetchDeletedWorkflows, @@ -18,11 +18,10 @@ import ComposeView from './ComposeView'; import TrashView from './TrashView'; // The three-pane Workflows body plus its title bar. The card wraps this with drag/resize geometry and passes the drag handlers in; the title bar lives here because Share needs to know which workflow is open. -const WorkflowsAppContent: React.FC<{ header: CardHeader; onTileZone?: (zone: string) => void }> = ({ header, onTileZone }) => { +const WorkflowsAppContent: React.FC<{ header: CardHeader; tileZone: string | undefined; onTileZone: (zone: string) => void }> = ({ header, tileZone, onTileZone }) => { const WC = useWC(); const dispatch = useAppDispatch(); const target = useAppSelector((s) => s.dashboardLayout.workflowsAppTarget); - const isFullscreen = useAppSelector((s) => !!s.dashboardLayout.workflowsHub?.fullscreen); const dashboardId = useAppSelector((s) => s.tempState.lastDashboardId) || undefined; const [mode, setMode] = useState('home'); @@ -84,14 +83,9 @@ const WorkflowsAppContent: React.FC<{ header: CardHeader; onTileZone?: (zone: st > dispatch(closeWorkflowsApp())} - onMinimize={() => dispatch(closeWorkflowsApp())} - onTile={(zone) => { - if (zone === 'fullscreen' || zone === 'restore') { dispatch(toggleWorkflowsHubFullscreen()); return; } - if (isFullscreen) dispatch(toggleWorkflowsHubFullscreen()); - onTileZone?.(zone); - }} - tiled={isFullscreen} - noTileMenu={isFullscreen} + onMinimize={() => dispatch(toggleMinimizeCard({ cardId: WORKFLOWS_HUB_ID }))} + onTile={onTileZone} + tiled={!!tileZone} />
diff --git a/frontend/src/app/pages/Workflows/app/cloudApi.ts b/frontend/src/app/pages/Workflows/app/cloudApi.ts new file mode 100644 index 00000000..8589f904 --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/cloudApi.ts @@ -0,0 +1,123 @@ +import { API_BASE, getAuthToken } from '@/shared/config'; + +// Mirrors backend/apps/workflows/cloud/status.py. `unknown` is a first-class answer, not a +// degraded `ready`: it carries no plan, no limits and no counts, so a failed fetch cannot be +// rendered as "you are not entitled" or "0 runs left". +export type CloudTarget = 'device' | 'cloud'; + +export interface CloudLimits { + workflows: number; + runs_per_month: number; + concurrent: number; +} + +export interface CloudUsage { + workflows_enabled: number; + runs_this_month: number; +} + +export interface CloudCapability { + ok: boolean; + reason: string | null; +} + +export interface HostedState { + id: string; + enabled: boolean; + next_run_at: string | null; + /** False when the workflow was edited after we pushed it, so the cloud holds older prose. */ + in_sync: boolean; +} + +interface CloudStatusShared { + target: CloudTarget; + schedule_supported: boolean; + schedule_reason: string | null; +} + +export interface CloudStatusReady extends CloudStatusShared { + state: 'ready'; + plan: string | null; + limits: CloudLimits; + usage: CloudUsage; + /** Null when the control plane could not tell us; create re-checks either way. */ + capability: CloudCapability | null; + hosted: HostedState | null; +} + +export interface CloudStatusSignedOut extends CloudStatusShared { + state: 'signed_out'; +} + +export interface CloudStatusUnknown extends CloudStatusShared { + state: 'unknown'; + detail: string; +} + +export type CloudStatus = CloudStatusReady | CloudStatusSignedOut | CloudStatusUnknown; + +export interface CloudRun { + id: string; + status: string; + started_at: string | null; + finished_at: string | null; + error: string | null; + answer: string | null; + notices: string[]; + cost_usd: number | null; +} + +export type CloudRunsResponse = + | { state: 'ready'; runs: CloudRun[] } + | { state: 'signed_out' | 'unknown'; detail: string | null }; + +export interface TargetOutcome { + ok: boolean; + message: string | null; +} + +const base = `${API_BASE}/cloud_workflows`; + +function headers(): Record { + let tok = ''; + try { tok = getAuthToken(); } catch { tok = ''; } + return { 'Content-Type': 'application/json', ...(tok ? { Authorization: `Bearer ${tok}` } : {}) }; +} + +// Null means our own backend did not answer, which the caller must render as "cannot tell" rather than as a denial. +export async function fetchCloudStatus(workflowId: string): Promise { + try { + const res = await fetch(`${base}/${encodeURIComponent(workflowId)}/status`, { headers: headers() }); + if (!res.ok) return null; + return (await res.json()) as CloudStatus; + } catch { + return null; + } +} + +export async function fetchCloudRuns(workflowId: string): Promise { + try { + const res = await fetch(`${base}/${encodeURIComponent(workflowId)}/runs`, { headers: headers() }); + if (!res.ok) return { state: 'unknown', detail: null }; + return (await res.json()) as CloudRunsResponse; + } catch { + return { state: 'unknown', detail: null }; + } +} + +export async function setCloudTarget( + workflowId: string, + body: { target: CloudTarget; enabled: boolean }, +): Promise { + try { + const res = await fetch(`${base}/${encodeURIComponent(workflowId)}/target`, { + method: 'POST', + headers: headers(), + body: JSON.stringify(body), + }); + if (!res.ok) return { ok: false, message: 'Something went wrong on this machine, so nothing changed.' }; + return (await res.json()) as TargetOutcome; + } catch { + return { ok: false, message: 'Something went wrong on this machine, so nothing changed.' }; + } +} diff --git a/frontend/src/app/pages/Workflows/app/cloudAvailability.ts b/frontend/src/app/pages/Workflows/app/cloudAvailability.ts new file mode 100644 index 00000000..ec08107f --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/cloudAvailability.ts @@ -0,0 +1,68 @@ +import type { CloudStatus, CloudStatusReady } from './cloudApi'; + +// One probe, three honest outcomes plus "still asking". Anything we have not heard back about is +// `unknown`, never a refusal: a hiccup that renders as "not entitled" is a paywall built out of a +// dropped packet. +export type CloudProbe = + | { phase: 'checking' } + | { phase: 'unreachable' } + | { phase: 'answered'; status: CloudStatus }; + +export type CloudAvailability = + | { kind: 'checking' } + | { kind: 'unknown'; detail: string | null } + | { kind: 'blocked'; reason: string; action: 'sign_in' | 'plans' | null } + | { kind: 'available' }; + +const PLAN_REQUIRED = 'Cloud runs come with Pro and up. On this plan, workflows run on this device.'; + +function blockedForAccount(status: CloudStatusReady): CloudAvailability | null { + if (status.limits.workflows === 0) { + return { kind: 'blocked', reason: PLAN_REQUIRED, action: 'plans' }; + } + // A workflow already up there is holding one of the slots, so its own slot must not read as full. + const holdsASlot = status.hosted !== null; + if (!holdsASlot && status.usage.workflows_enabled >= status.limits.workflows) { + return { + kind: 'blocked', + reason: `${status.usage.workflows_enabled} of ${status.limits.workflows} cloud workflows used. Turn one off to move this one up.`, + action: null, + }; + } + return null; +} + +/** Whether the Cloud choice can be offered, and if not, the sentence that says why. + * Reasons about the workflow itself come first: telling someone to upgrade for a job the runner + * could never do is a sale, not an answer. */ +export function cloudAvailability(probe: CloudProbe): CloudAvailability { + if (probe.phase === 'checking') return { kind: 'checking' }; + if (probe.phase === 'unreachable') return { kind: 'unknown', detail: null }; + const status = probe.status; + if (!status.schedule_supported && status.schedule_reason) { + return { kind: 'blocked', reason: status.schedule_reason, action: null }; + } + if (status.state === 'unknown') return { kind: 'unknown', detail: status.detail }; + if (status.state === 'signed_out') { + return { + kind: 'blocked', + reason: 'Sign in to your OpenSwarm account to run workflows in the cloud.', + action: 'sign_in', + }; + } + if (status.capability && !status.capability.ok && status.capability.reason) { + return { kind: 'blocked', reason: status.capability.reason, action: null }; + } + return blockedForAccount(status) ?? { kind: 'available' }; +} + +/** The account-wide ceiling, said so plainly nobody reads it as this one workflow's count. + * Null whenever we are unsure of the numbers, or cloud is not on the table for this workflow. */ +export function usageText(probe: CloudProbe, availability: CloudAvailability): string | null { + if (probe.phase !== 'answered' || probe.status.state !== 'ready') return null; + const onCloud = probe.status.target === 'cloud'; + if (!onCloud && availability.kind !== 'available') return null; + const { usage, limits } = probe.status; + if (limits.runs_per_month === 0) return null; + return `Your plan: ${usage.runs_this_month} of ${limits.runs_per_month} cloud runs this month`; +} diff --git a/frontend/src/app/pages/Workflows/app/cloudRunRow.ts b/frontend/src/app/pages/Workflows/app/cloudRunRow.ts new file mode 100644 index 00000000..c1e05559 --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/cloudRunRow.ts @@ -0,0 +1,86 @@ +import type { CloudRun } from './cloudApi'; +import type { RunStatus } from './uiKit'; + +// A cloud run that never started reports as "dispatch_unavailable: fly_capacity: ...", which is a +// sentence for us, not for the person who was expecting a report at 9am. Every refusal the +// dispatcher can produce gets a plain answer to the only question they have: did it run, and why not. +const REFUSAL_TEXT: Record = { + runner_not_configured: "Cloud runs weren't available on our side, so this didn't start. You weren't charged for it.", + callback_not_configured: "Cloud runs weren't available on our side, so this didn't start. You weren't charged for it.", + fly_unauthorized: "Cloud runs weren't available on our side, so this didn't start. You weren't charged for it.", + fly_rejected: "Cloud runs weren't available on our side, so this didn't start. You weren't charged for it.", + fly_capacity: "The cloud had no room at that moment, so this didn't start. You weren't charged for it.", + fly_unreachable: "We couldn't reach the machine meant to run this, so it didn't start. You weren't charged for it.", + workflow_definition_invalid: + "The cloud's copy of this workflow was unreadable, so nothing ran. Switch it back to this device and up to the cloud again to resend it.", + no_cloud_credential: + "No AI account is connected to the cloud for this workspace, so there was nothing to run this with.", + slot_already_run: 'This slot had already run, so it was not run a second time.', +}; + +const STATUS_LABEL: Record = { + pending: 'Starting', + running: 'Running', + succeeded: 'Success', + failed: 'Failed', + dispatch_unavailable: "Didn't run", +}; + +const STATUS_TONE: Record = { + pending: 'running', + running: 'running', + succeeded: 'success', + failed: 'failure', + dispatch_unavailable: 'skipped', +}; + +export interface CloudHistoryRow { + id: string; + label: string; + tone: RunStatus; + summary: string; + when: Date | null; + durationText: string; + costText: string; +} + +/** Split "reason: detail" on the FIRST colon only, and only accept a reason we actually know. + * An unrecognised prefix falls through to the raw text: showing the truth beats guessing at it. */ +export function explainCloudFailure(error: string | null): string { + if (!error) return ''; + const at = error.indexOf(': '); + if (at < 0) return error; + const reason = error.slice(0, at); + const detail = error.slice(at + 2); + // The runner-capability detail is already the sentence we would have written. + if (reason === 'runner_capability') return detail; + return REFUSAL_TEXT[reason] ?? error; +} + +function duration(run: CloudRun): string { + if (!run.started_at || !run.finished_at) return ''; + const ms = new Date(run.finished_at).getTime() - new Date(run.started_at).getTime(); + if (Number.isNaN(ms) || ms < 0) return ''; + const s = Math.round(ms / 1000); + return s < 60 ? `${s}s` : `${Math.floor(s / 60)}m ${s % 60}s`; +} + +function cost(run: CloudRun): string { + if (run.cost_usd === null || run.cost_usd === undefined) return ''; + if (run.cost_usd === 0) return '$0.00'; + return run.cost_usd < 0.01 ? '<$0.01' : `$${run.cost_usd.toFixed(2)}`; +} + +export function toCloudHistoryRow(run: CloudRun, fallbackTitle: string): CloudHistoryRow { + const failed = run.status === 'failed' || run.status === 'dispatch_unavailable'; + const explained = explainCloudFailure(run.error); + return { + id: run.id, + label: STATUS_LABEL[run.status] ?? run.status, + tone: STATUS_TONE[run.status] ?? 'skipped', + summary: (failed && explained) || run.answer || fallbackTitle, + when: run.started_at ? new Date(run.started_at) : null, + durationText: duration(run), + costText: cost(run), + }; +} diff --git a/frontend/src/app/pages/Workflows/app/useCloudRuns.ts b/frontend/src/app/pages/Workflows/app/useCloudRuns.ts new file mode 100644 index 00000000..6f58672c --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/useCloudRuns.ts @@ -0,0 +1,49 @@ +import { useEffect, useRef, useState } from 'react'; +import { fetchCloudRuns } from './cloudApi'; +import type { CloudRunsResponse } from './cloudApi'; + +export type CloudRunsProbe = + | { phase: 'idle' } + | { phase: 'checking' } + | { phase: 'answered'; response: CloudRunsResponse }; + +/** Run history for the cloud copy. Only asks when the workflow is actually up there, so a + * device-only workflow never makes a network call to find out it has no cloud runs. */ +export function useCloudRuns(workflowId: string, hosted: boolean, revision: string): CloudRunsProbe { + const [probe, setProbe] = useState({ phase: 'idle' }); + const live = useRef(true); + + useEffect(() => { + live.current = true; + return () => { live.current = false; }; + }, []); + + useEffect(() => { + if (!hosted) { + setProbe({ phase: 'idle' }); + return; + } + setProbe((prev) => (prev.phase === 'answered' ? prev : { phase: 'checking' })); + fetchCloudRuns(workflowId).then((response) => { + if (live.current) setProbe({ phase: 'answered', response }); + }); + }, [workflowId, hosted, revision]); + + // A cloud run reports to the cloud, not to us, so a run in flight is the one case worth asking again about. The poll stops itself the moment nothing is live. + const watching = + probe.phase === 'answered' && + probe.response.state === 'ready' && + probe.response.runs.some((r) => r.status === 'pending' || r.status === 'running'); + + useEffect(() => { + if (!hosted || !watching) return undefined; + const timer = setInterval(() => { + fetchCloudRuns(workflowId).then((response) => { + if (live.current) setProbe({ phase: 'answered', response }); + }); + }, 30000); + return () => clearInterval(timer); + }, [hosted, watching, workflowId]); + + return probe; +} diff --git a/frontend/src/app/pages/Workflows/app/useCloudStatus.ts b/frontend/src/app/pages/Workflows/app/useCloudStatus.ts new file mode 100644 index 00000000..d11b8314 --- /dev/null +++ b/frontend/src/app/pages/Workflows/app/useCloudStatus.ts @@ -0,0 +1,69 @@ +import { useCallback, useEffect, useRef, useState } from 'react'; +import { useAppDispatch } from '@/shared/hooks'; +import { fetchWorkflows } from '@/shared/state/workflowsSlice'; +import type { Workflow } from '@/shared/state/workflowsSlice'; +import { fetchCloudStatus, setCloudTarget } from './cloudApi'; +import type { CloudTarget, TargetOutcome } from './cloudApi'; +import type { CloudProbe } from './cloudAvailability'; + +export interface CloudStatusHandle { + probe: CloudProbe; + /** True while a flip is in flight; the control must not accept a second one. */ + pending: boolean; + /** Set only by a refused flip, and cleared by the next attempt. */ + refusal: string | null; + retry: () => void; + choose: (target: CloudTarget, enabled: boolean) => void; + refresh: () => void; +} + +/** The cloud's answer for one workflow, fetched off the render path so the Workflows app paints + * and stays usable whether or not there is a cloud, an account, or a network. */ +export function useCloudStatus(workflow: Workflow): CloudStatusHandle { + const dispatch = useAppDispatch(); + const [probe, setProbe] = useState({ phase: 'checking' }); + const [pending, setPending] = useState(false); + const [refusal, setRefusal] = useState(null); + const live = useRef(true); + const workflowId = workflow.id; + const dashboardId = workflow.dashboard_id; + + useEffect(() => { + live.current = true; + return () => { live.current = false; }; + }, []); + + const refresh = useCallback(() => { + fetchCloudStatus(workflowId).then((status) => { + if (!live.current) return; + setProbe(status ? { phase: 'answered', status } : { phase: 'unreachable' }); + }); + }, [workflowId]); + + // Only a different workflow blanks the answer. Re-asking about the SAME one keeps the last answer on screen while it happens, so an edit does not strobe the card. + useEffect(() => { setProbe({ phase: 'checking' }); }, [workflowId]); + + // Re-ask when the workflow changes in a way the answer depends on: a different schedule can stop being expressible in the cloud, and edited steps can stop being runnable there. + useEffect(() => { refresh(); }, [refresh, workflow.updated_at]); + + const choose = useCallback((target: CloudTarget, enabled: boolean) => { + if (pending) return; + setPending(true); + setRefusal(null); + setCloudTarget(workflowId, { target, enabled }).then((outcome: TargetOutcome) => { + if (!live.current) return; + setPending(false); + if (!outcome.ok) setRefusal(outcome.message); + // Refresh either way: a refusal usually means the reasons on screen are stale too. + refresh(); + if (outcome.ok) dispatch(fetchWorkflows(dashboardId ?? undefined)); + }); + }, [dispatch, pending, refresh, workflowId, dashboardId]); + + // refresh() deliberately leaves `refusal` alone: choose() calls it right after setting one, and + // clearing there would wipe the message before it rendered. A user-driven retry is a different + // intent, so it gets its own door; without it a refusal outlived the thing that caused it. + const retry = useCallback(() => { setRefusal(null); refresh(); }, [refresh]); + + return { probe, pending, refusal, choose, refresh, retry }; +} diff --git a/frontend/src/shared/TopLayerPortal.tsx b/frontend/src/shared/TopLayerPortal.tsx new file mode 100644 index 00000000..bdc47ceb --- /dev/null +++ b/frontend/src/shared/TopLayerPortal.tsx @@ -0,0 +1,16 @@ +import React from 'react'; +import { createPortal } from 'react-dom'; + +// A low-z or transformed ancestor traps position:fixed, so anything that must win every stacking fight mounts on body instead. +const TOP_LAYER = 2147483647; + +interface Props { + children: React.ReactNode; +} + +const TopLayerPortal: React.FC = ({ children }) => createPortal( +
{children}
, + document.body, +); + +export default TopLayerPortal; diff --git a/frontend/src/shared/cardScrollFocus.ts b/frontend/src/shared/cardScrollFocus.ts index 21489fa8..6c523a09 100644 --- a/frontend/src/shared/cardScrollFocus.ts +++ b/frontend/src/shared/cardScrollFocus.ts @@ -1,11 +1,32 @@ // The card you've clicked INTO, so plain scroll reads its content (chat transcript, scheduled-task // list) while scroll everywhere else zooms the canvas (Google Maps model). Imperative + read on the -// wheel handler so no re-render; cleared when you click blank canvas. Browser/app cards aren't tracked -// here: their guest page owns its own scroll/zoom (Maps, Figma), so plain wheel always stays in them. +// wheel handler so no re-render. Browser/app cards aren't tracked here: their guest page owns its own +// scroll/zoom (Maps, Figma), so plain wheel always stays in them. let scrollFocusedCardId: string | null = null; +// Message bubbles carry their own data-select-id, so match the card by walking, not by closest(). +function insideFocusedCard(target: EventTarget | null): boolean { + let el = target as HTMLElement | null; + while (el) { + if (el.getAttribute?.('data-select-id') === scrollFocusedCardId) return true; + el = el.parentElement; + } + return false; +} + +// Focus follows the cursor: the moment it leaves the card, scroll belongs to the canvas again, so a +// chat you once clicked can't own the wheel forever and make zoom look broken. Listen on pointermove, +// not pointerover: boundary events also fire when a re-render swaps the element under a still cursor. +function releaseOnPointerLeave(e: PointerEvent): void { + if (insideFocusedCard(e.target)) return; + setScrollFocusedCard(null); +} + export function setScrollFocusedCard(id: string | null): void { + if (id === scrollFocusedCardId) return; scrollFocusedCardId = id; + if (id) document.addEventListener('pointermove', releaseOnPointerLeave, true); + else document.removeEventListener('pointermove', releaseOnPointerLeave, true); } export function getScrollFocusedCard(): string | null { diff --git a/frontend/src/shared/notifications.ts b/frontend/src/shared/notifications.ts index f7f0beaa..b52d92cc 100644 --- a/frontend/src/shared/notifications.ts +++ b/frontend/src/shared/notifications.ts @@ -59,3 +59,66 @@ export function notifyAgentCompletion(p: AgentCompletionPayload): void { // Notification API can throw if sandboxed or headless; fail silently. } } + +export type WorkflowNotificationOutcome = 'open' | 'ack' | 'rerun' | 'edit'; + +export interface WorkflowRunNotification { + workflowId: string; + workflowTitle: string; + runId?: string; + sessionId?: string; + status: string; + tierKind?: string; + fallback?: boolean; +} + +const SUCCESS_TITLES = ['{name} is done', '{name} just wrapped up', 'Heads up: {name} finished', '{name} is ready']; +const FAILURE_TITLES = ['{name} hit a snag', "{name} couldn't finish", 'Something went sideways on {name}']; +const LATE_TITLES = ['{name} caught up late', '{name} ran late but made it']; + +function workflowTitleFor(p: WorkflowRunNotification): string { + const name = p.workflowTitle || 'Workflow'; + const pool = p.status === 'success' ? SUCCESS_TITLES + : p.status === 'failure' ? FAILURE_TITLES + : p.status === 'ran_late' ? LATE_TITLES + : null; + if (!pool) return `${name}: ${p.status}`; + // Seed by workflow id + current minute so two workflows pick different copy while one workflow stays stable across a few minutes. + const seed = Math.abs((p.workflowId.length + Math.floor(Date.now() / 60_000)) | 0); + return pool[seed % pool.length].replace('{name}', name); +} + +function workflowBodyFor(p: WorkflowRunNotification): string { + if (p.tierKind && p.fallback) { + return `Would have ${p.tierKind === 'call' ? 'called' : 'texted'} you. (Cloud SMS not wired yet.)`; + } + const verb = typeof navigator !== 'undefined' && /Mac/i.test(navigator.platform) ? 'Tap' : 'Click'; + if (p.status === 'success') return `${verb} to see what it did.`; + if (p.status === 'failure') return `${verb} to see what went wrong.`; + return `${verb} to open the run.`; +} + +/** Native OS notification for a finished workflow run. Prefers the Electron main process, which reaches Notification Center even with the window hidden or the renderer backgrounded; the renderer's own Notification API is the browser-only fallback and it only fires when the tab is hidden. */ +export function notifyWorkflowRun(p: WorkflowRunNotification): void { + const bridge = typeof window !== 'undefined' ? window.openswarm : undefined; + if (!bridge?.notify) { + notifyAgentCompletion({ + sessionId: p.sessionId || p.workflowId, + sessionName: p.workflowTitle || 'Workflow', + status: p.status === 'success' ? 'completed' : 'error', + }); + return; + } + bridge.notify({ + title: workflowTitleFor(p), + body: workflowBodyFor(p), + deepLink: `openswarm://workflow/${p.workflowId}/run/${p.runId || ''}`, + runId: p.runId, + workflowId: p.workflowId, + actions: [ + { text: 'Looks good', outcome: 'ack' }, + { text: 'Re-run', outcome: 'rerun' }, + { text: 'Adjust', outcome: 'edit' }, + ], + }).catch(() => { /* the OS refused it; the run is still on the canvas */ }); +} diff --git a/frontend/src/shared/state/dashboardLayoutSlice.ts b/frontend/src/shared/state/dashboardLayoutSlice.ts index c00233d9..f459634c 100644 --- a/frontend/src/shared/state/dashboardLayoutSlice.ts +++ b/frontend/src/shared/state/dashboardLayoutSlice.ts @@ -1,5 +1,6 @@ import { createSlice, createAsyncThunk, PayloadAction, createAction } from '@reduxjs/toolkit'; -import { launchAndSendFirstMessage, resumeSession } from './agentsSlice'; +import { launchAndSendFirstMessage, resumeSession, collapseSession, collapseAllSessions, setExpandedSessionIds } from './agentsSlice'; +import { untileClosedChats } from './untileClosedChats'; import { API_BASE } from '@/shared/config'; import { getLastDashboardId } from '@/shared/lastDashboardId'; @@ -26,7 +27,9 @@ export const DEFAULT_WORKFLOWS_HUB_W = DEFAULT_BROWSER_CARD_W; export const DEFAULT_WORKFLOWS_HUB_H = DEFAULT_BROWSER_CARD_H; export const DEFAULT_SETTINGS_CARD_W = 900; export const DEFAULT_SETTINGS_CARD_H = 640; +// The two singleton windows have no card map to key off, so they own these fixed ids everywhere (selection, minimize, z-order). export const SETTINGS_CARD_ID = 'settings'; +export const WORKFLOWS_HUB_ID = 'workflows-hub'; export const EXPANDED_CARD_MIN_H = 620; export const GRID_GAP = 24; // Gap between the Workflows window and the cards it spawns (run monitor, that monitor's browser). Keeps the hub -> monitor -> browser row evenly spaced. @@ -110,8 +113,6 @@ export interface WorkflowsHubPosition { width: number; height: number; zOrder: number; - // Full size view: the card fills the whole dashboard (reuses the fullscreen tile geometry). - fullscreen?: boolean; } // One entry in the Ctrl/Cmd+Shift+T "reopen last closed" stack: a full snapshot for browser/view/workflow/tab, just the session id for an agent (its session is brought back via resumeSession). @@ -344,10 +345,11 @@ export function findOpenGridCell( occupiedRects: Rect[], newW: number, newH: number, + colLimit?: number, ): { x: number; y: number } { const cellW = DEFAULT_CARD_W + GRID_GAP; const cellH = DEFAULT_CARD_H + GRID_GAP; - const maxCols = Math.max( + const maxCols = colLimit ?? Math.max( 1, Math.floor((window.innerWidth - GRID_ORIGIN.x) / cellW) || GRID_COLS_FALLBACK, ); @@ -364,6 +366,34 @@ export function findOpenGridCell( } } +// Tidy packs into the grid shape that fills the SCREEN best. The default column count is derived from +// window.innerWidth, which is screen pixels pretending to be world units: it laid 8 cards out as a +// 2-wide, 4-tall ribbon that the camera then had to pull back to 41% to show. +function tidyColumnCount(itemSizes: Array<{ w: number; h: number }>): number { + const cellW = DEFAULT_CARD_W + GRID_GAP; + const cellH = DEFAULT_CARD_H + GRID_GAP; + let cells = 0; + let widest = 1; + for (const s of itemSizes) { + const cols = Math.max(1, Math.ceil(s.w / cellW)); + cells += cols * Math.max(1, Math.ceil(s.h / cellH)); + widest = Math.max(widest, cols); + } + const vw = window.innerWidth || 1440; + const vh = window.innerHeight || 900; + let best = widest; + let bestZoom = 0; + for (let cols = widest; cols <= Math.max(widest, cells); cols++) { + const rows = Math.ceil(cells / cols); + const zoom = Math.min(vw / (cols * cellW), vh / (rows * cellH)); + if (zoom > bestZoom) { + bestZoom = zoom; + best = cols; + } + } + return best; +} + // Like findOpenGridCell but biased to stay near a proposed (x,y) anchor. Used when the backend hands us a card with a position that's already occupied (sub-agent or sub-browser spawning on top of its parent or a sibling). Spirals outward from the anchor on a grid, snapping to cell-aligned positions so the result still looks intentional, not dropped from orbit. Caps the spiral search at ~1000 cells to avoid pathological work in adversarial layouts, falls back to findOpenGridCell after that. Cost: O(rects × cells_scanned). Spawn events are rare (not per-frame), so this only runs when a new card appears. Typical scan resolves in <10 cells, well below the cap. No perf impact on steady-state UI. export function findOpenSpotNear( anchorX: number, @@ -569,16 +599,17 @@ const dashboardLayoutSlice = createSlice({ name: 'dashboardLayout', initialState, reducers: { - // Window controls (traffic lights). Minimize toggles a per-card pill; tiling snaps a card to a - // macOS-style viewport zone (green = 'fill'). Minimizing an un-tiles and vice-versa, so a card - // is never both pill'd and tiled at once. + // Window controls (traffic lights). Minimize parks a card in the right-edge rail; tiling snaps it + // to a macOS-style viewport zone. Rule 6 of the tiling set (see cards/useCardTiling.ts): a parked + // card keeps its zone and restores back into it, but never keeps 'fullscreen', which would leave + // an off-canvas card hiding the whole shell. toggleMinimizeCard(state, action: PayloadAction<{ cardId: string }>) { const id = action.payload.cardId; if (state.minimizedCards[id]) { delete state.minimizedCards[id]; } else { state.minimizedCards[id] = true; - if (state.tiledCards[id]) delete state.tiledCards[id]; + if (state.tiledCards[id] === 'fullscreen') delete state.tiledCards[id]; } }, setTiledCard(state, action: PayloadAction<{ cardId: string; zone: string }>) { @@ -766,19 +797,19 @@ const dashboardLayoutSlice = createSlice({ ]; allItems.sort((a, b) => a.y - b.y || a.x - b.x); + const sizeOf = (item: typeof allItems[number]): { w: number; h: number } => ({ + w: item.storedW, + h: item.kind === 'agent' && expanded.has(item.id) + ? Math.max(EXPANDED_CARD_MIN_H, item.storedH) + : item.storedH, + }); + const cols = tidyColumnCount(allItems.map(sizeOf)); const placedRects: Rect[] = []; for (const item of allItems) { - let w: number, h: number; - if (item.kind === 'agent') { - w = item.storedW; - h = expanded.has(item.id) ? Math.max(EXPANDED_CARD_MIN_H, item.storedH) : item.storedH; - } else { - w = item.storedW; - h = item.storedH; - } + const { w, h } = sizeOf(item); - const pos = findOpenGridCell(placedRects, w, h); + const pos = findOpenGridCell(placedRects, w, h, cols); placedRects.push({ x: pos.x, y: pos.y, w, h }); if (item.kind === 'agent') { @@ -1093,6 +1124,9 @@ const dashboardLayoutSlice = createSlice({ removeWorkflowCard(state, action: PayloadAction) { delete state.workflowCards[action.payload]; + // Rule 7: a dead card must never keep owning a tile; a stale entry poisons every reader of it. + delete state.tiledCards[action.payload]; + delete state.minimizedCards[action.payload]; }, // Rekey draft- id to the server-assigned id without visually hopping the card. @@ -1115,6 +1149,7 @@ const dashboardLayoutSlice = createSlice({ openWorkflowsHub(state, action: PayloadAction<{ expandedSessionIds?: string[] } | undefined>) { if (state.workflowsHub) { state.workflowsHub.zOrder = state.nextZOrder++; + delete state.minimizedCards[WORKFLOWS_HUB_ID]; state.pendingFocusWorkflowsHub = true; return; } @@ -1136,6 +1171,8 @@ const dashboardLayoutSlice = createSlice({ closeWorkflowsHub(state) { state.workflowsHub = null; + delete state.minimizedCards[WORKFLOWS_HUB_ID]; + delete state.tiledCards[WORKFLOWS_HUB_ID]; }, // The Workflows app is an on-canvas card (like chat/browser/view cards), backed by the singleton workflowsHub geometry. Opening it creates or raises that card and pans to it; an optional workflowId deep-links to that workflow's detail once the card mounts. @@ -1143,6 +1180,8 @@ const dashboardLayoutSlice = createSlice({ state.workflowsAppTarget = action.payload?.workflowId ?? null; if (state.workflowsHub) { state.workflowsHub.zOrder = state.nextZOrder++; + // Opening means visible: a parked window must come back to the canvas, or the focus pan flies to empty space. + delete state.minimizedCards[WORKFLOWS_HUB_ID]; state.pendingFocusWorkflowsHub = true; return; } @@ -1160,19 +1199,14 @@ const dashboardLayoutSlice = createSlice({ closeWorkflowsApp(state) { state.workflowsHub = null; + delete state.tiledCards[WORKFLOWS_HUB_ID]; + delete state.minimizedCards[WORKFLOWS_HUB_ID]; state.workflowsAppTarget = null; state.workflowsMonitorId = null; state.workflowsMonitorRunId = null; state.workflowsMonitorCard = null; }, - toggleWorkflowsHubFullscreen(state) { - if (state.workflowsHub) { - state.workflowsHub.fullscreen = !state.workflowsHub.fullscreen; - state.workflowsHub.zOrder = state.nextZOrder++; - } - }, - clearWorkflowsAppTarget(state) { state.workflowsAppTarget = null; }, @@ -1232,6 +1266,7 @@ const dashboardLayoutSlice = createSlice({ openSettingsCard(state, action: PayloadAction<{ expandedSessionIds?: string[] } | undefined>) { if (state.settingsCard) { state.settingsCard.zOrder = state.nextZOrder++; + delete state.minimizedCards[SETTINGS_CARD_ID]; state.pendingFocusSettingsCard = true; return; } @@ -1249,6 +1284,8 @@ const dashboardLayoutSlice = createSlice({ closeSettingsCard(state) { state.settingsCard = null; + delete state.minimizedCards[SETTINGS_CARD_ID]; + delete state.tiledCards[SETTINGS_CARD_ID]; state.pendingFocusSettingsCard = false; }, @@ -1256,12 +1293,6 @@ const dashboardLayoutSlice = createSlice({ state.pendingFocusSettingsCard = false; }, - toggleSettingsCardFullscreen(state) { - if (!state.settingsCard) return; - state.settingsCard.fullscreen = !state.settingsCard.fullscreen; - state.settingsCard.zOrder = state.nextZOrder++; - }, - setSettingsCardPosition(state, action: PayloadAction<{ x: number; y: number }>) { if (!state.settingsCard) return; state.settingsCard.x = action.payload.x; @@ -1453,8 +1484,14 @@ const dashboardLayoutSlice = createSlice({ const [moved] = source.tabs.splice(idx, 1); // Fresh id: reusing the old one makes the receiving BrowserCard think the tab is already initialized, so its webview never loads the URL and sits at about:blank. const tab = { ...moved, id: generateTabId() }; - if (source.tabs.length === 0) { + const spun = { + owner: source.spawned_by ?? null, + x: source.x, y: source.y, width: source.width, height: source.height, dashboard_id: source.dashboard_id, + }; + const dissolved = source.tabs.length === 0; + if (dissolved) { delete state.browserCards[fromBrowserId]; + delete state.suspendedBrowserCards[fromBrowserId]; } else if (source.activeTabId === tabId) { const nextActive = source.tabs[Math.min(idx, source.tabs.length - 1)]; source.activeTabId = nextActive.id; @@ -1466,18 +1503,24 @@ const dashboardLayoutSlice = createSlice({ target.url = tab.url; target.zOrder = state.nextZOrder++; } else { - const id = `browser-${Date.now().toString(36)}`; + // The last tab leaving is the card MOVING, so it keeps its id; either way it keeps its agent + // owner, or the agent stops recognising its own browser and spawns a second one next time it + // browses. keep_open because pulling it out is the user claiming it: the owner finishing must + // not delete it out from under them, exactly as when it had no owner at all. + const id = dissolved ? fromBrowserId : `browser-${Date.now().toString(36)}`; state.browserCards[id] = { browser_id: id, url: tab.url, tabs: [tab], activeTabId: tab.id, - x: x ?? source.x + 60, - y: y ?? source.y + 60, - width: source.width, - height: source.height, + x: x ?? spun.x + 60, + y: y ?? spun.y + 60, + width: spun.width, + height: spun.height, zOrder: state.nextZOrder++, - dashboard_id: source.dashboard_id, + dashboard_id: spun.dashboard_id, + spawned_by: spun.owner, + keep_open: true, }; } }, @@ -1746,6 +1789,17 @@ const dashboardLayoutSlice = createSlice({ // Fail-open for RENDERING only; saveArmed stays false so this client can never persist the empty layout it booted with over the server's real one (the wipe that hit 2026-07-20). state.initialized = true; }) + // Rule 8 of the tiling set: a chat's zone belongs to its OPEN state, so every action that closes + // chats untiles them here, in the same dispatch. See untileClosedChats. + .addCase(collapseSession, (state, action) => { + untileClosedChats(state.tiledCards, [action.payload], []); + }) + .addCase(collapseAllSessions, (state) => { + untileClosedChats(state.tiledCards, Object.keys(state.cards), []); + }) + .addCase(setExpandedSessionIds, (state, action) => { + untileClosedChats(state.tiledCards, Object.keys(state.cards), action.payload); + }) .addCase(fetchSessionRejectedAction, (state, action) => { // 404/410 means permanent; strip the card. Other failure modes leave it (next fetch may succeed). const payload = action.payload; @@ -1766,6 +1820,13 @@ const dashboardLayoutSlice = createSlice({ delete state.cards[draftId]; state.cards[session.id] = { ...card, session_id: session.id, zOrder: state.nextZOrder++ }; } + // The zone rides the re-key too: left behind, a tiled draft pops out of its tile AND strands an + // entry no reader can ever clear (a stranded 'fullscreen' hides the whole shell until reload). + const draftZone = state.tiledCards[draftId]; + if (draftZone) { + delete state.tiledCards[draftId]; + state.tiledCards[session.id] = draftZone; + } // Carry an optimistic browser tether from the draft id to the real session id, in place (no flicker, no stale draft endpoint). for (const entry of Object.values(state.glowingBrowserCards)) { if (entry.sourceId === draftId) entry.sourceId = session.id; @@ -1853,7 +1914,6 @@ export const { closeWorkflowsHub, openWorkflowsApp, closeWorkflowsApp, - toggleWorkflowsHubFullscreen, clearWorkflowsAppTarget, openWorkflowMonitor, closeWorkflowMonitor, @@ -1866,7 +1926,6 @@ export const { openSettingsCard, closeSettingsCard, clearPendingFocusSettingsCard, - toggleSettingsCardFullscreen, setSettingsCardPosition, setSettingsCardSize, recordClosedCard, @@ -1901,8 +1960,9 @@ export const selectFullscreenCardId = (state: { dashboardLayout: DashboardLayout if (!entry) return null; const id = entry[0]; // Belt over the reducer hygiene: an entry whose card is gone (any removal path) must not hold the app in fullscreen. - const exists = id in s.cards || id in s.viewCards || id in s.browserCards || id in s.workflowCards; - return exists ? id : null; + const exists = id in s.cards || id in s.viewCards || id in s.browserCards || id in s.workflowCards + || (id === WORKFLOWS_HUB_ID && !!s.workflowsHub) || (id === SETTINGS_CARD_ID && !!s.settingsCard); + return exists && !s.minimizedCards[id] ? id : null; }; export default dashboardLayoutSlice.reducer; diff --git a/frontend/src/shared/state/isUserLaunchedSession.ts b/frontend/src/shared/state/isUserLaunchedSession.ts new file mode 100644 index 00000000..20c595cf --- /dev/null +++ b/frontend/src/shared/state/isUserLaunchedSession.ts @@ -0,0 +1,14 @@ +// Plumbing chats the UI spins up for itself: a workflow card's Edit Agent, a workflow run, a browser +// or sub agent working for a parent. They are real sessions, they just aren't things the user started. +const PLUMBING_MODES: ReadonlySet = new Set(['browser-agent', 'invoked-agent', 'sub-agent']); + +export interface SessionOrigin { + mode: string; + workflow_run_id?: string | null; + workflow_edit_id?: string | null; +} + +/** True for a chat the user started themselves, which is the only kind that earns a card or a notification. */ +export function isUserLaunchedSession(session: SessionOrigin): boolean { + return !session.workflow_run_id && !session.workflow_edit_id && !PLUMBING_MODES.has(session.mode); +} diff --git a/frontend/src/shared/state/settingsSlice.ts b/frontend/src/shared/state/settingsSlice.ts index 82295bb9..47f10c40 100644 --- a/frontend/src/shared/state/settingsSlice.ts +++ b/frontend/src/shared/state/settingsSlice.ts @@ -46,6 +46,7 @@ export interface AppSettings { theme: 'light' | 'dark'; new_agent_shortcut: string; dictation_shortcut?: string | null; + dictation_model?: string | null; anthropic_api_key: string | null; openai_api_key?: string | null; google_api_key?: string | null; @@ -139,7 +140,12 @@ export interface BrowseResult { interface SettingsState { data: AppSettings; loading: boolean; + /** We have the user's real settings. Anything that judges the user (no model connected, onboarding + * not done, free runs spent) must read THIS, never `settled`: a failed fetch is not an answer. */ loaded: boolean; + /** The first fetch finished, either way. Only the paint gate wants this, so a dead backend shows a + * window with an honest error instead of a blank one. */ + settled: boolean; modalOpen: boolean; /** When non-null, Settings opens to this tab instead of 'general'. */ initialTab: string | null; @@ -168,6 +174,7 @@ export const DEFAULT_SETTINGS: AppSettings = { theme: 'light', new_agent_shortcut: 'Meta+l', dictation_shortcut: null, + dictation_model: null, anthropic_api_key: null, browser_homepage: 'https://duckduckgo.com', browser_import_signins: false, @@ -182,6 +189,7 @@ const initialState: SettingsState = { data: DEFAULT_SETTINGS, loading: false, loaded: false, + settled: false, modalOpen: false, initialTab: null, draft: null, @@ -334,6 +342,7 @@ const settingsSlice = createSlice({ .addCase(fetchSettings.fulfilled, (state, action) => { state.loading = false; state.loaded = true; + state.settled = true; // Drop a stale response: on boot three fetches race (initial, sub-sync, free-trial mint); if the pre-mint one resolves last it would wipe the armed trial. Newest wins. if (state.latestWriteId && action.meta.requestId !== state.latestWriteId) return; // Fill any field an older backend shape omitted so no consumer reads undefined; the payload still wins for everything it does send. @@ -347,7 +356,8 @@ const settingsSlice = createSlice({ }) .addCase(fetchSettings.rejected, (state) => { state.loading = false; - state.loaded = true; + state.settled = true; + // `loaded` deliberately stays put. A failed fetch is not an answer, and claiming one made every gate read DEFAULT_SETTINGS as fact: with the backend down the app greeted a configured user as a brand-new one and offered to sell them a subscription. A refresh that fails keeps the last good copy; a first fetch that fails stays unknown. }) .addCase(updateSettingsPatch.fulfilled, (state, action) => { // A user save is authoritative; claim newest so an in-flight GET can't overwrite it, and consume the draft so reopening shows the saved state. diff --git a/frontend/src/shared/state/untileClosedChats.test.ts b/frontend/src/shared/state/untileClosedChats.test.ts new file mode 100644 index 00000000..6d208f1f --- /dev/null +++ b/frontend/src/shared/state/untileClosedChats.test.ts @@ -0,0 +1,28 @@ +// Run: node --test frontend/src/shared/state/untileClosedChats.test.ts +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { untileClosedChats } from './untileClosedChats.ts'; + +test('collapsing one chat drops that chat tile and nothing else', () => { + const tiled: Record = { a: 'left', b: 'fullscreen' }; + untileClosedChats(tiled, ['a'], []); + assert.deepEqual(tiled, { b: 'fullscreen' }); +}); + +test('restoring a saved expansion list untiles every chat missing from it', () => { + const tiled: Record = { a: 'left', b: 'fullscreen', c: 'right' }; + untileClosedChats(tiled, ['a', 'b', 'c'], ['b']); + assert.deepEqual(tiled, { b: 'fullscreen' }); +}); + +test('non-chat tiles (browsers, apps, windows) are never touched', () => { + const tiled: Record = { chat: 'left', 'browser-1': 'right' }; + untileClosedChats(tiled, ['chat'], []); + assert.deepEqual(tiled, { 'browser-1': 'right' }); +}); + +test('an open chat keeps its tile', () => { + const tiled: Record = { a: 'fullscreen' }; + untileClosedChats(tiled, ['a'], ['a']); + assert.deepEqual(tiled, { a: 'fullscreen' }); +}); diff --git a/frontend/src/shared/state/untileClosedChats.ts b/frontend/src/shared/state/untileClosedChats.ts new file mode 100644 index 00000000..8c93fc75 --- /dev/null +++ b/frontend/src/shared/state/untileClosedChats.ts @@ -0,0 +1,14 @@ +// A tiled chat IS an open chat, so the two states can never be allowed to drift apart. Anything that +// closes chats drops their zones in the SAME reducer: an after-the-fact effect leaves a window where +// the card is tile-sized but wearing the collapsed skin, and on a busy machine that window has been +// measured at 17 seconds (the "fullscreen turns white" bug). +export function untileClosedChats( + tiledCards: Record, + chatCardIds: string[], + openChatIds: readonly string[], +): void { + const open = new Set(openChatIds); + for (const id of chatCardIds) { + if (!open.has(id)) delete tiledCards[id]; + } +} diff --git a/frontend/src/shared/state/workflowsSlice.ts b/frontend/src/shared/state/workflowsSlice.ts index 445e0250..b5f77347 100644 --- a/frontend/src/shared/state/workflowsSlice.ts +++ b/frontend/src/shared/state/workflowsSlice.ts @@ -74,6 +74,9 @@ export interface Workflow { steps: WorkflowStep[]; actions: ActionsConfig; schedule: ScheduleConfig; + /** Where a SCHEDULED fire runs. Only the cloud_workflows routes may change it, and only once + * the cloud has actually taken (or released) the workflow. Manual runs are always local. */ + execution_target?: 'device' | 'cloud'; permissions: PermissionTier[]; source_session_id?: string | null; dashboard_id?: string | null; @@ -200,12 +203,14 @@ interface State { allRuns: WorkflowRun[]; allRunsLoading: boolean; runningToast: RunningToast | null; + /** One-off explanation for something the user asked for that the server refused. */ + noticeToast: string | null; runControlPending: Record; deleted: Workflow[]; deletedLoading: boolean; } -const initialState: State = { items: {}, runs: {}, openCards: {}, loaded: false, loading: false, paused: false, active: [], cloudSmsEnabled: false, allRuns: [], allRunsLoading: false, runningToast: null, runControlPending: {}, deleted: [], deletedLoading: false }; +const initialState: State = { items: {}, runs: {}, openCards: {}, loaded: false, loading: false, paused: false, active: [], cloudSmsEnabled: false, allRuns: [], allRunsLoading: false, runningToast: null, noticeToast: null, runControlPending: {}, deleted: [], deletedLoading: false }; function mergeRunIntoState(state: State, r: WorkflowRun) { const arr = state.runs[r.workflow_id] || []; @@ -402,8 +407,13 @@ export const discardDraft = createAsyncThunk('workflows/discardDraft', async (id return (await res.json()) as Workflow; }); -export const deleteWorkflow = createAsyncThunk('workflows/delete', async (id: string) => { - await fetch(`${API}/${id}`, { method: 'DELETE' }); +export const deleteWorkflow = createAsyncThunk('workflows/delete', async (id: string, { rejectWithValue }) => { + const res = await fetch(`${API}/${id}`, { method: 'DELETE' }); + // A refused delete must not remove the card: the workflow is still there, and if it is cloud-hosted it is still running. + if (!res.ok) { + const body = await res.json().catch(() => null); + return rejectWithValue(typeof body?.detail === 'string' ? body.detail : "Couldn't delete this workflow. Try again in a moment."); + } return id; }); @@ -414,9 +424,12 @@ export const fetchDeletedWorkflows = createAsyncThunk('workflows/fetchDeleted', return data.workflows as Workflow[]; }); -export const restoreWorkflow = createAsyncThunk('workflows/restore', async (id: string) => { +export const restoreWorkflow = createAsyncThunk('workflows/restore', async (id: string, { dispatch }) => { const res = await fetch(`${API}/${id}/restore`, { method: 'POST' }); if (!res.ok) throw new Error(`restore failed ${res.status}`); + // Trashing drops the run rows from the store; the server kept them, so pull them back or a restored workflow claims "No runs yet" over a real history. + void dispatch(fetchRuns(id)); + void dispatch(fetchAllRuns(200)); return (await res.json()) as Workflow; }); @@ -567,6 +580,9 @@ const slice = createSlice({ dismissRunningToast(state) { state.runningToast = null; }, + dismissNoticeToast(state) { + state.noticeToast = null; + }, }, extraReducers: (builder) => { builder @@ -588,6 +604,9 @@ const slice = createSlice({ .addCase(updateWorkflow.fulfilled, (state, action) => { state.items[action.payload.id] = action.payload; }) .addCase(commitDraft.fulfilled, (state, action) => { state.items[action.payload.id] = action.payload; }) .addCase(discardDraft.fulfilled, (state, action) => { state.items[action.payload.id] = action.payload; }) + .addCase(deleteWorkflow.rejected, (state, action) => { + state.noticeToast = typeof action.payload === 'string' ? action.payload : "Couldn't delete this workflow. Try again in a moment."; + }) .addCase(deleteWorkflow.fulfilled, (state, action) => { delete state.items[action.payload]; delete state.runs[action.payload]; @@ -667,5 +686,6 @@ export const { upsertWorkflow, removeWorkflow, dismissRunningToast, + dismissNoticeToast, } = slice.actions; export default slice.reducer; diff --git a/frontend/src/shared/voice/VoiceDictationContext.tsx b/frontend/src/shared/voice/VoiceDictationContext.tsx index b78a2d65..06c216e1 100644 --- a/frontend/src/shared/voice/VoiceDictationContext.tsx +++ b/frontend/src/shared/voice/VoiceDictationContext.tsx @@ -12,12 +12,19 @@ export function VoiceDictationProvider({ children }: { children: React.ReactNode const { state, lastText, error, pct, feedback, toggle, start, stop, volumeRef } = useVoiceDictation(); const holdMode = useAppSelector((s) => s.settings.data.voice_hold_to_talk ?? true); const dictationShortcut = useAppSelector((s) => s.settings.data.dictation_shortcut ?? null); + const dictationModel = useAppSelector((s) => s.settings.data.dictation_model ?? null); // Push the user's combo to main on boot and on change so every hotkey tier rebinds live. useEffect(() => { const bridge = window as unknown as { openswarm?: { setVoiceHotkey?: (combo: string | null) => void } }; bridge.openswarm?.setVoiceHotkey?.(dictationShortcut); }, [dictationShortcut]); + + // Main boots with the catalog default, so a user who picked something else has to say so on every + // launch or dictation quietly runs on the wrong model. + useEffect(() => { + if (dictationModel) void window.openswarm?.voiceSetModel?.(dictationModel); + }, [dictationModel]); const stateRef = useRef(state); stateRef.current = state; const heldRef = useRef(false); @@ -28,7 +35,7 @@ export function VoiceDictationProvider({ children }: { children: React.ReactNode // flips state to 'recording', so without this the mic could be started by a click but never // stopped by one. if (stateRef.current === 'recording') { heldRef.current = false; void stop(); return; } - if (stateRef.current === 'idle') { heldRef.current = true; void start(); } + if (stateRef.current === 'idle') { heldRef.current = true; void start(true); } } else { toggle(); } diff --git a/frontend/src/shared/voice/createSilenceDetector.ts b/frontend/src/shared/voice/createSilenceDetector.ts new file mode 100644 index 00000000..f14b4ba6 --- /dev/null +++ b/frontend/src/shared/voice/createSilenceDetector.ts @@ -0,0 +1,65 @@ +// Endpointing: decide when the speaker has finished so tap-to-dictate stops itself instead of +// recording (and transcribing) the gap between "done talking" and "remembered to press again". +// Energy based on purpose: the mic stream is already float PCM in hand, and a Silero ONNX model +// would mean shipping onnxruntime into the renderer for a decision this cheap. +// +// Every constant is borrowed from shipping code rather than guessed: the 0.004 RMS speech floor is +// TypeWhisper's, the 0.02 peak companion is openwhispr's, and the 1400ms silence window with a +// 400ms minimum-speech guard is what Whispering's Silero endpointer uses. +// +// The peak test is the load-bearing half. Speech is spiky and room tone is flat, so RMS alone calls +// a noisy room "talking" forever and the recording never ends; requiring a real peak separates them. + +const FRAME_MS = 20; +const SPEECH_RMS = 0.004; +const SPEECH_PEAK = 0.02; +const MIN_SPEECH_MS = 400; +const SILENCE_HOLD_MS = 1400; +const MAX_UTTERANCE_MS = 120_000; + +export type SilenceVerdict = 'listening' | 'ended' | 'too-long'; + +export interface SilenceDetector { + // Feed every captured chunk; any verdict other than 'listening' means stop recording now. + push(samples: Float32Array): SilenceVerdict; +} + +export function createSilenceDetector(sampleRate: number): SilenceDetector { + const frameSize = Math.max(1, Math.round((sampleRate * FRAME_MS) / 1000)); + let elapsedMs = 0; + let speechMs = 0; + let silenceMs = 0; + // Frames straddle chunk boundaries, so carry the partial frame across pushes; dropping the + // remainder would make every constant here quietly run ~6% long. + let sumSquares = 0; + let peak = 0; + let framed = 0; + + return { + push(samples: Float32Array): SilenceVerdict { + for (let i = 0; i < samples.length; i++) { + const v = samples[i]; + sumSquares += v * v; + const mag = v < 0 ? -v : v; + if (mag > peak) peak = mag; + if (++framed < frameSize) continue; + const rms = Math.sqrt(sumSquares / frameSize); + const isSpeech = rms >= SPEECH_RMS && peak >= SPEECH_PEAK; + sumSquares = 0; + peak = 0; + framed = 0; + elapsedMs += FRAME_MS; + if (elapsedMs >= MAX_UTTERANCE_MS) return 'too-long'; + if (isSpeech) { + speechMs += FRAME_MS; + silenceMs = 0; + } else if (speechMs >= MIN_SPEECH_MS) { + // Only count quiet AFTER real speech, so an empty room never ends a recording by itself. + silenceMs += FRAME_MS; + } + if (speechMs >= MIN_SPEECH_MS && silenceMs >= SILENCE_HOLD_MS) return 'ended'; + } + return 'listening'; + }, + }; +} diff --git a/frontend/src/shared/voice/useVoiceDictation.ts b/frontend/src/shared/voice/useVoiceDictation.ts index 23426ec2..e2028a31 100644 --- a/frontend/src/shared/voice/useVoiceDictation.ts +++ b/frontend/src/shared/voice/useVoiceDictation.ts @@ -3,6 +3,7 @@ import { API_BASE } from '@/shared/config'; import { encodeWav, VOICE_SAMPLE_RATE } from './encodeWav'; import { playVoiceCue } from './voiceCues'; import { injectAtFocus } from './injectAtFocus'; +import { createSilenceDetector } from './createSilenceDetector'; export type VoiceState = 'idle' | 'recording' | 'transcribing' | 'preparing'; @@ -63,6 +64,8 @@ export function useVoiceDictation() { const stateRef = useRef('idle'); // Live mic level (0..1) for the aurora; a ref, not state, so 60Hz visuals never re-render React. const volumeRef = useRef(0); + // The capture callback has to reach stop(), which is defined after start(); a ref breaks the knot. + const stopRef = useRef<(() => Promise) | null>(null); stateRef.current = state; // First-run: the model is downloading. Poll progress until it lands, then drop back to idle so the @@ -95,7 +98,9 @@ export function useVoiceDictation() { return out; }, []); - const start = useCallback(async (): Promise => { + // `hold` = the user is physically holding a key or button, so they own the end of the utterance + // and silence must never cut them off mid-thought. Only a tap session endpoints itself. + const start = useCallback(async (hold = false): Promise => { if (stateRef.current !== 'idle') return; if (!window.openswarm?.voiceTranscribe) { setError('desktop-only'); return; } // no Electron bridge = web build setError(null); @@ -105,6 +110,7 @@ export function useVoiceDictation() { const source = ctx.createMediaStreamSource(stream); const node = ctx.createScriptProcessor(4096, 1, 1); const chunks: Float32Array[] = []; + const endpointer = hold ? null : createSilenceDetector(ctx.sampleRate); node.onaudioprocess = (e): void => { const data = e.inputBuffer.getChannelData(0); chunks.push(new Float32Array(data)); @@ -113,6 +119,7 @@ export function useVoiceDictation() { for (let i = 0; i < data.length; i += 8) sum += data[i] * data[i]; const rms = Math.sqrt(sum / (data.length / 8)); volumeRef.current = volumeRef.current * 0.7 + Math.min(1, rms * 6) * 0.3; + if (endpointer && endpointer.push(data) !== 'listening') void stopRef.current?.(); }; source.connect(node); node.connect(ctx.destination); @@ -180,6 +187,8 @@ export function useVoiceDictation() { } }, [teardown, pollModel]); + stopRef.current = stop; + const toggle = useCallback((): void => { if (stateRef.current === 'recording') void stop(); else if (stateRef.current === 'idle') void start(); diff --git a/frontend/src/shared/ws/WebSocketManager.ts b/frontend/src/shared/ws/WebSocketManager.ts index 41af5a00..f671b9cd 100644 --- a/frontend/src/shared/ws/WebSocketManager.ts +++ b/frontend/src/shared/ws/WebSocketManager.ts @@ -35,7 +35,7 @@ import { displaySessionName } from '../state/sessionDisplay'; import { upsertRun, ackRun, runWorkflowNow, openWorkflowCard, upsertWorkflow, removeWorkflow } from '../state/workflowsSlice'; import { stepsSignature } from '@/app/pages/Workflows/scheduleUtils'; import { getAuthToken } from '../config'; -import { notifyAgentCompletion } from '../notifications'; +import { notifyAgentCompletion, notifyWorkflowRun } from '../notifications'; // Phase 0 boot instrumentation: one-shot flag so we report the first streamed agent token to Electron main exactly once per app launch. Module scope (not instance) because multiple WebSocketManagers exist (one per session WS). let firstAgentResponseMarked = false; @@ -771,55 +771,17 @@ class WebSocketManager { break; case 'workflow:notify': - try { - notifyAgentCompletion({ - sessionId: data.session_id || data.workflow_id, - sessionName: data.workflow_title || 'Workflow', - status: data.status === 'success' ? 'completed' : 'error', + if (data.workflow_id) { + notifyWorkflowRun({ + workflowId: data.workflow_id, + workflowTitle: data.workflow_title || 'Workflow', + runId: data.run_id, + sessionId: data.session_id, + status: data.status, + tierKind: data.tier_kind, + fallback: data.fallback, }); - } catch { /* notifications are best-effort */ } - try { - const w: any = (window as any).openswarm; - if (w?.notify) { - // Seed by workflow id + current minute so multiple workflows pick different copy while a single workflow stays stable within a few minutes. - const seed = ((data.workflow_id || '').length + Math.floor(Date.now() / 60000)) | 0; - const SUCCESS_TITLES = [ - `${data.workflow_title || 'Workflow'} — done`, - `${data.workflow_title || 'Workflow'} just wrapped up`, - `Heads up: ${data.workflow_title || 'Workflow'} finished`, - `${data.workflow_title || 'Workflow'} is ready`, - ]; - const FAILURE_TITLES = [ - `${data.workflow_title || 'Workflow'} hit a snag`, - `${data.workflow_title || 'Workflow'} couldn't finish`, - `Something went sideways on ${data.workflow_title || 'Workflow'}`, - ]; - const LATE_TITLES = [ - `${data.workflow_title || 'Workflow'} caught up late`, - `${data.workflow_title || 'Workflow'} ran late but made it`, - ]; - const pool = data.status === 'success' ? SUCCESS_TITLES - : data.status === 'failure' ? FAILURE_TITLES - : data.status === 'ran_late' ? LATE_TITLES - : [`${data.workflow_title || 'Workflow'} • ${data.status}`]; - const title = pool[Math.abs(seed) % pool.length]; - const isMac = (typeof navigator !== 'undefined' && /Mac/i.test(navigator.platform)); - const body = data.tier_kind && data.fallback - ? `Would have ${data.tier_kind === 'call' ? 'called' : 'texted'} you. (Cloud SMS not wired yet.)` - : data.status === 'success' - ? (isMac ? 'Tap to see what it did.' : 'Click to see what it did.') - : data.status === 'failure' - ? (isMac ? 'Tap to see what went wrong.' : 'Click to see what went wrong.') - : (isMac ? 'Tap to open the run.' : 'Click to open the run.'); - const deepLink = data.workflow_id ? `openswarm://workflow/${data.workflow_id}/run/${data.run_id || ''}` : undefined; - const actions = [ - { text: 'Looks good', outcome: 'ack' }, - { text: 'Re-run', outcome: 'rerun' }, - { text: 'Adjust', outcome: 'edit' }, - ]; - w.notify({ title, body, deepLink, runId: data.run_id, workflowId: data.workflow_id, actions }); - } - } catch { /* native notif optional */ } + } break; case 'dashboard:browser_card_keep': diff --git a/frontend/src/types/electron.d.ts b/frontend/src/types/electron.d.ts index 55b62ef6..9d782dde 100644 --- a/frontend/src/types/electron.d.ts +++ b/frontend/src/types/electron.d.ts @@ -1,5 +1,14 @@ export {}; +// One entry in the dictation model catalog, as the main process reports it. +export interface VoiceModel { + id: string; + label: string; + note: string; + sizeMb: number; + installed: boolean; +} + declare global { namespace JSX { interface IntrinsicElements { @@ -31,6 +40,23 @@ declare global { total: number; } + // A finished-run notification handed to the OS by the Electron main process. + interface OpenSwarmNotifyRequest { + title: string; + body?: string; + deepLink?: string; + runId?: string; + workflowId?: string; + actions?: Array<{ text: string; outcome: 'open' | 'ack' | 'rerun' | 'edit' }>; + } + + interface OpenSwarmNotifyAction { + outcome: 'open' | 'ack' | 'rerun' | 'edit'; + runId?: string; + workflowId?: string; + deepLink?: string; + } + interface OpenSwarmAPI { getBackendPort: () => number; getWebviewPreloadPath: () => string; @@ -55,7 +81,9 @@ declare global { hardReset?: () => Promise; clearBrowserData?: () => Promise<{ ok: boolean }>; voiceWarmup?: () => Promise<{ ok: boolean; error?: string }>; - voiceStatus?: () => Promise<{ downloading: boolean; pct: number; error: string | null }>; + voiceStatus?: () => Promise<{ downloading: boolean; id: string | null; pct: number; error: string | null }>; + voiceModels?: () => Promise<{ models: VoiceModel[]; selected: string }>; + voiceSetModel?: (id: string) => Promise<{ ok: boolean; ready: boolean }>; voiceTranscribe?: (wav: ArrayBuffer) => Promise<{ ok: boolean; text?: string; error?: string }>; voiceInject?: (text: string) => Promise<{ ok: boolean; pasted?: boolean; error?: string }>; onVoiceToggle?: (cb: () => void) => () => void; @@ -63,6 +91,8 @@ declare global { voiceRequestHoldPermission?: () => Promise; onAuthUrl?: (cb: (url: string) => void) => () => void; onOauthClaim?: (cb: (url: string) => void) => () => void; + notify?: (payload: OpenSwarmNotifyRequest) => Promise; + onNotificationAction?: (cb: (payload: OpenSwarmNotifyAction) => void) => () => void; } interface Window { diff --git a/linter/README.md b/linter/README.md index 5a877caa..4676a74c 100644 --- a/linter/README.md +++ b/linter/README.md @@ -42,6 +42,16 @@ These rules apply to `.py`, `.ts`, `.tsx`, `.js`, and `.jsx` files. Both are backend-only and grandfather pre-existing debt via the `no-underscore-names` / `p-private` exception lists; new code must be clean. (Ported from Haik's linter, which also adds Pyright + Ruff and should eventually supersede this subset.) +### Cross-entity references + +**Declared reference targets (`dangling-refs`)** — Backend entities are JSON records that point at each other with bare strings (`Workflow.edit_agent_session_id`, `Output.workspace_id`, `CardPosition.session_id`). The type says `str`, so nothing warns a reader that the referent may be gone, and the miss renders as a blank card instead of a designed empty state. + +Every field named `*_id` / `*_ids` on a pydantic `BaseModel` under `backend/` must therefore be declared in `backend/config/entity_references.py`, which names the entity it points at and the store that resolves that entity by id. A model's own primary key is spelled `id`, so it never matches; neither do words that merely end in "id" (`uuid`, `grid`, `valid`), since the underscore is required. + +The registry is verified in both directions: an entry for a field that no longer exists is an error, and so is an `EntityStore` whose lookup function has been renamed away. A registry nobody checks is a registry that rots. + +Pre-existing fields are grandfathered per FIELD, not per file — the exception entries are keyed `::.`, so a new id field added to an already-listed model is still caught. A file glob would exempt `workflows/models.py` forever, which is exactly where the next dangling pointer lands. + ## How it runs ### Linter watch (automatic) @@ -154,6 +164,7 @@ The project's code conventions live here (a tracked file) rather than in `CLAUDE ### Enforced by the linter - **No leading `_`** — use `p_` for private. (`no-underscore-names`, backend) - **`p_` is a private access boundary** — a `p_` name used across files/classes must be public. (`p-private`, backend) +- **Cross-entity id fields declare their target.** A new `*_id` on a backend model must be registered in `backend/config/entity_references.py`. (`dangling-refs`, backend) - **No runtime import cycles.** (`import-cycles`) - **File and folder size caps.** (`max-file-lines`, `max-folder-items`) @@ -190,6 +201,7 @@ linter/ knip.py # knip unused-code runner endpoints.py # orphaned endpoint detection classes.py # class-level dead code detection + dangling_refs.py # cross-entity id fields must declare a target entity config/ # all configuration files config.json # enabled checks, rules, exclusions, exceptions pyrightconfig.json # python type checking config @@ -211,6 +223,7 @@ deferred. | `max-file-lines` (300) | on | Our 300-line precedence. Active for new files; existing debt is grandfathered (see below). | | `max-folder-items` (7) | on | Grandfathered per subtree via `.lintignore-max-folder-items` markers in `backend/`, `frontend/`, `debugger/`, `electron/`, `scripts/`. | | `vulture` | on | Dead-code detection over `backend/`. Runs against `backend/.venv/bin/vulture`. | +| `dangling-refs` | on | Cross-entity id fields on backend models must declare a target entity. 42 of the 74 existing fields are in the registry; the other 32 are grandfathered per field. | | `no-nested-imports` | off | We deliberately use function-level / lazy imports to break import cycles (400+ sites). Flagging them all is wrong for this codebase. | | `eslint`, `knip` | off | Node tooling, deferred to a later pass. | | `endpoints` | off | Orphaned-endpoint triage deferred. | diff --git a/linter/checks/dangling_refs.py b/linter/checks/dangling_refs.py new file mode 100644 index 00000000..bd2ba3cd --- /dev/null +++ b/linter/checks/dangling_refs.py @@ -0,0 +1,255 @@ +"""Every cross-entity id field on a backend pydantic model must say what it points at. + +OpenSwarm stores entities as JSON records that reference each other with bare strings +(``Workflow.edit_agent_session_id``, ``Output.workspace_id``, ``CardPosition.session_id``). The +type says ``str``, so nothing tells a reader the referent may be gone, and a reader that forgets +renders a blank instead of a designed empty state. Measured on real data: 0 of 5 workflow chat +pointers resolved, and 8 of 10 app workspaces had no record at all. + +The rule: a field named ``*_id`` / ``*_ids`` on a class that inherits from pydantic ``BaseModel`` +must be declared in ``backend/config/entity_references.py``, which names the entity it points at +and the store that resolves it. A model's own primary key is spelled ``id`` and so never matches. +Pre-existing fields are grandfathered per FIELD (not per file) in the ``dangling-refs`` exception +list, keyed ``::.``, so a new field in an old model is still caught. + +The registry is checked back: an entry for a field that no longer exists is an error, and so is a +store whose lookup function has been renamed away. A registry nobody verifies is a registry that +rots. + +Scoped to ``backend/`` Python, like checks/classes.py. One AST pass, no imports of backend code +(CI lints with a bare interpreter that has no pydantic). +""" + +from __future__ import annotations + +import ast +from pathlib import Path +from typing import Dict, List, Optional, Set, Tuple + +from . import CheckError, is_excepted, is_excluded, is_lintignored + +RULE = "dangling-refs" +REGISTRY_REL = "backend/config/entity_references.py" +REFERENCES_NAME = "CROSS_ENTITY_REFERENCES" +STORES_NAME = "ENTITY_STORES" +KIND_ENUM_NAME = "EntityKind" + +# (dotted module, model name, field name) +FieldKey = Tuple[str, str, str] + + +def p_dotted(rel: str) -> str: + """``backend/apps/foo/models.py`` -> ``backend.apps.foo.models``.""" + parts = list(Path(rel).with_suffix("").parts) + if parts and parts[-1] == "__init__": + parts.pop() + return ".".join(parts) + + +def p_const_str(node: Optional[ast.AST]) -> Optional[str]: + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return node.value + return None + + +def p_attr_name(node: Optional[ast.AST]) -> Optional[str]: + """``EntityKind.SESSION`` -> ``SESSION``.""" + return node.attr if isinstance(node, ast.Attribute) else None + + +def p_base_names(node: ast.ClassDef) -> List[str]: + names: List[str] = [] + for base in node.bases: + target: ast.AST = base.value if isinstance(base, ast.Subscript) else base + if isinstance(target, ast.Name): + names.append(target.id) + elif isinstance(target, ast.Attribute): + names.append(target.attr) + return names + + +def p_is_class_var(node: ast.AnnAssign) -> bool: + annotation: ast.AST = node.annotation + if isinstance(annotation, ast.Subscript): + annotation = annotation.value + if isinstance(annotation, ast.Name): + return annotation.id == "ClassVar" + return isinstance(annotation, ast.Attribute) and annotation.attr == "ClassVar" + + +class BackendIndex: + """Everything one AST pass over ``backend/`` needs to hand the rest of the check.""" + + def __init__(self) -> None: + self.bases: Dict[str, List[str]] = {} + self.top_level: Dict[str, Set[str]] = {} + # (rel path, dotted module, model, field, lineno, col) + self.id_fields: List[Tuple[str, str, str, str, int, int]] = [] + self.p_model_cache: Dict[str, bool] = {} + + def is_model(self, name: str, seen: Optional[Set[str]] = None) -> bool: + if name == "BaseModel": + return True + cached = self.p_model_cache.get(name) + if cached is not None: + return cached + seen = seen if seen is not None else set() + if name in seen: + return False + seen.add(name) + result = any(self.is_model(b, seen) for b in self.bases.get(name, [])) + self.p_model_cache[name] = result + return result + + +def p_index_file(index: BackendIndex, tree: ast.Module, rel: str) -> None: + module = p_dotted(rel) + names: Set[str] = set() + for node in tree.body: + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): + names.add(node.name) + elif isinstance(node, ast.Assign): + names.update(t.id for t in node.targets if isinstance(t, ast.Name)) + elif isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name): + names.add(node.target.id) + index.top_level[module] = names + + for node in ast.walk(tree): + if not isinstance(node, ast.ClassDef): + continue + index.bases.setdefault(node.name, []).extend(p_base_names(node)) + for stmt in node.body: + if not isinstance(stmt, ast.AnnAssign) or not isinstance(stmt.target, ast.Name): + continue + field = stmt.target.id + if not (field.endswith("_id") or field.endswith("_ids")) or p_is_class_var(stmt): + continue + index.id_fields.append((rel, module, node.name, field, stmt.lineno, stmt.col_offset)) + + +class Registry: + """The parsed contents of entity_references.py, as data.""" + + def __init__(self) -> None: + self.kinds: Set[str] = set() + self.stores: Dict[str, Tuple[str, str, int]] = {} + self.references: Dict[FieldKey, Tuple[str, int]] = {} + + +def p_list_literal(tree: ast.Module, name: str) -> Optional[List[ast.expr]]: + for node in tree.body: + if isinstance(node, ast.Assign): + targets: List[ast.expr] = list(node.targets) + elif isinstance(node, ast.AnnAssign): + targets = [node.target] + else: + continue + if any(isinstance(t, ast.Name) and t.id == name for t in targets): + return list(node.value.elts) if isinstance(node.value, ast.List) else None + return None + + +def p_parse_registry(path: Path) -> Registry: + try: + tree = ast.parse(path.read_text(), filename=REGISTRY_REL) + except (OSError, SyntaxError) as exc: + raise CheckError(f"cannot read the reference registry at {REGISTRY_REL}: {exc}") from exc + + registry = Registry() + for node in tree.body: + if isinstance(node, ast.ClassDef) and node.name == KIND_ENUM_NAME: + for stmt in node.body: + if isinstance(stmt, ast.Assign): + registry.kinds.update(t.id for t in stmt.targets if isinstance(t, ast.Name)) + + stores = p_list_literal(tree, STORES_NAME) + references = p_list_literal(tree, REFERENCES_NAME) + if stores is None or references is None: + raise CheckError(f"{REGISTRY_REL} must assign {STORES_NAME} and {REFERENCES_NAME} to list literals") + + for element in stores: + if not isinstance(element, ast.Call): + continue + kw = {k.arg: k.value for k in element.keywords if k.arg} + kind = p_attr_name(kw.get("kind")) + module = p_const_str(kw.get("module")) + lookup = p_const_str(kw.get("lookup")) + if kind and module and lookup: + registry.stores[kind] = (module, lookup, element.lineno) + + for element in references: + if not isinstance(element, ast.Call): + continue + kw = {k.arg: k.value for k in element.keywords if k.arg} + module = p_const_str(kw.get("module")) + model = p_const_str(kw.get("model")) + field = p_const_str(kw.get("field")) + target = p_attr_name(kw.get("target")) + if module and model and field and target: + registry.references[(module, model, field)] = (target, element.lineno) + return registry + + +def p_registry_error(lineno: int, message: str) -> str: + return f"{REGISTRY_REL}:{lineno}:1: error: [{RULE}] {message}" + + +def p_check_registry(registry: Registry, index: BackendIndex) -> List[str]: + """Fail loudly when the registry has drifted from the code it describes.""" + errors: List[str] = [] + for kind, (module, lookup, lineno) in sorted(registry.stores.items()): + if kind not in registry.kinds: + errors.append(p_registry_error(lineno, f"store kind '{kind}' is not a member of {KIND_ENUM_NAME}")) + elif lookup not in index.top_level.get(module, set()): + errors.append(p_registry_error(lineno, f"store for '{kind}' points at {module}.{lookup}, which no longer exists")) + + declared = {(module, model, field) for _, module, model, field, _, _ in index.id_fields} + for (module, model, field), (target, lineno) in sorted(registry.references.items()): + if target not in registry.stores: + errors.append(p_registry_error(lineno, f"'{model}.{field}' targets '{target}', which has no {STORES_NAME} row")) + if (module, model, field) not in declared: + errors.append(p_registry_error(lineno, f"'{model}.{field}' in {module} matches no field on a backend model; it was renamed or removed")) + return errors + + +def run_dangling_refs_check( + root: Path, + exceptions: Dict[str, List[str]], + excludes: List[str], + ignores: Optional[Dict[Path, Set[str]]] = None, +) -> List[str]: + """Flag ``*_id`` / ``*_ids`` fields on backend models that declare no target entity.""" + backend = root / "backend" + if not backend.is_dir(): + return [] + + index = BackendIndex() + for pyfile in sorted(backend.rglob("*.py")): + if is_excluded(pyfile, root, excludes): + continue + # Forward slashes even on Windows, so the config's exception globs match there too. + rel = pyfile.relative_to(root).as_posix() + try: + tree = ast.parse(pyfile.read_text(), filename=rel) + except (OSError, SyntaxError): + continue + p_index_file(index, tree, rel) + + registry = p_parse_registry(root / REGISTRY_REL) + errors = p_check_registry(registry, index) + + for rel, module, model, field, lineno, col in index.id_fields: + if not index.is_model(model): + continue + if (module, model, field) in registry.references: + continue + if is_excepted(f"{rel}::{model}.{field}", RULE, exceptions): + continue + if ignores and is_lintignored(root / rel, root, RULE, ignores): + continue + errors.append( + f"{rel}:{lineno}:{col + 1}: error: [{RULE}] cross-entity id field " + f"'{model}.{field}' declares no target entity; add an EntityReference for it in " + f"{REGISTRY_REL}, or grandfather '{rel}::{model}.{field}' in the dangling-refs exceptions" + ) + return errors diff --git a/linter/config/config.json b/linter/config/config.json index b26b5225..88b64536 100644 --- a/linter/config/config.json +++ b/linter/config/config.json @@ -11,6 +11,7 @@ "classes": false, "no-underscore-names": true, "p-private": true, + "dangling-refs": true, "ruff": true, "pyright": true }, @@ -19,10 +20,11 @@ "eslint-knip": "Node tooling deferred to a later pass.", "classes": "Placeholder check, not wired up. endpoints: orphaned-endpoint triage deferred.", "max-file-lines-exceptions": "Grandfather list of pre-existing >300-line files (existing debt, not new). Paths updated after the folder-tree restructure moved several of them. The two manager/prompt/* entries are from the agent_manager decomposition: prompt_context.py aggregates the system-prompt context builders and attachments.py is one cohesive 230-line attachment resolver; both are single-responsibility and a few lines over, not splittable without an artificial seam.", - "max-folder-items-exceptions": "Exact-path allow for folders intentionally over the cap. The rule trips at >7 (7 items is fine, the 8th tips it), so only genuinely 8+ folders are listed. backend/ and backend/apps are FastAPI feature-package registries (each child is an app mounted in main.py); agents/ aggregates agent subsystems; agents/manager/ is the agent_manager god-object decomposition (cohesive AgentManager mixins + standalone run helpers + the streaming/permissions/prompt/session subtrees), conventionally flat like agents/ and core/ since its standalone helpers are heterogeneous and don't group cleanly; agents/manager/streaming and agents/manager/session are flat peer collections of one-module-per-concern handlers; core/, tools_lib/, tests/ are conventionally flat. Frontend: app/pages is the page registry, AgentChat/ChatInput/Settings-sections/Onboarding are organizational parents, and shared/state (Redux slices) plus hooks/steps/mcp-cards/Views are flat peer collections. scripts/, electron/, linter/checks/ are flat tool dirs. These replaced blanket .lintignore-max-folder-items sentinels (backend, frontend, scripts, electron, linter/checks) so the rule still catches NEW unplanned bloat everywhere else. Kept as whole-subtree sentinels on purpose: debugger/ (self-contained injected sub-tool with its own Vite GUI), webapp_template (Vite scaffold payload), and vendored mcp-bundles. 2026-07 desktop-shell additions: Dashboard canvas/cards/desktop + hooks/interaction + hooks/lifecycle, AgentChat bubbles/tool-ui, and shared/styles are flat peer collections (one component or hook per concern) that crossed 7 as the redesign surface grew. frontend/src/toolui carries a whole-subtree .lintignore: vendored tool-ui component library (pierre), same treatment as mcp-bundles.", + "max-folder-items-exceptions": "Exact-path allow for folders intentionally over the cap. The rule trips at >7 (7 items is fine, the 8th tips it), so only genuinely 8+ folders are listed. backend/ and backend/apps are FastAPI feature-package registries (each child is an app mounted in main.py); agents/ aggregates agent subsystems; agents/manager/ is the agent_manager god-object decomposition (cohesive AgentManager mixins + standalone run helpers + the streaming/permissions/prompt/session subtrees), conventionally flat like agents/ and core/ since its standalone helpers are heterogeneous and don't group cleanly; agents/manager/streaming and agents/manager/session are flat peer collections of one-module-per-concern handlers; core/, tools_lib/, tests/ are conventionally flat. Frontend: app/pages is the page registry, AgentChat/ChatInput/Settings-sections/Onboarding are organizational parents, and shared/state (Redux slices) plus hooks/steps/mcp-cards/Views are flat peer collections. scripts/, electron/, linter/checks/ are flat tool dirs. These replaced blanket .lintignore-max-folder-items sentinels (backend, frontend, scripts, electron, linter/checks) so the rule still catches NEW unplanned bloat everywhere else. Kept as whole-subtree sentinels on purpose: debugger/ (self-contained injected sub-tool with its own Vite GUI), webapp_template (Vite scaffold payload), and vendored mcp-bundles. 2026-07 desktop-shell additions: Dashboard canvas/cards/desktop + hooks/interaction + hooks/lifecycle, AgentChat bubbles/tool-ui, and shared/styles are flat peer collections (one component or hook per concern) that crossed 7 as the redesign surface grew. frontend/src/toolui carries a whole-subtree .lintignore: vendored tool-ui component library (pierre), same treatment as mcp-bundles. openswarm-edge/app is the edge's flat one-module-per-concern set (routing, bundles, inject, ratelimit, sandbox, and the vendored code_safety gate); it crossed 7 when the sandbox's static gate was split out to mirror the desktop file byte for byte. AgentChat/parsing joined when the narration/deliverable classifier landed: it is the same flat one-module-per-parser collection as the rest of that subtree.", "import-cycles": "Flags RUNTIME circular imports only (SCC>1). Skips type-only imports (import type / export type) and dynamic import() since neither runs at module init, which is why the idiomatic Redux store<->hooks type cycle is not flagged. Frontend alias resolution comes from import-cycle-aliases. Zero cycles today; the check keeps it that way.", "ruff + pyright": "Ported from Haik's linter (haik/feat/ingest). ruff is narrowed to F401/F811/F841 (unused imports/redefs/locals) and intentionally DROPS Haik's ARG001/ARG002 (unused args): our SDK-callback signatures require unused params (can_use_tool/pre_tool_hook take a `context` they don't use) and we ban the `_unused` prefix, so ARG is noise here. pyright runs Haik's existence-only config (typeCheckingMode off) with reportAttributeAccessIssue ENABLED: the AgentManager behavior classes now inherit a typing-only AgentManagerProtocol base (manager/AgentManagerProtocol.py) that declares the composed __init__ state + cross-class methods, so the checker sees self.sessions etc. from inside a mixin. pyright caught real bugs: a dangling `_conns` ref + TWO broken lazy imports (`_load_all`/`_load` from outputs.py, renamed to load_all/load in workspace_io but the import sites weren't updated \u2014 App Builder workspace seeding/name-sync was silently failing in a try/except). The one grandfathered SURFACE file (handle_assistant_message) is the SDK-optional try/except-import boundary (TextBlock=object fallback defeats isinstance narrowing). Both grandfather pre-existing debt by file; the refactor surface is clean. Requires `ruff` + `pyright` on PATH (added to requirements-dev.txt); pyright's config expects the venv at backend/.venv.", - "no-underscore-names + p-private": "Convention checks ported verbatim from Haik's linter (haik/feat/ingest): no-underscore-names bans leading-underscore names (a dead-code-tooling blind spot; use p_ for private), p-private enforces that p_-prefixed names are accessed only inside their owning file/class (cross-file/class use means the name should be public). Backend Python only. The exception lists grandfather pre-existing debt that landed with the workflows/analytics forward-ports (eric's 'don't mass-migrate untouched files' rule); the agent_manager refactor surface is clean. NOTE: Haik's full linter (his branch also adds pyright + ruff and runs a different enabled set) should eventually supersede this; these two were lifted to enforce the p_ conventions on eric/dev now. browser_cookies.py and its Windows round-trip test are excepted for `_fields_` only: a ctypes.Structure protocol name required by the ctypes metaclass, not our naming." + "no-underscore-names + p-private": "Convention checks ported verbatim from Haik's linter (haik/feat/ingest): no-underscore-names bans leading-underscore names (a dead-code-tooling blind spot; use p_ for private), p-private enforces that p_-prefixed names are accessed only inside their owning file/class (cross-file/class use means the name should be public). Backend Python only. The exception lists grandfather pre-existing debt that landed with the workflows/analytics forward-ports (eric's 'don't mass-migrate untouched files' rule); the agent_manager refactor surface is clean. NOTE: Haik's full linter (his branch also adds pyright + ruff and runs a different enabled set) should eventually supersede this; these two were lifted to enforce the p_ conventions on eric/dev now. browser_cookies.py and its Windows round-trip test are excepted for `_fields_` only: a ctypes.Structure protocol name required by the ctypes metaclass, not our naming.", + "dangling-refs": "Every *_id / *_ids field on a backend pydantic model must name the entity it points at, in backend/config/entity_references.py. A model's own primary key is spelled `id`, which never matches the suffix, and neither do words that merely END in id (uuid, grid, valid) since the underscore is required. 42 of the 74 existing fields are declared in the registry (sessions, dashboards, workflows, workflow runs, apps/outputs, workspaces); the 32 listed here are grandfathered debt, and the entry is keyed ::. rather than by file ON PURPOSE, so a NEW id field added to an already-listed model is still caught (a file glob would exempt workflows/models.py forever, which is exactly where the next dangling pointer lands). The grandfathered set is what does not resolve against a store: renderer-owned live objects (browser_id, selected_browser_ids, selected_setting_ids), ids internal to a single record (active_branch_id, msg_id, parent_id, fork_point_message_id, compacted_through_msg_id), external protocol ids we do not own (sdk_session_id, client_message_id, connection_id, installation_id, user_id), telemetry echoes (analytics bridges), and the skill-registry / .swarm-bundle entities that have no backend store module yet. Move an entry out of this list and into the registry when its entity gets one. backend/tests/*::* is blanket-exempt: a test-local model is not a persisted entity. The registry is checked back both ways, so an entry for a deleted field, or a store whose lookup function was renamed, is an error too." }, "rules": { "max-file-lines": 300, @@ -151,46 +153,50 @@ "frontend/src/shared/browserCommandHandler.ts", "frontend/src/shared/state/agentsSlice.ts", "frontend/src/shared/state/dashboardLayoutSlice.ts", - "frontend/src/shared/ws/WebSocketManager.ts" + "frontend/src/shared/ws/WebSocketManager.ts", + "backend/apps/workflows/cloud/client.py" ], "max-folder-items": [ "backend", "backend/apps", - "backend/apps/reddit_mcp_shim", - "backend/apps/x_mcp_shim", - "backend/apps/tiktok_mcp_shim", "backend/apps/agents", "backend/apps/agents/core", "backend/apps/agents/manager", "backend/apps/agents/manager/session", "backend/apps/agents/manager/streaming", "backend/apps/outputs", - "backend/apps/tools_lib", + "backend/apps/reddit_mcp_shim", "backend/apps/service", + "backend/apps/tiktok_mcp_shim", + "backend/apps/tools_lib", + "backend/apps/x_mcp_shim", "backend/tests", - "frontend/src/shared", - "frontend/src/shared/state", - "frontend/src/shared/hooks", + "electron", + "frontend/src/app/components/Onboarding", + "frontend/src/app/components/Onboarding/steps", "frontend/src/app/pages", "frontend/src/app/pages/AgentChat", "frontend/src/app/pages/AgentChat/ChatInput", "frontend/src/app/pages/AgentChat/ChatInput/hooks", - "frontend/src/app/pages/AgentChat/mcp-cards", "frontend/src/app/pages/AgentChat/bubbles", + "frontend/src/app/pages/AgentChat/mcp-cards", + "frontend/src/app/pages/AgentChat/parsing", "frontend/src/app/pages/AgentChat/tool-ui", "frontend/src/app/pages/Dashboard/canvas", "frontend/src/app/pages/Dashboard/cards", "frontend/src/app/pages/Dashboard/desktop", "frontend/src/app/pages/Dashboard/hooks/interaction", "frontend/src/app/pages/Dashboard/hooks/lifecycle", - "frontend/src/shared/styles", "frontend/src/app/pages/Settings/sections", "frontend/src/app/pages/Views", - "frontend/src/app/components/Onboarding", - "frontend/src/app/components/Onboarding/steps", + "frontend/src/shared", + "frontend/src/shared/hooks", + "frontend/src/shared/state", + "frontend/src/shared/styles", + "linter/checks", + "openswarm-edge/app", "scripts", - "electron", - "linter/checks" + "backend/apps/workflows/cloud" ], "no-nested-imports": [], "import-cycles": [], @@ -231,6 +237,41 @@ "backend/tests/test_workflows_semantics.py", "backend/tests/test_workflows_storage.py" ], + "dangling-refs": [ + "backend/apps/agents/core/models.py::AgentSession.active_branch_id", + "backend/apps/agents/core/models.py::AgentSession.browser_id", + "backend/apps/agents/core/models.py::AgentSession.compacted_through_msg_id", + "backend/apps/agents/core/models.py::AgentSession.sdk_session_id", + "backend/apps/agents/core/models.py::ApprovalResponse.request_id", + "backend/apps/agents/core/models.py::Message.branch_id", + "backend/apps/agents/core/models.py::Message.client_message_id", + "backend/apps/agents/core/models.py::Message.parent_id", + "backend/apps/agents/core/models.py::MessageBranch.fork_point_message_id", + "backend/apps/agents/core/models.py::MessageBranch.parent_branch_id", + "backend/apps/agents/manager/Messaging.py::QueuedMessage.client_message_id", + "backend/apps/agents/manager/Messaging.py::QueuedMessage.selected_browser_ids", + "backend/apps/agents/manager/Messaging.py::QueuedMessage.selected_setting_ids", + "backend/apps/agents/manager/permissions/workflow_approval.py::WorkflowApprovalMemory.current_step_id", + "backend/apps/agents/manager/streaming/PartialReply.py::PartialReply.branch_id", + "backend/apps/agents/manager/streaming/PartialReply.py::PartialReply.msg_id", + "backend/apps/agents/manager/streaming/state.py::ThinkingState.msg_id", + "backend/apps/agents/manager/streaming/state.py::TurnState.stream_text_msg_id", + "backend/apps/dashboards/models.py::BrowserCardPosition.browser_id", + "backend/apps/nine_router/credential_store.py::ProviderCredential.connection_id", + "backend/apps/outputs/models.py::OutputVersion.parent_id", + "backend/apps/service/analytics/agent_bridge.py::BroadcastMessage.branch_id", + "backend/apps/service/analytics/agent_bridge.py::BroadcastMessage.parent_id", + "backend/apps/service/analytics/frontend_bridge.py::FrontendEventProps.dashboard_id", + "backend/apps/service/analytics/frontend_bridge.py::FrontendEventProps.step_id", + "backend/apps/settings/models.py::AppSettings.installation_id", + "backend/apps/settings/models.py::AppSettings.user_id", + "backend/apps/skill_registry/skill_registry.py::p_InstallRequest.skill_id", + "backend/apps/skill_registry/skill_registry.py::p_UpdateRequest.skill_id", + "backend/apps/swarm/models.py::EntityRef.bundle_id", + "backend/apps/swarm/models.py::ImportCommitResponse.root_id", + "backend/apps/swarm/models.py::Manifest.bundle_id", + "backend/tests/*::*" + ], "ruff": [ "backend/apps/agents/agents.py", "backend/apps/agents/browser/browser_agent.py", diff --git a/linter/config/vulture_whitelist.py b/linter/config/vulture_whitelist.py index 91ddb509..5de8dfcc 100644 --- a/linter/config/vulture_whitelist.py +++ b/linter/config/vulture_whitelist.py @@ -72,3 +72,9 @@ resolve_forced_tools used_llm usage_summary last_run_at + +# config/entity_references.py: the cross-entity reference registry. Its consumer is +# the dangling-refs linter check, which reads the file as data rather than importing +# it, so vulture sees two module-level tables nobody touches. +ENTITY_STORES +CROSS_ENTITY_REFERENCES diff --git a/linter/lint.py b/linter/lint.py index ccd605f2..b318d26c 100644 --- a/linter/lint.py +++ b/linter/lint.py @@ -18,6 +18,7 @@ from checks.knip import run_knip from checks.endpoints import run_endpoint_check from checks.classes import run_class_check from checks.cycles import run_cycle_check +from checks.dangling_refs import run_dangling_refs_check from checks.no_underscore_names import run_underscore_check from checks.p_private import run_p_private_check from checks.ruff import run_ruff @@ -33,7 +34,7 @@ def load_config() -> dict[str, Any]: return json.load(f) -def run_checks(root: Path) -> tuple[list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str]]: +def run_checks(root: Path) -> tuple[list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str], list[str]]: config = load_config() enabled: dict[str, bool] = config.get("enabled", {}) rules: dict[str, int] = config["rules"] @@ -114,6 +115,14 @@ def run_checks(root: Path) -> tuple[list[str], list[str], list[str], list[str], underscore_errors = run_underscore_check(root, exceptions, excludes, ignores) if enabled.get("no-underscore-names", False) else [] p_private_errors = run_p_private_check(root, exceptions, excludes, ignores) if enabled.get("p-private", False) else [] + # A dangling reference becomes a linter error at declaration time, not a blank card six months later. + dangling_ref_errors: list[str] = [] + if enabled.get("dangling-refs", False): + try: + dangling_ref_errors = run_dangling_refs_check(root, exceptions, excludes, ignores) + except CheckError as e: + dangling_ref_errors = [f"dangling-refs: check could not run: {e.reason}"] + # ruff (scoped dead-code codes) + pyright (existence errors), also from Haik's # linter. Both shell out to a tool, so a missing tool / timeout raises CheckError # and is surfaced as a loud error rather than a silently-clean empty result. @@ -130,7 +139,7 @@ def run_checks(root: Path) -> tuple[list[str], list[str], list[str], list[str], except CheckError as e: pyright_errors = [f"pyright: check could not run: {e.reason}"] - return sorted(structural_errors), sorted(vulture_errors), sorted(eslint_errors), sorted(knip_errors), sorted(endpoint_errors), sorted(class_errors), sorted(cycle_errors), sorted(underscore_errors), sorted(p_private_errors), sorted(ruff_errors), sorted(pyright_errors) + return sorted(structural_errors), sorted(vulture_errors), sorted(eslint_errors), sorted(knip_errors), sorted(endpoint_errors), sorted(class_errors), sorted(cycle_errors), sorted(underscore_errors), sorted(p_private_errors), sorted(dangling_ref_errors), sorted(ruff_errors), sorted(pyright_errors) def _print_section(name: str, errors: list[str]) -> None: @@ -145,8 +154,8 @@ def print_results( eslint_errors: list[str], knip_errors: list[str], endpoint_errors: list[str], class_errors: list[str], cycle_errors: list[str], underscore_errors: list[str], - p_private_errors: list[str], ruff_errors: list[str], - pyright_errors: list[str], + p_private_errors: list[str], dangling_ref_errors: list[str], + ruff_errors: list[str], pyright_errors: list[str], ) -> None: _print_section("structural", structural_errors) _print_section("vulture", vulture_errors) @@ -157,6 +166,7 @@ def print_results( _print_section("import-cycles", cycle_errors) _print_section("no-underscore-names", underscore_errors) _print_section("p-private", p_private_errors) + _print_section("dangling-refs", dangling_ref_errors) _print_section("ruff", ruff_errors) _print_section("pyright", pyright_errors) diff --git a/openswarm-edge/app/code_safety.py b/openswarm-edge/app/code_safety.py new file mode 100644 index 00000000..d05364ff --- /dev/null +++ b/openswarm-edge/app/code_safety.py @@ -0,0 +1,223 @@ +"""Static safety gate for published apps' backend.py compute. + +VENDORED, verbatim below this docstring, from backend/apps/outputs/code_safety.py +(the desktop App Builder gate). test_edge_sandbox_mirrors_backend.py in the +backend suite fails if the two drift, because a gate that is only tightened on +the desktop leaves the internet-facing copy open. +""" + +import ast +import importlib +import types +from typing import Dict, List, Optional, Set + +# Modules backend code is allowed to import. This is not an OS-level jail, so keep the list to "data shaping" libraries; no I/O, no networking, no subprocess. It pairs with cwd=tempdir + minimal env so the blast radius stays small. +ALLOWED_MODULES = frozenset({ + "json", "math", "re", "datetime", "collections", "itertools", + "functools", "statistics", "decimal", "fractions", "random", + "string", "textwrap", "unicodedata", "csv", "copy", "enum", + "dataclasses", "typing", "abc", "numbers", "uuid", "hashlib", + "base64", "binascii", "operator", "heapq", "bisect", "array", +}) + +# Builtins that punch holes through the allowlist or do I/O. Most are also deleted off `builtins` inside the subprocess; exec/compile/__import__ can't be, because the import machinery runs on them. +P_BLOCKED_BUILTINS = frozenset({ + "exec", "eval", "compile", "__import__", "open", "input", + "breakpoint", "exit", "quit", +}) + +# These hand back a live namespace dict, which is every blocked name again through a different door. Warned about but never deleted: library code calls them constantly, and a scrubbed `builtins` would break `import csv` itself. +P_NAMESPACE_BUILTINS = frozenset({"vars", "globals", "locals"}) + +# Spell an attribute as a string and the AST can't read it, so these are allowed only with a plain literal that would have passed written out longhand. +P_DYNAMIC_ATTR_BUILTINS = frozenset({"getattr", "setattr", "delattr"}) + +# Modules the executor preamble binds into the user's namespace with no import. `json` stays usable (it is allowlisted anyway); these three were the free handles that made the whole allowlist decorative. +P_SANDBOX_MODULE_HANDLES = frozenset({"sys", "io", "builtins"}) + +# Dunders that hold a reference to nothing at all, and `if __name__ == "__main__"` is far too common to punish. +P_INERT_DUNDERS = frozenset({"__name__", "__file__", "__doc__"}) + + +class UnsafeCodeError(Exception): + """Raised when the static gate rejects user-supplied backend code.""" + + +def p_is_dunder(name: str) -> bool: + return len(name) > 4 and name.startswith("__") and name.endswith("__") + + +def p_dotted_chain(node: ast.expr) -> Optional[List[str]]: + """['json', 'codecs', 'open'] for `json.codecs.open`; None when the chain + doesn't start at a plain name.""" + parts: List[str] = [] + current: ast.expr = node + while isinstance(current, ast.Attribute): + parts.append(current.attr) + current = current.value + if not isinstance(current, ast.Name): + return None + parts.append(current.id) + parts.reverse() + return parts + + +def p_resolved_module(chain: List[str], aliases: Dict[str, str]) -> Optional[str]: + """The name of the module an attribute chain resolves to, or None if it + resolves to something that isn't a module. + + `json.codecs` is a module and `datetime.time` is a class, and only the live + object knows which; matching attribute names against a list of module names + would flag both. So resolve against the module actually imported. Safe to + import here because `aliases` only ever holds allowlisted stdlib roots. + """ + root = aliases.get(chain[0]) + if root is None: + return None + try: + value: object = importlib.import_module(root) + for attr in chain[1:]: + value = getattr(value, attr) + except Exception: + return None + return value.__name__ if isinstance(value, types.ModuleType) else None + + +def p_module_aliases(tree: ast.Module) -> Dict[str, str]: + """Local name -> allowlisted module it holds. Plain assignment counts, so + `m = json` doesn't launder `m.codecs` past the chain check. Modules outside + the allowlist are never recorded, which is what keeps the resolver above + from importing anything a hostile file names.""" + aliases: Dict[str, str] = {"json": "json"} + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + root = alias.name.split(".")[0] + if root in ALLOWED_MODULES: + aliases[alias.asname or root] = root + assigns = [n for n in ast.walk(tree) if isinstance(n, ast.Assign)] + for _ in range(len(assigns)): + before = len(aliases) + for node in assigns: + if len(node.targets) != 1 or not isinstance(node.targets[0], ast.Name): + continue + chain = p_dotted_chain(node.value) + resolved = p_resolved_module(chain, aliases) if chain else None + if resolved and resolved.split(".")[0] in ALLOWED_MODULES: + aliases[node.targets[0].id] = resolved + if len(aliases) == before: + break + return aliases + + +def p_dunder_warning(attr: str, prefix: str) -> Optional[str]: + if p_is_dunder(attr) and attr not in P_INERT_DUNDERS: + return f"Uses dunder '{prefix}{attr}', which walks the object graph past the allowlist" + return None + + +def p_attribute_warning(chain: List[str], aliases: Dict[str, str]) -> Optional[str]: + """The verdict on one resolved attribute chain, dunders first.""" + for attr in chain[1:]: + dunder = p_dunder_warning(attr, ".") + if dunder: + return dunder + for depth in range(2, len(chain) + 1): + reached = p_resolved_module(chain[:depth], aliases) + if reached and reached.split(".")[0] not in ALLOWED_MODULES: + return f"Reaches module '{reached}' via '{'.'.join(chain[:depth])}' (outside the safe-data-shaping allowlist)" + return None + + +def p_call_warning(node: ast.Call, aliases: Dict[str, str]) -> Optional[str]: + if not isinstance(node.func, ast.Name): + return None + name = node.func.id + if name in P_BLOCKED_BUILTINS: + return f"Calls builtin '{name}()' which can escape the sandbox" + if name in P_NAMESPACE_BUILTINS: + return f"Calls '{name}()', which hands back the sandbox's own namespace" + if name not in P_DYNAMIC_ATTR_BUILTINS: + return None + attr = node.args[1] if len(node.args) > 1 else None + if not isinstance(attr, ast.Constant) or not isinstance(attr.value, str): + return f"Computes an attribute name for '{name}()', which can spell any escape as a string" + base = p_dotted_chain(node.args[0]) + if base is None: + return p_dunder_warning(attr.value, ".") + return p_attribute_warning(base + [attr.value], aliases) + + +def p_star_import_warning(module: str, aliases: Dict[str, str]) -> Optional[str]: + """`from json import *` binds whatever json's __all__ names, which is a + short list of functions today but is not ours to assume.""" + try: + imported = importlib.import_module(module) + except Exception: + return None + names = getattr(imported, "__all__", None) or [n for n in dir(imported) if not n.startswith("_")] + for name in names: + warning = p_attribute_warning([module, str(name)], aliases) + if warning: + return warning + return None + + +def get_code_warnings(code: str) -> List[str]: + """Return human-readable warnings for every static risk, without raising. + + `/api/outputs/execute` surfaces these in the run dialog, so an Output that + genuinely needs `pandas` gets a "review and click Run Anyway" affordance + instead of a silent 500. An empty list is what buys the no-prompt auto-run + path, so anything that could reach past the allowlist has to land in it. A + syntax error is reported as a warning rather than raised, so the dialog can + show it next to the code. + """ + try: + tree = ast.parse(code) + except SyntaxError as e: + return [f"Syntax error: {e}"] + + aliases = p_module_aliases(tree) + warnings: List[str] = [] + seen: Set[str] = set() + + def note(msg: Optional[str]) -> None: + if msg and msg not in seen: + seen.add(msg) + warnings.append(msg) + + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + if alias.name.split(".")[0] not in ALLOWED_MODULES: + note(f"Imports '{alias.name}' (outside the safe-data-shaping allowlist)") + elif isinstance(node, ast.ImportFrom): + root = (node.module or "").split(".")[0] + if root not in ALLOWED_MODULES: + note(f"Imports from '{node.module}' (outside the safe-data-shaping allowlist)") + continue + for alias in node.names: + if alias.name == "*": + note(p_star_import_warning(root, aliases)) + else: + note(p_attribute_warning([root, alias.name], aliases)) + elif isinstance(node, ast.Call): + note(p_call_warning(node, aliases)) + elif isinstance(node, ast.Attribute): + chain = p_dotted_chain(node) + note(p_attribute_warning(chain, aliases) if chain else p_dunder_warning(node.attr, ".")) + elif isinstance(node, ast.Name): + if node.id in P_SANDBOX_MODULE_HANDLES: + note(f"References '{node.id}', a live module the sandbox binds but the allowlist withholds") + else: + note(p_dunder_warning(node.id, "")) + return warnings + + +def validate_code_safety(code: str) -> None: + """Raise UnsafeCodeError on the first static risk. The strict wrapper around + get_code_warnings, for callers with no user to ask.""" + warnings = get_code_warnings(code) + if warnings: + raise UnsafeCodeError(warnings[0]) diff --git a/openswarm-edge/app/main.py b/openswarm-edge/app/main.py index 1c0cecd2..fa819cf3 100644 --- a/openswarm-edge/app/main.py +++ b/openswarm-edge/app/main.py @@ -20,7 +20,8 @@ from .bundles import get_bundle, resolve_file from .fallback import apex_page, not_found_page from .inject import inject_runtime from .ratelimit import RateLimiter -from .sandbox import UnsafeCodeError, run_backend +from .code_safety import UnsafeCodeError +from .sandbox import run_backend APPS_BASE_DOMAIN = os.environ.get("APPS_BASE_DOMAIN", "openswarm.host") # The metered-LLM call goes to the cloud over Fly's PRIVATE 6PN mesh (encrypted, diff --git a/openswarm-edge/app/sandbox.py b/openswarm-edge/app/sandbox.py index c7a0947e..585c6830 100644 --- a/openswarm-edge/app/sandbox.py +++ b/openswarm-edge/app/sandbox.py @@ -1,13 +1,13 @@ """Sandboxed Python runner for published apps' backend.py compute. VENDORED from backend/apps/outputs/executor.py (the desktop App Builder runtime). -Keep the allow/deny lists + the subprocess hardening in sync with that file; this -is the same data-shaping sandbox, just running in the edge instead of on the -desktop. Pure compute only: no network, no disk, no subprocess, no secrets. Safe -to run multi-tenant on one machine because nothing here can reach shared state.""" +Keep the subprocess hardening in sync with that file; this is the same +data-shaping sandbox, just running in the edge instead of on the desktop. The +static gate it runs on every call lives in the vendored app/code_safety.py. Pure +compute only: no network, no disk, no subprocess, no secrets. Safe to run +multi-tenant on one machine because nothing here can reach shared state.""" from __future__ import annotations -import ast import asyncio import json import os @@ -15,46 +15,10 @@ import sys import tempfile from dataclasses import dataclass +from app.code_safety import ALLOWED_MODULES, validate_code_safety + TIMEOUT_SECONDS = 30 -_ALLOWED_MODULES = frozenset({ - "json", "math", "re", "datetime", "collections", "itertools", - "functools", "statistics", "decimal", "fractions", "random", - "string", "textwrap", "unicodedata", "csv", "copy", "enum", - "dataclasses", "typing", "abc", "numbers", "uuid", "hashlib", - "base64", "binascii", "operator", "heapq", "bisect", "array", -}) - -_BLOCKED_BUILTINS = frozenset({ - "exec", "eval", "compile", "__import__", "open", "input", - "breakpoint", "exit", "quit", -}) - - -class UnsafeCodeError(Exception): - """AST validation rejected the backend code.""" - - -def validate_code_safety(code: str) -> None: - """Raise UnsafeCodeError on the first AST-visible risk. Published apps are - vetted at publish time, but we re-check here: the edge never trusts that the - bundle in storage matches what was scanned.""" - try: - tree = ast.parse(code) - except SyntaxError as e: - raise UnsafeCodeError(f"Syntax error: {e}") - for node in ast.walk(tree): - if isinstance(node, ast.Import): - for alias in node.names: - if alias.name.split(".")[0] not in _ALLOWED_MODULES: - raise UnsafeCodeError(f"import '{alias.name}' is not allowed") - elif isinstance(node, ast.ImportFrom): - if node.module and node.module.split(".")[0] not in _ALLOWED_MODULES: - raise UnsafeCodeError(f"import from '{node.module}' is not allowed") - elif isinstance(node, ast.Call): - if isinstance(node.func, ast.Name) and node.func.id in _BLOCKED_BUILTINS: - raise UnsafeCodeError(f"builtin '{node.func.id}()' is not allowed") - def _minimal_env() -> dict: return { @@ -79,19 +43,22 @@ async def run_backend(code: str, input_data: dict) -> ComputeResult: preamble = ( "import json, sys, io, builtins\n" - "for _b in ('exec','eval','compile','open','input',\n" - " 'breakpoint','exit','quit'):\n" - " try: delattr(builtins, _b)\n" - " except AttributeError: pass\n" - "_orig_stdout = sys.stdout\n" - "_capture = io.StringIO()\n" - "sys.stdout = _capture\n" + "p_stdout = sys.stdout\n" + "p_capture = io.StringIO()\n" + "sys.stdout = p_capture\n" "input_data = json.loads(sys.stdin.read())\n" "result = {}\n" + # Warm the allowlist BEFORE scrubbing builtins: half the stdlib borrows the builtins the scrub deletes while it loads (tokenize does `from builtins import open`, taking `import dataclasses` with it). Then the module handles go, because leaving `sys` bound hands gate-passing code a live `sys.modules['os']` with no import statement in sight. + f"for p_name in {tuple(sorted(ALLOWED_MODULES))!r}:\n" + " try: __import__(p_name)\n" + " except ImportError: pass\n" + "for p_name in ('open','input','breakpoint','exit','quit'):\n" + " try: delattr(builtins, p_name)\n" + " except AttributeError: pass\n" + "del sys, io, builtins, p_name\n" ) postamble = ( - "\nsys.stdout = _orig_stdout\n" - 'json.dump({"__stdout__": _capture.getvalue(), "__result__": result}, sys.stdout)\n' + "\np_stdout.write(json.dumps({\"__stdout__\": p_capture.getvalue(), \"__result__\": result}))\n" ) wrapper = preamble + code + postamble diff --git a/openswarm-edge/tests/test_edge.py b/openswarm-edge/tests/test_edge.py index 8d05a695..6d5a6f1a 100644 --- a/openswarm-edge/tests/test_edge.py +++ b/openswarm-edge/tests/test_edge.py @@ -21,7 +21,9 @@ from app.main import slug_from_host from app.bundles import unpack, resolve_file from app.inject import inject_runtime from app.ratelimit import RateLimiter -from app.sandbox import validate_code_safety, run_backend, UnsafeCodeError +from app.code_safety import validate_code_safety, UnsafeCodeError +from app import sandbox as edge_sandbox +from app.sandbox import run_backend def test_slug_from_host(): @@ -179,11 +181,51 @@ def test_sandbox_rejects_unsafe_and_allows_safe(): validate_code_safety("import math\nresult={'x': math.pi}") # no raise +def test_sandbox_rejects_the_module_handle_escapes(): + """Issue #134 at the public tier: the preamble's own `sys`/`io` handles, an + attribute chain onto a withheld module, and the dunder walk.""" + for code in ( + "result = {'cwd': sys.modules['os'].getcwd()}", + "result = {'x': str(io.open)}", + "result = {'c': str(json.codecs)}", + "result = {'n': len(().__class__.__bases__[0].__subclasses__())}", + "result = {'c': str(getattr(json, 'codecs'))}", + ): + try: + validate_code_safety(code) + assert False, f"expected UnsafeCodeError for {code!r}" + except UnsafeCodeError: + pass + + def test_sandbox_runs_safe_code(): res = asyncio.run(run_backend("result = {'sum': sum(input_data['nums'])}", {"nums": [1, 2, 3]})) assert res.result == {"sum": 6} +def test_sandbox_runs_allowlisted_imports(): + """The builtins scrub used to delete exec/eval, which broke `import statistics` + and every namedtuple; a sandbox that can't run real code isn't secure, it's off.""" + res = asyncio.run(run_backend( + "import statistics, datetime\n" + "result = {'mean': statistics.mean(input_data['nums']), 'd': datetime.time(9, 0).isoformat()}", + {"nums": [1, 2, 3]}, + )) + assert res.result == {"mean": 2, "d": "09:00:00"} + + +def test_sandbox_subprocess_has_no_module_handles(monkeypatch): + """Second wall: pretend a payload beats the gate, and the subprocess still + has no module left to grab.""" + monkeypatch.setattr(edge_sandbox, "validate_code_safety", lambda code: None) + for handle in ("sys", "io", "builtins"): + try: + asyncio.run(run_backend(f"result = {{'x': str({handle})}}", {})) + assert False, f"{handle} was still reachable" + except RuntimeError as e: + assert "NameError" in str(e) + + def test_inject_runtime(): out = inject_runtime(b"xhi").decode() assert "OUTPUT_COMPUTE" in out and "OUTPUT_LLM" in out diff --git a/openswarm-runner/Dockerfile b/openswarm-runner/Dockerfile new file mode 100644 index 00000000..31fec01b --- /dev/null +++ b/openswarm-runner/Dockerfile @@ -0,0 +1,177 @@ +# syntax=docker/dockerfile:1 +# +# One OpenSwarm workflow run, then exit. Build context is the REPO ROOT, not this +# directory, because the image needs backend/ and requirements.lock: +# +# docker build --platform linux/amd64 -f openswarm-runner/Dockerfile -t openswarm-runner . +# +# Layout the image commits to (all three are load-bearing, backend code resolves +# them with zero changes when OPENSWARM_PACKAGED=1): +# /app/backend the FastAPI orchestrator +# /app/router 9router's standalone server, found by p_find_9router_dir() +# /app/python-env UV_PYTHON target probed by tools_lib/mcp_config.py +# +# Plus the renderer half, which exists so browser tools work the way they do on a +# laptop instead of being denied: +# /app/electron-runtime the same CastLabs Electron build the desktop app ships +# /app/electron the desktop shell's own main process, unmodified +# /app/frontend the production webpack bundle, served off loopback +# +# amd64 only. CastLabs publishes no linux-arm64 build, and running a DIFFERENT +# Electron than the desktop app ships would quietly undo the point of this image. + +ARG PYTHON_VERSION=3.13 +ARG NODE_VERSION=20 +ARG ROUTER_VERSION=0.3.60 +ARG UV_VERSION=0.11.8 +# Must track electron/package.json's devDependency, or the container drives a different browser than the laptop does. +ARG ELECTRON_VERSION=42.3.3+wvcus +ARG ELECTRON_SHA256=5b6ce3a4d13f07fc63d79e884f6a40d1bc8a1cdf82cb2d130c26e8c1530649cb + +FROM node:${NODE_VERSION}-bookworm-slim AS node + +# 9router 0.3.60 is pure JavaScript; --ignore-scripts skips a postinstall that only rebuilds a native addon the standalone server never loads. +FROM node AS router +ARG ROUTER_VERSION +WORKDIR /stage +RUN printf '{"name":"router-stage","version":"0.0.0","private":true}\n' > package.json \ + && npm install "9router@${ROUTER_VERSION}" --no-save --no-audit --no-fund --silent --ignore-scripts \ + && test -f node_modules/9router/app/server.js \ + && test -z "$(find node_modules/9router -name '*.node' -print -quit)" + +# The App Builder's template dependencies, installed once here and shipped ALREADY EXTRACTED at +# the digest path backend/apps/outputs/view_builder_templates.py already probes. Without it the +# first CreateApp in a run pays a cold npm install against the public registry, and a run with no +# egress just fails. NOT $BUILDPLATFORM: vite pulls in a platform-specific esbuild, so this has to +# resolve on the arch the container will actually run on. +FROM node:${NODE_VERSION}-bookworm-slim AS webapp-template +WORKDIR /stage +COPY backend/apps/outputs/webapp_template/frontend/package.json ./package.json +RUN set -eux; \ + npm install --no-audit --no-fund --loglevel=error --ignore-scripts; \ + test -x node_modules/.bin/vite; \ + digest="$(sha256sum package.json | cut -c1-12)"; \ + mkdir -p "/out/${digest}"; \ + mv node_modules "/out/${digest}/node_modules" + +# Webpack output is architecture-independent, so this runs natively on the build host rather than under emulation. +FROM --platform=$BUILDPLATFORM node:${NODE_VERSION}-bookworm-slim AS frontend +WORKDIR /src +COPY frontend/package.json frontend/package-lock.json ./ +RUN npm ci --no-audit --no-fund --silent +COPY frontend ./ +RUN npm run build && test -f dist/index.html + +# The shell's runtime deps only. --ignore-scripts leaves uiohook-napi without its prebuilt addon, which is correct: it taps a real keyboard, there isn't one here, and voiceHotkey already requires it inside a try. +FROM --platform=$BUILDPLATFORM node:${NODE_VERSION}-bookworm-slim AS shell-deps +WORKDIR /stage +COPY electron/package.json electron/package-lock.json ./ +RUN npm install --omit=dev --ignore-scripts --no-audit --no-fund --silent + +FROM debian:bookworm-slim AS electron +ARG ELECTRON_VERSION +ARG ELECTRON_SHA256 +ARG TARGETARCH +RUN set -eux; \ + test "${TARGETARCH}" = "amd64" || { echo "the renderer half is amd64-only: CastLabs ships no linux-${TARGETARCH} Electron" >&2; exit 1; }; \ + apt-get update && apt-get install -y --no-install-recommends curl ca-certificates unzip; \ + url="https://github.com/castlabs/electron-releases/releases/download/v${ELECTRON_VERSION}/electron-v${ELECTRON_VERSION}-linux-x64.zip"; \ + curl -fsSL -o /tmp/electron.zip "${url}"; \ + echo "${ELECTRON_SHA256} /tmp/electron.zip" | sha256sum -c -; \ + mkdir -p /stage; \ + unzip -q /tmp/electron.zip -d /stage; \ + rm /tmp/electron.zip; \ + test -x /stage/electron + +FROM python:${PYTHON_VERSION}-slim-bookworm AS uv +ARG UV_VERSION +ARG TARGETARCH +RUN set -eux; \ + apt-get update && apt-get install -y --no-install-recommends curl ca-certificates; \ + case "${TARGETARCH}" in \ + amd64) triple=x86_64-unknown-linux-gnu; sha=56dd1b66701ecb62fe896abb919444e4b83c5e8645cca953e6ddd496ff8a0feb ;; \ + arm64) triple=aarch64-unknown-linux-gnu; sha=eee8dd658d20e5ac85fec9c2326b6cbc9d83a1eef09ef07433e58698ac849591 ;; \ + *) echo "unsupported TARGETARCH ${TARGETARCH}" >&2; exit 1 ;; \ + esac; \ + curl -fsSL -o /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-${triple}.tar.gz"; \ + echo "${sha} /tmp/uv.tar.gz" | sha256sum -c -; \ + mkdir -p /stage; \ + tar -xzf /tmp/uv.tar.gz -C /stage --strip-components=1 + +# Wheels only: the runtime image ships no compiler, so a source build here is a build-time failure rather than a 3am surprise. +FROM python:${PYTHON_VERSION}-slim-bookworm AS pydeps +COPY backend/requirements.lock /tmp/requirements.lock +RUN pip install --no-cache-dir --require-hashes --only-binary=:all: \ + --prefix=/opt/pydeps -r /tmp/requirements.lock + +FROM python:${PYTHON_VERSION}-slim-bookworm + +# The X server plus every shared object `ldd` reports the Electron binary wanting, and the fonts without which every page renders as boxes. Derived from ldd on the real binary, not from a blog post. +RUN set -eux; \ + apt-get update; \ + apt-get install -y --no-install-recommends \ + git ca-certificates \ + xvfb fonts-liberation \ + libasound2 libatk-bridge2.0-0 libatk1.0-0 libatspi2.0-0 libcairo2 libcups2 \ + libdbus-1-3 libdrm2 libexpat1 libgbm1 libglib2.0-0 libgtk-3-0 libnss3 \ + libpango-1.0-0 libx11-6 libxcb1 libxcomposite1 libxdamage1 libxext6 \ + libxfixes3 libxkbcommon0 libxrandr2 libxtst6; \ + rm -rf /var/lib/apt/lists/* + +COPY --from=node /usr/local/bin/node /usr/local/bin/node +# npm and npx too, not just node. They are shims into lib/node_modules, so copying the tree and +# re-linking is the only way to get them; a `node` with no `npm` is what left the App Builder +# scaffolding an app it could never install, build or serve. +COPY --from=node /usr/local/lib/node_modules /usr/local/lib/node_modules +COPY --from=pydeps /opt/pydeps /usr/local + +# Numeric owner on every /app copy, because a `chown -R /app` afterwards rewrites the whole tree into a second layer and the image pays for it twice (that cost 493MB before this line existed). Numeric, not `runner`, because the user is created further down. +COPY --from=router --chown=10001:10001 /stage/node_modules/9router/app /app/router +COPY --chown=10001:10001 backend /app/backend +COPY --chown=10001:10001 openswarm-runner/runner /app/runner +COPY --chown=10001:10001 electron /app/electron +COPY --from=shell-deps --chown=10001:10001 /stage/node_modules /app/electron/node_modules +COPY --from=electron --chown=10001:10001 /stage /app/electron-runtime +COPY --from=frontend --chown=10001:10001 /src/dist /app/frontend + +# After backend/, never before: mcp_config.resolve_command probes uv-bin last, and the repo's own copy is Mach-O. +COPY --from=uv --chown=10001:10001 /stage/uv /app/backend/uv-bin/uv +COPY --from=uv --chown=10001:10001 /stage/uvx /app/backend/uv-bin/uvx + +# Also after backend/, and at the exact path bundled_extracted_modules() looks for. +COPY --from=webapp-template --chown=10001:10001 /out /app/backend/apps/outputs/webapp_template_cache + +RUN set -eux; \ + if ls /app/backend/.env* >/dev/null 2>&1; then echo "a dotenv reached the image; fix Dockerfile.dockerignore" >&2; exit 1; fi; \ + ln -s ../lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm; \ + ln -s ../lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx; \ + npm --version >/dev/null; \ + mkdir -p /app/python-env/bin; \ + ln -s /usr/local/bin/python3 /app/python-env/bin/python3; \ + find /app/backend -name '__pycache__' -type d -prune -exec rm -rf {} +; \ + useradd --create-home --uid 10001 --shell /usr/sbin/nologin runner; \ + mkdir -p /data; \ + ln -s /data/openswarm /app/backend/data; \ + mkdir -p /tmp/.X11-unix; \ + chmod 1777 /tmp/.X11-unix; \ + chown runner:runner /data; \ + printf '[user]\n\tname = OpenSwarm Cloud Run\n\temail = cloud-run@openswarm.local\n[init]\n\tdefaultBranch = main\n[safe]\n\tdirectory = *\n' > /etc/gitconfig + +USER runner +WORKDIR /app +ENV HOME=/home/runner \ + PYTHONPATH=/app \ + PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 \ + OPENSWARM_HEADLESS=1 \ + OPENSWARM_PACKAGED=1 \ + OPENSWARM_DATA_ROOT=/data/openswarm \ + OPENSWARM_HOST=127.0.0.1 \ + OPENSWARM_PORT=8324 \ + DATA_DIR=/data/9router \ + NODE_ENV=production \ + ELECTRON_BIN=/app/electron-runtime/electron \ + OPENSWARM_RUN_WORKSPACE=/data/workspace \ + OPENSWARM_NODE_PATH=/usr/local/bin/node + +ENTRYPOINT ["python3", "-m", "runner.main"] diff --git a/openswarm-runner/Dockerfile.dockerignore b/openswarm-runner/Dockerfile.dockerignore new file mode 100644 index 00000000..3eb6885b --- /dev/null +++ b/openswarm-runner/Dockerfile.dockerignore @@ -0,0 +1,33 @@ +* +!backend +!openswarm-runner/runner +!electron +!frontend + +# A developer's real OAuth client secrets live here; baking them into an image that +# gets pushed to a registry is how a laptop leaks credentials. The Dockerfile asserts +# they are gone, so this list failing open fails the build instead of shipping. +backend/.env +backend/.env.* + +backend/data +backend/.venv +backend/uv-bin +backend/tests +backend/.pytest_cache + +# The bundle is built in a stage inside the image. A developer's stale local dist must +# never be what a cloud run renders. +frontend/dist +frontend/node_modules +electron/node_modules +electron/dist +electron/build-staging +electron/python-env +# Prebuilt Mach-O addons for the mac trackpad; main.js already skips a missing one. +electron/native +**/*.test.js + +**/__pycache__ +**/*.pyc +**/.DS_Store diff --git a/openswarm-runner/README.md b/openswarm-runner/README.md new file mode 100644 index 00000000..3e8606a5 --- /dev/null +++ b/openswarm-runner/README.md @@ -0,0 +1,158 @@ +# openswarm-runner + +One ephemeral Linux container that executes ONE OpenSwarm workflow run and exits. +One Fly Firecracker machine per run, no state kept. + +## Build + +The build context is the **repo root**, not this directory (the image needs `backend/`, +`electron/`, `frontend/` and `backend/requirements.lock`). **amd64 only**, see the +renderer section: + +```bash +docker build --platform linux/amd64 -f openswarm-runner/Dockerfile -t openswarm-runner . +``` + +Nothing has to be built on the host first: the frontend bundle and the shell's node +modules are built in their own stages inside the image. + +## Run + +The container is told everything it needs by one JSON run spec in `OPENSWARM_RUN_SPEC` +(or a path in `OPENSWARM_RUN_SPEC_FILE`). See `runner/run_spec.py` for the typed shape. + +```json +{ + "run_id": "cr_01J...", + "workflow": { "id": "wf_1", "title": "Daily digest", "model": "opus-5", + "steps": [{ "text": "summarize my inbox" }] }, + "credentials": [ + { "provider": "claude", "auth_type": "oauth", + "access_token": "", + "expires_at": "2026-07-31T20:00:00Z" } + ], + "callback": { "url": "https://api.openswarm.com/api/cloud-runs/cr_01J.../report", + "token": "" }, + "max_run_seconds": 1800, + "needs_browser": true +} +``` + +Exit codes: `0` ok, `1` runner crash, `2` bad spec, `3` credential expired on arrival, +`4` backend never came up, `5` workflow failed, `6` wall-clock cap hit, +`7` no Electron window ever registered. + +## Files the run makes + +`/data/workspace` is the agent's working directory and **the only path whose contents survive**. +It is seeded as `default_folder` before the backend boots, and the agent is told in its system +prompt that files saved there come back and everything else is destroyed. + +After the workflow reaches a terminal state (including a timeout, so partial work still lands) the +runner walks that directory and POSTs each file to `callback.artifacts_url`, then sends the +terminal report. That order is load-bearing: the per-run callback token is refused once the run is +closed, so uploading afterwards would be rejected. + +Caps, applied in the runner AND again at the control plane, which is the one that counts: + +| Limit | Value | +| --- | --- | +| One file | 20 MB | +| One run, all files | 50 MB | +| Files per run | 40 | + +Nothing is ever truncated. A file past a cap is not sent and instead arrives as a row in the +report's `files[]` carrying a written reason, so the user reads "your 512 MB render could not be +sent" rather than finding a 20 MB fragment. `.git`, `node_modules`, `__pycache__`, `.venv`, +`.claude` and the usual caches are skipped, and symlinks are never followed. + +## Skills and connected apps + +`skills[]` in the run spec is written to `~/.claude/skills//` before boot. This is not a +nicety: the backend registers the Skill tool only when at least one non-built-in skill exists on +disk, so a container without them has no Skill tool at all and answers from general knowledge in +the same confident voice it would use with the real thing. + +`unavailable_mcp_servers[]` is **names only**. The user's MCP credentials (Slack session cookies, +Notion and GitHub access tokens, Google refresh tokens) never leave their machine, so the names go +up purely so the run's system prompt can tell the agent which apps exist and are out of reach. +`McpServerNote` forbids extra fields, so there is no shape a secret could travel in. + +## What the image carries for the App Builder + +`node`, `npm` and `npx`, plus the App Builder template's `node_modules` pre-installed at the digest +path `bundled_extracted_modules()` probes. Without npm, `CreateApp` scaffolded an app that could +never install, build or serve; without the baked cache, the first `CreateApp` in a run would pay a +cold registry install. `git` also carries a system identity (`/etc/gitconfig`), so a workflow that +commits does not die on "Author identity unknown". + +## The renderer + +OpenSwarm's browser tier is not an HTTP client. Element serialization and every click, +type and scroll live in `frontend/src/shared/browserCommandHandler.ts` and drive a live +Electron ``; the backend only relays commands over the dashboard WebSocket. So +the container runs **the real desktop shell**, unmodified, on a virtual display: + +``` +Xvfb :99 -> Electron (ELECTRON_DEV=1, OPENSWARM_DEV_URL=#/dashboard/cloud-run) + -> registers on /ws/dashboard -> browser tools are live +``` + +`ELECTRON_DEV=1` is the same path `bash run.sh` uses: the shell attaches to the backend +already running here instead of spawning a second one. The bundle is served off loopback +on `:4173`, the same port the packaged app prefers, and deep-linked at the run's one +dashboard so no human has to click anything. + +Three things follow from this and are worth knowing before you touch it: + +- **amd64 only.** CastLabs (whose Electron the desktop app ships) publishes no + linux-arm64 build. Running a *different* Electron in the cloud than users run on their + laptops would quietly undo the point of the image, so the build refuses other arches. +- **`--no-sandbox`.** Chromium's setuid sandbox needs a root-owned binary and its + namespace sandbox needs unprivileged user namespaces; a non-root container under + Docker's default seccomp has neither. The wall this run relies on is the Firecracker VM + around the whole container, not Chromium's own layer. The flag lives in a named constant + in `runner/renderer_process.py` rather than inside a launch string, on purpose. +- **`needs_browser: false` skips it.** Boot costs roughly 15s and ~500MB of the run's + memory, so a workflow that never opens a page can opt out. Default is on: parity is the + reason this image exists, and opting out should be the thing you have to say. + +If Electron starts but no window ever registers, the run **fails** (exit 7) rather than +proceeding without a browser. A browser workflow that silently ran blind produces a +confident wrong answer, which is worse than no answer. + +## The credential rule + +**A `providerConnections[]` entry this runner writes never contains a `refreshToken`.** +9Router's refresh dispatcher bails on `if (!b || !b.refreshToken) return null`, so +omitting the field is what makes the container incapable of rotating the user's grant. +If it ever rotated, the user's laptop would be left replaying a dead token and the +provider would revoke the whole grant family. + +Two independent walls enforce it, and a third makes a leak require deleting the code +that builds the entry: + +1. `ProviderCredential` forbids extra fields, so a spec carrying `refreshToken` fails + validation before the backend boots. +2. `assert_no_refresh_token` re-reads the assembled db payload just before the write. +3. `router_connection` assembles the entry from a fixed key list, never a passthrough. + +All three live in `runner/seed/router_credentials.py`. + +An access token that arrives expired fails the run (exit 3). The runner never refreshes. + +## Test + +```bash +PYTHONPATH=.:openswarm-runner backend/.venv/bin/python3 -m pytest openswarm-runner/tests -q +``` + +The Electron boot itself needs Linux and a display, so the tests pin the contract around +it (the deep link, the bundle check, the three ways "no window" ends) rather than the +boot. Proving the browser tier actually behaves means running a real page in both places +and comparing; see the parity matrix in the cloud-browser work notes. + +## Deploy + +Not deployed. `fly.toml` is written but never applied; read its header first, the app +has to be created onto its own isolated private network by hand before any deploy. diff --git a/openswarm-runner/fly.toml b/openswarm-runner/fly.toml new file mode 100644 index 00000000..ff2ed454 --- /dev/null +++ b/openswarm-runner/fly.toml @@ -0,0 +1,51 @@ +# openswarm-runner: one ephemeral Firecracker machine per workflow run. Boots the +# backend headless, runs the workflow, reports, exits. Nothing here is long-lived. +# +# THIS APP MUST NOT SHARE THE TRUSTED 6PN MESH with openswarm-cloud / openswarm-edge. +# The agent inside has Bash and executes user prose, so from in here +# `curl http://openswarm-cloud.internal:8080` must resolve to nothing. Fly decides an +# app's private network AT CREATE TIME and fly.toml cannot express it, so the app is +# created once, by hand, onto its own isolated network: +# +# fly apps create openswarm-runner --org openswarm --network openswarm-runner-isolated +# fly deploy . --config openswarm-runner/fly.toml --dockerfile openswarm-runner/Dockerfile +# +# (deploy runs from the REPO ROOT: the image needs backend/ in its build context.) +# Verify the isolation after the first deploy, do not assume it: +# fly ssh console -a openswarm-runner -C "getent hosts openswarm-cloud.internal" # must fail +# +# There is deliberately no [http_service] and no [[services]]: the runner takes no +# inbound traffic and gets no public IP. It reaches the control plane outbound over +# the public internet with the callback token in the run spec, which is why the two +# do not need a shared private network in the first place. +# +# Machines are created per run by the control plane (Machines API, auto_destroy=true, +# run spec passed as OPENSWARM_RUN_SPEC). This file is the app-level shape they inherit. + +app = 'openswarm-runner' +primary_region = 'iad' +kill_signal = 'SIGTERM' +kill_timeout = '30s' + +[build] + dockerfile = 'Dockerfile' + +[env] + # Hard wall-clock cap, enforced twice inside the container: the poll loop stops the + # run at this mark, and an independent thread kills the process 90s later. A run + # spec asking for more is clamped down to this, never up. + RUNNER_MAX_RUN_SECONDS = '1800' + OPENSWARM_HEADLESS = '1' + OPENSWARM_PACKAGED = '1' + OPENSWARM_DATA_ROOT = '/data/openswarm' + OPENSWARM_HOST = '127.0.0.1' + OPENSWARM_PORT = '8324' + DATA_DIR = '/data/9router' + +# No [[mounts]]: a run's state is garbage the moment it ends, and an ephemeral rootfs +# means one run cannot leave a credential lying around for the next tenant to find. + +[[vm]] + cpu_kind = 'shared' + cpus = 2 + memory_mb = 4096 diff --git a/openswarm-runner/runner/__init__.py b/openswarm-runner/runner/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/openswarm-runner/runner/boot/__init__.py b/openswarm-runner/runner/boot/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/openswarm-runner/runner/boot/backend_process.py b/openswarm-runner/runner/boot/backend_process.py new file mode 100644 index 00000000..b4114933 --- /dev/null +++ b/openswarm-runner/runner/boot/backend_process.py @@ -0,0 +1,111 @@ +"""Boot the OpenSwarm backend inside the container and wait until it answers. + +Spawned the same way the desktop shell spawns it (`python -m uvicorn backend.main:app` +on loopback) so the cloud path and the laptop path are the same code on the same +socket. Loopback, not 0.0.0.0: every caller of this API lives in this container, and +the agent running inside it has Bash, so there is no reason to publish the port onto +the machine's private network. +""" + +import os +import subprocess +import time +from typing import Dict, List, Optional + +import httpx +from pydantic import BaseModel, ConfigDict, InstanceOf +from typeguard import typechecked + +HOST = "127.0.0.1" +HEALTH_PATH = "/api/health/check" +# 9Router is a Next.js standalone server: it binds `process.env.HOSTNAME || '0.0.0.0'`, and Docker sets HOSTNAME to the container id, so left alone it listens on the container's eth0 address and every probe of 127.0.0.1:20128 gets ECONNREFUSED. Set on the backend's env because the backend is what spawns node. +ROUTER_BIND_HOSTNAME = "127.0.0.1" +AUTH_TOKEN_FILENAME = "auth.token" +# The backend imports the whole app graph before it binds; on a cold Fly machine that has been measured in tens of seconds, so the budget is generous rather than tight. +BOOT_TIMEOUT_SECONDS = 120.0 +SHUTDOWN_GRACE_SECONDS = 10.0 + + +class BackendUnavailable(RuntimeError): + """The backend never came up, or died while we were using it.""" + + +class BackendProcess(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + process: InstanceOf[subprocess.Popen] + base_url: str + token: str + + @typechecked + def headers(self) -> Dict[str, str]: + return {"Authorization": f"Bearer {self.token}"} + + @typechecked + def is_alive(self) -> bool: + return self.process.poll() is None + + +@typechecked +def p_command(port: int) -> List[str]: + return ["python3", "-m", "uvicorn", "backend.main:app", "--host", HOST, "--port", str(port)] + + +@typechecked +def p_read_auth_token(data_root: str) -> str: + """The backend mints this before it binds, so by the time health passes the file exists.""" + path = os.path.join(data_root, AUTH_TOKEN_FILENAME) + try: + with open(path, "r", encoding="utf-8") as handle: + return handle.read().strip() + except OSError as exc: + raise BackendUnavailable(f"backend is up but its auth token is unreadable at {path}: {exc}") from exc + + +@typechecked +def start_backend(app_root: str, data_root: str, port: int, deadline: float) -> BackendProcess: + """Spawn the backend and block until it answers health, or raise BackendUnavailable.""" + environment = dict(os.environ) + environment["OPENSWARM_DATA_ROOT"] = data_root + environment["OPENSWARM_HEADLESS"] = "1" + environment["OPENSWARM_PORT"] = str(port) + environment["OPENSWARM_HOST"] = HOST + environment["HOSTNAME"] = ROUTER_BIND_HOSTNAME + environment["PYTHONPATH"] = app_root + + process = subprocess.Popen(p_command(port), cwd=app_root, env=environment) + base_url = f"http://{HOST}:{port}" + budget = min(time.monotonic() + BOOT_TIMEOUT_SECONDS, deadline) + + with httpx.Client(timeout=2.0) as client: + while time.monotonic() < budget: + if process.poll() is not None: + raise BackendUnavailable(f"backend exited during startup with code {process.returncode}") + try: + healthy = client.get(f"{base_url}{HEALTH_PATH}").status_code == 200 + except httpx.HTTPError: + healthy = False + if healthy: + try: + token = p_read_auth_token(data_root) + except BackendUnavailable: + stop_backend(process) + raise + return BackendProcess(process=process, base_url=base_url, token=token) + time.sleep(0.25) + + stop_backend(process) + raise BackendUnavailable(f"backend did not answer {HEALTH_PATH} within its startup budget") + + +@typechecked +def stop_backend(process: Optional[subprocess.Popen]) -> None: + """SIGTERM then SIGKILL. The machine is about to die anyway; this just stops the logs mid-sentence.""" + if process is None or process.poll() is not None: + return + process.terminate() + try: + process.wait(timeout=SHUTDOWN_GRACE_SECONDS) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=SHUTDOWN_GRACE_SECONDS) diff --git a/openswarm-runner/runner/boot/renderer_process.py b/openswarm-runner/runner/boot/renderer_process.py new file mode 100644 index 00000000..19267da0 --- /dev/null +++ b/openswarm-runner/runner/boot/renderer_process.py @@ -0,0 +1,233 @@ +"""Boot the real Electron shell inside the container so browser tools have a window to drive. + +The browser tier is not an HTTP client: the element serialization and every click, type and +scroll live in the frontend and drive a live Electron ``. There is no way to get +laptop-identical behaviour without the laptop's actual renderer, so this starts one on a +virtual display and points it at the backend that is already running in this container. + +The Electron process runs the same `ELECTRON_DEV=1` path a developer uses (`bash run.sh`): +the shell attaches to an existing backend on OPENSWARM_PORT instead of spawning its own, and +loads whatever OPENSWARM_DEV_URL says. Here that URL is the packaged frontend bundle served +off loopback, deep-linked straight at the run's dashboard so the window registers without a +human clicking anything. +""" + +import functools +import http.server +import logging +import os +import shutil +import socketserver +import subprocess +import threading +import time +from typing import Dict, List, Optional + +import httpx +from pydantic import BaseModel, ConfigDict, InstanceOf +from typeguard import typechecked + +logger = logging.getLogger("runner.renderer") + +HOST = "127.0.0.1" +RENDERER_HEALTH_PATH = "/api/health/renderer" +# Same port the packaged app prefers, so the renderer's origin (and therefore its localStorage) is the one the frontend was built expecting. +FRONTEND_PORT = 4173 +DISPLAY = ":99" +XVFB_SCREEN = "1920x1080x24" +# Cold Electron under a virtual display: Xvfb, Chromium boot, React mount, then the deferred dashboard socket. Measured in the low tens of seconds, so the budget is generous rather than tight. +RENDERER_TIMEOUT_SECONDS = 180.0 +XVFB_READY_TIMEOUT_SECONDS = 20.0 +SHUTDOWN_GRACE_SECONDS = 5.0 + +# Chromium's setuid sandbox needs a root-owned binary and its namespace sandbox needs unprivileged +# user namespaces, neither of which a non-root container under Docker's default seccomp profile has. +# The isolation this run relies on is the Firecracker VM around the whole container, not Chromium's +# own layer. Stated here rather than buried in a launch string, because dropping a sandbox is a +# choice and the reader deserves to see it made. +SANDBOX_FLAGS: List[str] = ["--no-sandbox"] +# /dev/shm defaults to 64MB in a container, which Chromium overruns and then crashes on. +CONTAINER_CHROMIUM_FLAGS: List[str] = ["--disable-dev-shm-usage", "--disable-gpu"] + + +class RendererUnavailable(RuntimeError): + """No Electron window ever registered, so browser tools would be dead this run.""" + + +class RendererProcess(BaseModel): + """The virtual display and the Electron shell drawing into it. Dies with the run.""" + + model_config = ConfigDict(validate_assignment=True) + + xvfb: InstanceOf[subprocess.Popen] + electron: InstanceOf[subprocess.Popen] + url: str + + @typechecked + def is_alive(self) -> bool: + return self.electron.poll() is None and self.xvfb.poll() is None + + +@typechecked +def dashboard_url(port: int, dashboard_id: str) -> str: + """The frontend deep-link that mounts a dashboard directly. HashRouter, so the route is a fragment.""" + return f"http://{HOST}:{port}/index.html#/dashboard/{dashboard_id}" + + +@typechecked +def serve_frontend(frontend_dir: str) -> int: + """Serve the built bundle off loopback in a daemon thread; returns the port it landed on. + + Falls back to an OS-assigned port if 4173 is held, exactly like the packaged shell does. + """ + if not os.path.isfile(os.path.join(frontend_dir, "index.html")): + raise RendererUnavailable( + f"no frontend bundle at {frontend_dir}; the image must be built with frontend/dist in it" + ) + + handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=frontend_dir) + + class p_Server(socketserver.ThreadingTCPServer): + daemon_threads = True + allow_reuse_address = True + + try: + server = p_Server((HOST, FRONTEND_PORT), handler) + except OSError: + server = p_Server((HOST, 0), handler) + port = server.server_address[1] + threading.Thread(target=server.serve_forever, daemon=True, name="frontend-server").start() + logger.info("frontend bundle served from %s on %s:%d", frontend_dir, HOST, port) + return port + + +@typechecked +def p_x_socket_ready(display: str) -> bool: + return os.path.exists(f"/tmp/.X11-unix/X{display.lstrip(':')}") + + +@typechecked +def start_xvfb(display: str = DISPLAY) -> subprocess.Popen: + """Bring up the virtual display and wait for its socket, so Electron never races it.""" + if shutil.which("Xvfb") is None: + raise RendererUnavailable("Xvfb is not installed in this image, so there is no display to draw on") + process = subprocess.Popen( + ["Xvfb", display, "-screen", "0", XVFB_SCREEN, "-nolisten", "tcp"], + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + budget = time.monotonic() + XVFB_READY_TIMEOUT_SECONDS + while time.monotonic() < budget: + if process.poll() is not None: + raise RendererUnavailable(f"Xvfb exited immediately with code {process.returncode}") + if p_x_socket_ready(display): + logger.info("virtual display %s up (%s)", display, XVFB_SCREEN) + return process + time.sleep(0.1) + p_stop(process) + raise RendererUnavailable(f"Xvfb never created a socket for {display}") + + +@typechecked +def p_electron_env(backend_port: int, url: str, display: str) -> Dict[str, str]: + environment = dict(os.environ) + # The dev path: attach to the backend already running here rather than spawning a second one, and load the bundle we are serving instead of a webpack dev server. + environment["ELECTRON_DEV"] = "1" + environment["OPENSWARM_DEV_URL"] = url + environment["OPENSWARM_PORT"] = str(backend_port) + environment["DISPLAY"] = display + environment["ELECTRON_DISABLE_SECURITY_WARNINGS"] = "1" + environment.pop("OPENSWARM_PACKAGED", None) + return environment + + +@typechecked +def start_electron(app_root: str, backend_port: int, url: str, display: str = DISPLAY) -> subprocess.Popen: + binary = os.environ.get("ELECTRON_BIN", "/app/electron-runtime/electron") + if not os.path.isfile(binary): + raise RendererUnavailable(f"no Electron binary at {binary}; the image was built without a renderer") + app_dir = os.path.join(app_root, "electron") + command = [binary, app_dir, *SANDBOX_FLAGS, *CONTAINER_CHROMIUM_FLAGS] + logger.info("starting Electron: %s", " ".join(command)) + return subprocess.Popen(command, cwd=app_dir, env=p_electron_env(backend_port, url, display)) + + +@typechecked +def await_registration( + base_url: str, + headers: Dict[str, str], + electron: subprocess.Popen, + deadline: float, +) -> None: + """Block until the backend reports a renderer on its dashboard socket, or raise. + + Polls the backend rather than the Electron process because "the window is up" and "the + window can be driven" are different claims, and only the second one matters. + """ + budget = min(time.monotonic() + RENDERER_TIMEOUT_SECONDS, deadline) + with httpx.Client(timeout=5.0) as client: + while time.monotonic() < budget: + if electron.poll() is not None: + raise RendererUnavailable( + f"Electron exited with code {electron.returncode} before any window registered" + ) + try: + body = client.get(f"{base_url}{RENDERER_HEALTH_PATH}", headers=headers).json() + except (httpx.HTTPError, ValueError): + body = {} + if body.get("attached"): + logger.info("renderer registered (%s dashboard connection(s))", body.get("connections")) + return + time.sleep(0.5) + raise RendererUnavailable( + "Electron started but no renderer ever registered on the dashboard WebSocket within " + f"{RENDERER_TIMEOUT_SECONDS:.0f}s, so browser tools would be dead this run" + ) + + +@typechecked +def start_renderer( + app_root: str, + frontend_dir: str, + backend_base_url: str, + backend_headers: Dict[str, str], + backend_port: int, + dashboard_id: str, + deadline: float, +) -> RendererProcess: + """Display, bundle server, Electron, then block until the window is actually drivable.""" + url = dashboard_url(serve_frontend(frontend_dir), dashboard_id) + xvfb = start_xvfb() + try: + electron = start_electron(app_root, backend_port, url) + except BaseException: + p_stop(xvfb) + raise + try: + await_registration(backend_base_url, backend_headers, electron, deadline) + except BaseException: + p_stop(electron) + p_stop(xvfb) + raise + return RendererProcess(xvfb=xvfb, electron=electron, url=url) + + +@typechecked +def p_stop(process: Optional[subprocess.Popen]) -> None: + if process is None or process.poll() is not None: + return + process.terminate() + try: + process.wait(timeout=SHUTDOWN_GRACE_SECONDS) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=SHUTDOWN_GRACE_SECONDS) + + +@typechecked +def stop_renderer(renderer: Optional[RendererProcess]) -> None: + """Electron first, then the display it was drawing on.""" + if renderer is None: + return + p_stop(renderer.electron) + p_stop(renderer.xvfb) diff --git a/openswarm-runner/runner/main.py b/openswarm-runner/runner/main.py new file mode 100644 index 00000000..ac6b177f --- /dev/null +++ b/openswarm-runner/runner/main.py @@ -0,0 +1,260 @@ +"""One workflow run, one container, one exit code. + +Boots the backend headless, executes the workflow the control plane asked for, +reports the result, and dies. Nothing here is meant to survive the run. +""" + +import logging +import os +import subprocess +import sys +import threading +import time +from datetime import datetime, timezone +from typing import Optional + +from pydantic import BaseModel, ConfigDict +from typeguard import typechecked + +from runner.boot.backend_process import BackendProcess, BackendUnavailable, start_backend, stop_backend +from runner.boot.renderer_process import RendererProcess, RendererUnavailable, start_renderer, stop_renderer +from runner.results.deliverables import collect +from runner.results.report import RunReport, deliver_files, send_report +from runner.run_spec import CLOUD_RUN_DASHBOARD_ID, CallbackTarget, InvalidRunSpec, RunSpec, load_run_spec +from runner.seed.data_root import seed_data_root +from runner.seed.router_credentials import write_router_db +from runner.seed.skills import write_skills +from runner.workflow_run import RunOutcome, RunProgress, WorkflowRunFailed, execute_workflow + +EXIT_OK = 0 +EXIT_INTERNAL = 1 +EXIT_BAD_SPEC = 2 +EXIT_CREDENTIAL_EXPIRED = 3 +EXIT_BACKEND_UNAVAILABLE = 4 +EXIT_WORKFLOW_FAILED = 5 +EXIT_DEADLINE = 6 +EXIT_RENDERER_UNAVAILABLE = 7 + +DEFAULT_APP_ROOT = "/app" +DEFAULT_FRONTEND_DIR = "/app/frontend" +DEFAULT_DATA_ROOT = "/data/openswarm" +DEFAULT_ROUTER_DATA_DIR = "/data/9router" +# The agent's own folder, and the only place on this machine whose contents come home. +DEFAULT_RUN_WORKSPACE = "/data/workspace" +DEFAULT_PORT = 8324 +# Slack between the soft deadline (stop the run, report it) and the hard one (kill the process). +# Has to cover the file upload as well as the report's retries, so it is minutes, not seconds; the +# control plane's own kill sits further out again (dispatch.ts MACHINE_GRACE_MS). +REPORT_GRACE_SECONDS = 240.0 +# Ceiling the control plane cannot raise. A cap a caller can override is not a cap. +MAX_RUN_SECONDS_ENV = "RUNNER_MAX_RUN_SECONDS" +DEFAULT_MAX_RUN_SECONDS = 1800 + +logger = logging.getLogger("runner") + + +class Heartbeat(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + run_id: str + interval_seconds: float + callback: Optional[CallbackTarget] = None + last_sent: float = 0.0 + + @typechecked + def maybe_send(self, progress: RunProgress) -> None: + now = time.monotonic() + if now - self.last_sent < self.interval_seconds: + return + self.last_sent = now + send_report(self.callback, RunReport( + run_id=self.run_id, + phase="heartbeat", + status=progress.status, + active_step_idx=progress.active_step_idx, + last_tool_label=progress.last_tool_label, + )) + + +@typechecked +def effective_max_run_seconds(spec: RunSpec) -> int: + """The shorter of what the job asked for and what this machine's config allows.""" + try: + ceiling = int(os.environ.get(MAX_RUN_SECONDS_ENV, "") or DEFAULT_MAX_RUN_SECONDS) + except ValueError: + ceiling = DEFAULT_MAX_RUN_SECONDS + return max(60, min(spec.max_run_seconds, ceiling)) + + +@typechecked +def arm_hard_stop(seconds: float) -> None: + """Independent backstop on machine-seconds; fires even if the graceful path is wedged.""" + def p_fire() -> None: + time.sleep(seconds) + logger.error("hard wall-clock stop after %.0fs, killing the run", seconds) + os._exit(EXIT_DEADLINE) + + threading.Thread(target=p_fire, daemon=True, name="hard-stop").start() + + +@typechecked +def p_fail( + spec: Optional[RunSpec], + status: str, + message: str, + code: int, + workspace: Optional[str] = None, +) -> int: + """Report a failure, handing over anything the run managed to make first. + + `workspace` is passed only where the agent actually ran: a workflow that died on step 3 may + have written a perfectly good report on step 1, and losing it because a later step threw is + exactly the "the file died with the machine" problem this whole path exists to fix. + """ + logger.error("%s: %s", status, message) + files = ( + deliver_files(spec.callback, workspace, collect(workspace)) + if spec is not None and workspace is not None + else [] + ) + send_report( + spec.callback if spec else None, + RunReport( + run_id=spec.run_id if spec else "unknown", + phase="finished", + status=status, + exit_code=code, + error=message, + files=files, + ), + ) + return code + + +@typechecked +def p_exit_code_for(outcome: RunOutcome) -> int: + if outcome.status in ("success", "ran_late"): + return EXIT_OK + if outcome.status == "timed_out": + return EXIT_DEADLINE + return EXIT_WORKFLOW_FAILED + + +@typechecked +def p_run(spec: RunSpec, deadline: float) -> int: + now = datetime.now(timezone.utc) + expired = spec.expired_credentials(now) + if expired: + names = ", ".join(credential.provider for credential in expired) + return p_fail( + spec, + "failure", + f"access token for {names} is expired or about to expire; the runner never refreshes, " + "so the control plane must re-issue it", + EXIT_CREDENTIAL_EXPIRED, + ) + + app_root = os.environ.get("OPENSWARM_APP_ROOT", DEFAULT_APP_ROOT) + data_root = os.environ.get("OPENSWARM_DATA_ROOT", DEFAULT_DATA_ROOT) + router_data_dir = os.environ.get("DATA_DIR", DEFAULT_ROUTER_DATA_DIR) + workspace = os.environ.get("OPENSWARM_RUN_WORKSPACE", DEFAULT_RUN_WORKSPACE) + port = int(os.environ.get("OPENSWARM_PORT", str(DEFAULT_PORT))) + + write_router_db(router_data_dir, spec.credentials, now) + seed_data_root(data_root, workspace, spec) + write_skills(os.path.expanduser("~"), spec.skills) + logger.info("seeded data root %s and router db in %s", data_root, router_data_dir) + + backend: Optional[BackendProcess] = None + process: Optional[subprocess.Popen] = None + renderer: Optional[RendererProcess] = None + try: + backend = start_backend(app_root, data_root, port, deadline) + process = backend.process + logger.info("backend healthy at %s", backend.base_url) + + if spec.needs_browser: + renderer = start_renderer( + app_root=app_root, + frontend_dir=os.environ.get("OPENSWARM_FRONTEND_DIR", DEFAULT_FRONTEND_DIR), + backend_base_url=backend.base_url, + backend_headers=backend.headers(), + backend_port=port, + dashboard_id=CLOUD_RUN_DASHBOARD_ID, + deadline=deadline, + ) + logger.info("renderer attached at %s, browser tools are live", renderer.url) + + send_report(spec.callback, RunReport(run_id=spec.run_id, phase="started", status="running")) + heartbeat = Heartbeat( + run_id=spec.run_id, + interval_seconds=float(spec.callback.heartbeat_seconds) if spec.callback else 30.0, + callback=spec.callback, + ) + outcome = execute_workflow(backend, spec.workflow.id, deadline, heartbeat.maybe_send) + except BackendUnavailable as exc: + return p_fail(spec, "failure", str(exc), EXIT_BACKEND_UNAVAILABLE) + except RendererUnavailable as exc: + # Loud, not silent: a browser workflow that quietly ran without a window produces a + # confident wrong answer, which is worse than no answer. + return p_fail(spec, "failure", str(exc), EXIT_RENDERER_UNAVAILABLE) + except WorkflowRunFailed as exc: + return p_fail(spec, "failure", str(exc), EXIT_WORKFLOW_FAILED, workspace) + finally: + stop_renderer(renderer) + stop_backend(process) + + # Files before the terminal report, always: the callback token is refused the moment the run + # is closed, so this is the only order in which both the files and the receipt can land. + files = deliver_files(spec.callback, workspace, collect(workspace)) + + code = p_exit_code_for(outcome) + logger.info("run %s finished as %s (exit %d)", spec.run_id, outcome.status, code) + send_report(spec.callback, RunReport( + run_id=spec.run_id, + phase="finished", + status=outcome.status, + exit_code=code, + error=outcome.error, + cost_usd=outcome.cost_usd, + answer=outcome.answer, + transcript=outcome.transcript, + system_notices=outcome.system_notices, + files=files, + )) + return code + + +@typechecked +def main() -> int: + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s %(levelname).1s %(name)s: %(message)s", + datefmt="%H:%M:%S", + ) + try: + spec = load_run_spec() + except InvalidRunSpec as exc: + return p_fail(None, "failure", str(exc), EXIT_BAD_SPEC) + + budget = effective_max_run_seconds(spec) + arm_hard_stop(budget + REPORT_GRACE_SECONDS) + deadline = time.monotonic() + budget + try: + return p_run(spec, deadline) + except Exception as exc: + logger.exception("runner crashed") + # Re-uploading a file the successful path already sent is harmless: the control plane keys + # a run's files on their path, so a second delivery overwrites one row rather than billing + # the budget twice. + return p_fail( + spec, + "failure", + f"runner crashed: {exc}", + EXIT_INTERNAL, + os.environ.get("OPENSWARM_RUN_WORKSPACE", DEFAULT_RUN_WORKSPACE), + ) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/openswarm-runner/runner/results/__init__.py b/openswarm-runner/runner/results/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/openswarm-runner/runner/results/deliverables.py b/openswarm-runner/runner/results/deliverables.py new file mode 100644 index 00000000..5aaba20f --- /dev/null +++ b/openswarm-runner/runner/results/deliverables.py @@ -0,0 +1,193 @@ +"""What the run made, and what of it is allowed to come home. + +A cloud run's machine is destroyed the moment it exits, so a file it wrote is gone +unless something carries it out. This module is the "what": it walks the one +directory a run is given as its working folder and decides, per file, deliver or +refuse. The "how" (handing bytes to the control plane) lives in runner.results.report. + +Refusing loudly is the whole point of the caps. A user who asked for a video and +got a 20MB fragment of one is worse off than a user who was told the video was too +big, so nothing here ever truncates a file: it either arrives whole or it arrives +as a sentence explaining why it did not. +""" + +import hashlib +import logging +import os +from typing import List, Optional, Tuple + +from pydantic import BaseModel, ConfigDict, Field +from typeguard import typechecked + +logger = logging.getLogger(__name__) + +# Per-file ceiling. Deliverables are reports, spreadsheets, charts and small archives; a run that +# produces something bigger is doing a different job than this pipe was built for. +MAX_FILE_BYTES = 20 * 1024 * 1024 +# Per-run ceiling, enforced in walk order so the first files still arrive when a later one blows it. +MAX_TOTAL_BYTES = 50 * 1024 * 1024 +MAX_FILES = 40 +# Longest path we will accept, so a deep tree cannot produce a name no filesystem will take back. +MAX_RELATIVE_PATH_CHARS = 180 + +# Machinery, not deliverables. Everything here is either regenerable (dependencies, caches, +# compiled bytecode) or the run's own plumbing, and shipping it would blow the file budget on +# things nobody asked for. +EXCLUDED_DIRS = frozenset({ + ".git", + ".claude", + "node_modules", + "__pycache__", + ".venv", + "venv", + ".pytest_cache", + ".ruff_cache", + ".mypy_cache", + ".cache", + ".npm", +}) +EXCLUDED_NAMES = frozenset({".DS_Store", ".gitignore", ".gitkeep"}) + +READ_CHUNK_BYTES = 1024 * 1024 + + +class Deliverable(BaseModel): + """One file that fits, addressed by its path relative to the run's workspace.""" + + model_config = ConfigDict(validate_assignment=True) + + path: str + size_bytes: int + sha256: str + + +class Refused(BaseModel): + """One file that does not come home, and the sentence the user gets instead.""" + + model_config = ConfigDict(validate_assignment=True) + + path: str + size_bytes: int + reason: str + + +class Harvest(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + files: List[Deliverable] = Field(default_factory=list) + refused: List[Refused] = Field(default_factory=list) + + @typechecked + def total_bytes(self) -> int: + return sum(item.size_bytes for item in self.files) + + +@typechecked +def human_bytes(count: int) -> str: + if count < 1024: + return f"{count} B" + if count < 1024 * 1024: + return f"{count / 1024:.0f} KB" + if count < 1024 * 1024 * 1024: + return f"{count / (1024 * 1024):.1f} MB" + return f"{count / (1024 * 1024 * 1024):.1f} GB" + + +@typechecked +def p_digest(path: str) -> Optional[str]: + """sha256, streamed. None when the file went away mid-walk, which is not an error.""" + digest = hashlib.sha256() + try: + with open(path, "rb") as handle: + while True: + chunk = handle.read(READ_CHUNK_BYTES) + if not chunk: + break + digest.update(chunk) + except OSError as exc: + logger.warning("could not read %s while harvesting: %s", path, exc) + return None + return digest.hexdigest() + + +@typechecked +def p_walk(workspace: str) -> List[Tuple[str, int]]: + """Every candidate file under the workspace as (relative path, size), sorted for determinism.""" + found: List[Tuple[str, int]] = [] + for directory, subdirs, filenames in os.walk(workspace): + subdirs[:] = sorted(name for name in subdirs if name not in EXCLUDED_DIRS) + for filename in sorted(filenames): + if filename in EXCLUDED_NAMES: + continue + absolute = os.path.join(directory, filename) + # Symlinks are not followed: a run that linked to /etc/passwd must not exfiltrate it. + if os.path.islink(absolute) or not os.path.isfile(absolute): + continue + try: + size = os.path.getsize(absolute) + except OSError: + continue + found.append((os.path.relpath(absolute, workspace), size)) + return found + + +@typechecked +def collect(workspace: str) -> Harvest: + """Decide, for every file in the run's workspace, whether it comes home.""" + harvest = Harvest() + if not os.path.isdir(workspace): + return harvest + + running_total = 0 + for relative, size in p_walk(workspace): + if len(relative) > MAX_RELATIVE_PATH_CHARS: + harvest.refused.append(Refused( + path=relative[:MAX_RELATIVE_PATH_CHARS] + "...", + size_bytes=size, + reason="its path is too long to save anywhere", + )) + continue + if size == 0: + continue + if size > MAX_FILE_BYTES: + harvest.refused.append(Refused( + path=relative, + size_bytes=size, + reason=( + f"it is {human_bytes(size)} and a single cloud-run file cannot exceed " + f"{human_bytes(MAX_FILE_BYTES)}" + ), + )) + continue + if len(harvest.files) >= MAX_FILES: + harvest.refused.append(Refused( + path=relative, + size_bytes=size, + reason=f"this run already produced the maximum of {MAX_FILES} files", + )) + continue + if running_total + size > MAX_TOTAL_BYTES: + harvest.refused.append(Refused( + path=relative, + size_bytes=size, + reason=( + f"the run's files already total {human_bytes(running_total)} and the limit " + f"is {human_bytes(MAX_TOTAL_BYTES)}" + ), + )) + continue + digest = p_digest(os.path.join(workspace, relative)) + if digest is None: + harvest.refused.append(Refused(path=relative, size_bytes=size, reason="it could not be read")) + continue + harvest.files.append(Deliverable(path=relative, size_bytes=size, sha256=digest)) + running_total += size + + logger.info( + "harvested %d file(s) totalling %s from %s, refused %d", + len(harvest.files), + human_bytes(running_total), + workspace, + len(harvest.refused), + ) + return harvest diff --git a/openswarm-runner/runner/results/report.py b/openswarm-runner/runner/results/report.py new file mode 100644 index 00000000..ca769138 --- /dev/null +++ b/openswarm-runner/runner/results/report.py @@ -0,0 +1,168 @@ +"""Tell the control plane what happened, and hand it whatever the run made. + +The terminal report is the run's only receipt. Files go up BEFORE it, on purpose: +the per-run callback token stops working the instant the run reaches a terminal +state, so "upload, then close" is the only order in which both can succeed, and it +means the window for writing files to a run closes exactly when the run does. +""" + +import logging +import os +import time +from typing import List, Literal, Optional +from urllib.parse import quote + +import httpx +from pydantic import BaseModel, ConfigDict, Field +from typeguard import typechecked + +from runner.results.deliverables import Harvest, human_bytes +from runner.run_spec import CallbackTarget + +logger = logging.getLogger(__name__) + +TERMINAL_ATTEMPTS = 5 +TERMINAL_BACKOFF_SECONDS = 2.0 +REQUEST_TIMEOUT_SECONDS = 15.0 +# One file, one shot, generous: 20MB over a cold uplink is slower than any report. +UPLOAD_TIMEOUT_SECONDS = 120.0 +FILE_PATH_HEADER = "X-Openswarm-File-Path" +FILE_SHA256_HEADER = "X-Openswarm-File-Sha256" + + +class ReportedFile(BaseModel): + """One file the run produced, delivered or not, always named.""" + + model_config = ConfigDict(validate_assignment=True) + + path: str + size_bytes: int + delivered: bool + # Present only when delivered is False. Written for a human, because it is shown to one. + reason: Optional[str] = None + + +class RunReport(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + run_id: str + phase: Literal["started", "heartbeat", "finished"] + status: str + exit_code: Optional[int] = None + error: Optional[str] = None + cost_usd: float = 0.0 + active_step_idx: Optional[int] = None + last_tool_label: Optional[str] = None + answer: str = "" + transcript: str = "" + # The backend calls a run "success" even when the provider rejected the token; these are how the control plane sees that. + system_notices: List[str] = Field(default_factory=list) + # Every file the run made, including the ones that were too big to send. A deliverable that + # silently vanished is the failure this list exists to make impossible. + files: List[ReportedFile] = Field(default_factory=list) + + +@typechecked +def p_post_once(callback: CallbackTarget, report: RunReport) -> bool: + try: + with httpx.Client(timeout=REQUEST_TIMEOUT_SECONDS) as client: + response = client.post( + callback.url, + headers={"Authorization": f"Bearer {callback.token}"}, + json=report.model_dump(mode="json"), + ) + if response.status_code < 300: + return True + logger.warning("report %s rejected with HTTP %s", report.phase, response.status_code) + return False + except httpx.HTTPError as exc: + logger.warning("report %s failed to send: %s", report.phase, exc) + return False + + +@typechecked +def send_report(callback: Optional[CallbackTarget], report: RunReport) -> bool: + """Post a report. Terminal reports retry; heartbeats get one shot and are never retried.""" + if callback is None: + logger.info("no callback configured; %s report kept local: %s", report.phase, report.status) + return True + if report.phase != "finished": + return p_post_once(callback, report) + for attempt in range(TERMINAL_ATTEMPTS): + if p_post_once(callback, report): + return True + if attempt + 1 < TERMINAL_ATTEMPTS: + time.sleep(TERMINAL_BACKOFF_SECONDS * (attempt + 1)) + logger.error("terminal report for run %s never landed after %d attempts", report.run_id, TERMINAL_ATTEMPTS) + return False + + +@typechecked +def p_upload_one(client: httpx.Client, callback: CallbackTarget, workspace: str, relative: str, sha256: str) -> Optional[str]: + """Push one file. Returns None on success, or the sentence explaining the failure.""" + try: + with open(os.path.join(workspace, relative), "rb") as handle: + response = client.post( + str(callback.artifacts_url), + headers={ + "Authorization": f"Bearer {callback.token}", + "Content-Type": "application/octet-stream", + FILE_PATH_HEADER: quote(relative, safe="/"), + FILE_SHA256_HEADER: sha256, + }, + content=handle.read(), + ) + except OSError as exc: + return f"it could not be read back off disk ({exc.strerror or exc})" + except httpx.HTTPError as exc: + return f"the upload did not complete ({type(exc).__name__})" + if response.status_code < 300: + return None + # The control plane refuses with prose it wrote for the user; keep its words rather than ours. + detail = (response.text or "").strip() + try: + body = response.json() + if isinstance(body, dict) and isinstance(body.get("error"), str): + detail = body["error"] + except ValueError: + pass + return detail[:300] if detail else f"the storage service answered HTTP {response.status_code}" + + +@typechecked +def deliver_files(callback: Optional[CallbackTarget], workspace: str, harvest: Harvest) -> List[ReportedFile]: + """Hand every deliverable to the control plane and report honestly on each one. + + Never raises and never fails the run: a workflow whose answer is good and whose + attachment did not make it should still deliver the answer, with the miss stated. + """ + reported = [ + ReportedFile(path=item.path, size_bytes=item.size_bytes, delivered=False, reason=item.reason) + for item in harvest.refused + ] + if not harvest.files: + return reported + if callback is None or not callback.artifacts_url: + for item in harvest.files: + reported.append(ReportedFile( + path=item.path, + size_bytes=item.size_bytes, + delivered=False, + reason="this run had nowhere to send files, so it kept them on the machine", + )) + return reported + + with httpx.Client(timeout=UPLOAD_TIMEOUT_SECONDS) as client: + for item in harvest.files: + failure = p_upload_one(client, callback, workspace, item.path, item.sha256) + if failure is None: + logger.info("delivered %s (%s)", item.path, human_bytes(item.size_bytes)) + else: + logger.warning("could not deliver %s: %s", item.path, failure) + reported.append(ReportedFile( + path=item.path, + size_bytes=item.size_bytes, + delivered=failure is None, + reason=failure, + )) + return reported diff --git a/openswarm-runner/runner/run_spec.py b/openswarm-runner/runner/run_spec.py new file mode 100644 index 00000000..db6487e7 --- /dev/null +++ b/openswarm-runner/runner/run_spec.py @@ -0,0 +1,215 @@ +"""Typed description of the single workflow run this container exists to execute. + +The control plane hands the container exactly one of these (JSON in +OPENSWARM_RUN_SPEC, or a path in OPENSWARM_RUN_SPEC_FILE) and nothing else. Every +field is validated before the backend boots, so a malformed job dies in under a +second instead of burning a machine-minute discovering it. +""" + +import json +import os +from datetime import datetime, timedelta, timezone +from typing import List, Literal, Optional + +from pydantic import BaseModel, ConfigDict, Field, model_validator +from typeguard import typechecked + +from backend.apps.workflows.models import Workflow + +SPEC_ENV = "OPENSWARM_RUN_SPEC" +SPEC_FILE_ENV = "OPENSWARM_RUN_SPEC_FILE" + +# The one dashboard a cloud run has. Fixed rather than generated so the Electron window can be +# deep-linked at it before the backend has even booted, and so a workflow arriving with the +# laptop dashboard id it was authored against gets repointed at a dashboard that exists here. +CLOUD_RUN_DASHBOARD_ID = "cloud-run" + +# Headroom the access token must still have on arrival. The control plane refreshes right before dispatch; anything thinner than this means its clock or its queue is broken, and we must not paper over that by refreshing ourselves. +MIN_TOKEN_LIFETIME = timedelta(minutes=2) + +# A skill is prose plus the odd small script. These bounds keep a run spec from becoming a file +# transfer, and are enforced here so an oversized one dies before a machine is billed for it. +MAX_SKILLS = 60 +MAX_SKILL_FILE_CHARS = 200_000 + + +class InvalidRunSpec(ValueError): + """The control plane handed us something we refuse to run.""" + + +class ProviderCredential(BaseModel): + """One already-refreshed provider credential, spendable but not rotatable. + + `extra="forbid"` is the first of two walls keeping a refresh token out of this + container: a payload carrying `refreshToken` fails validation here and the run + dies loudly. See runner.router_credentials for the second wall and the why. + """ + + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + provider: str = Field(min_length=1) + auth_type: Literal["oauth", "api_key"] + label: str = "OpenSwarm cloud run" + access_token: Optional[str] = None + api_key: Optional[str] = None + expires_at: Optional[datetime] = None + scope: Optional[str] = None + + @model_validator(mode="after") + def p_require_matching_secret(self) -> "ProviderCredential": + if self.auth_type == "oauth": + if not self.access_token: + raise ValueError(f"credential for {self.provider!r} is oauth but carries no access_token") + if self.expires_at is None: + raise ValueError(f"credential for {self.provider!r} is oauth but carries no expires_at") + if self.api_key: + raise ValueError(f"credential for {self.provider!r} carries both an access_token and an api_key") + else: + if not self.api_key: + raise ValueError(f"credential for {self.provider!r} is api_key but carries no api_key") + if self.access_token: + raise ValueError(f"credential for {self.provider!r} carries both an access_token and an api_key") + return self + + @typechecked + def remaining_lifetime(self, now: datetime) -> Optional[timedelta]: + """How long this credential is still good for; None when it cannot expire.""" + if self.expires_at is None: + return None + return self.expires_at.astimezone(timezone.utc) - now.astimezone(timezone.utc) + + +class CallbackTarget(BaseModel): + """Where the run reports back. The token is a dedicated two-party secret, never a user credential.""" + + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + url: str = Field(min_length=1) + token: str = Field(min_length=1) + heartbeat_seconds: int = Field(default=30, ge=5, le=300) + # Where the run's files go. Named outright rather than derived from `url`, so a control plane + # that cannot accept files says so by leaving it out instead of being guessed at. + artifacts_url: Optional[str] = None + + +class SkillFile(BaseModel): + """One file inside a skill folder. Text only: a skill is prose plus small scripts.""" + + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + path: str = Field(min_length=1, max_length=180) + text: str = Field(max_length=MAX_SKILL_FILE_CHARS) + + @model_validator(mode="after") + def p_reject_escaping_path(self) -> "SkillFile": + parts = self.path.split("/") + if self.path.startswith("/") or "\\" in self.path or any(p in ("", ".", "..") for p in parts): + raise ValueError(f"skill file path {self.path!r} is not a plain relative path") + return self + + +class SkillPayload(BaseModel): + """One of the user's skills, carried up so the agent has the same know-how it has at home. + + Skills are the user's own writing, not credentials. Nothing in here is a token, and the + control plane never learns anything from it that it could spend. + """ + + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + # Also the folder name under ~/.claude/skills, so it has to survive being a directory. + id: str = Field(min_length=1, max_length=80, pattern=r"^[A-Za-z0-9][A-Za-z0-9_-]*$") + files: List[SkillFile] = Field(min_length=1) + + @model_validator(mode="after") + def p_require_skill_md(self) -> "SkillPayload": + if not any(file.path == "SKILL.md" for file in self.files): + raise ValueError(f"skill {self.id!r} has no SKILL.md, so nothing would ever load it") + return self + + +class McpServerNote(BaseModel): + """A server the user has connected at home, named so the agent can say it cannot reach it. + + Deliberately carries NO transport and NO secret. The user's MCP credentials (Slack session + cookies, Notion and GitHub access tokens, Google refresh tokens) stay on their laptop, so this + is a list of names and nothing else. Its whole job is to stop the agent quietly answering + "update my Notion" from general knowledge because it never knew Notion existed. + """ + + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + name: str = Field(min_length=1, max_length=120) + + +class RunSpec(BaseModel): + model_config = ConfigDict(validate_assignment=True, extra="forbid") + + run_id: str = Field(min_length=1) + workflow: Workflow + credentials: List[ProviderCredential] = Field(min_length=1) + callback: Optional[CallbackTarget] = None + # The user's skills, shipped so a cloud run is as capable as the same workflow at home. + skills: List[SkillPayload] = Field(default_factory=list, max_length=MAX_SKILLS) + # Names only. See McpServerNote for why there is no config here. + unavailable_mcp_servers: List[McpServerNote] = Field(default_factory=list, max_length=100) + # Hard wall-clock ceiling. Fly bills by machine-second, so an agent that wedges must cost a bounded amount. + max_run_seconds: int = Field(default=1800, ge=60, le=7200) + # Boot Electron under a virtual display so browser steps work. On by default: parity is the + # point of running in a container at all, and a workflow that never touches a browser is the + # exception that should have to say so. Costs a few seconds and a few hundred MB when on. + needs_browser: bool = True + + @typechecked + def expired_credentials(self, now: datetime) -> List[ProviderCredential]: + """Credentials too close to expiry to spend. The runner cannot refresh, so this is fatal, not a retry.""" + stale: List[ProviderCredential] = [] + for credential in self.credentials: + remaining = credential.remaining_lifetime(now) + if remaining is not None and remaining < MIN_TOKEN_LIFETIME: + stale.append(credential) + return stale + + @typechecked + def workflow_for_disk(self) -> Workflow: + """The workflow as this container should see it: one run, never a schedule. + + A cloud-executed workflow arrives with its schedule still configured. Left + enabled, the container's own scheduler would fire it a second time inside + the box, so the timer is stripped here rather than trusted to stay off. + + It also arrives pointing at whatever dashboard it was authored on, which does not + exist in this container; left alone, the first browser card would 404 looking for + it. Repointed at the one dashboard this run has. + """ + copy = self.workflow.model_copy(deep=True) + copy.dashboard_id = CLOUD_RUN_DASHBOARD_ID + copy.schedule.enabled = False + copy.deleted_at = None + copy.draft_steps = None + copy.next_run_at = None + return copy + + +@typechecked +def load_run_spec() -> RunSpec: + """Parse the run spec from the environment, or raise InvalidRunSpec with a legible reason.""" + raw = os.environ.get(SPEC_ENV, "").strip() + source = SPEC_ENV + if not raw: + path = os.environ.get(SPEC_FILE_ENV, "").strip() + if not path: + raise InvalidRunSpec(f"no run spec: set {SPEC_ENV} to JSON or {SPEC_FILE_ENV} to a file path") + source = f"{SPEC_FILE_ENV}={path}" + try: + with open(path, "r", encoding="utf-8") as handle: + raw = handle.read() + except OSError as exc: + raise InvalidRunSpec(f"cannot read run spec from {source}: {exc}") from exc + + try: + return RunSpec.model_validate_json(raw) + except json.JSONDecodeError as exc: + raise InvalidRunSpec(f"run spec from {source} is not valid JSON: {exc}") from exc + except ValueError as exc: + raise InvalidRunSpec(f"run spec from {source} is not a valid RunSpec: {exc}") from exc diff --git a/openswarm-runner/runner/seed/__init__.py b/openswarm-runner/runner/seed/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/openswarm-runner/runner/seed/data_root.py b/openswarm-runner/runner/seed/data_root.py new file mode 100644 index 00000000..7741ff7c --- /dev/null +++ b/openswarm-runner/runner/seed/data_root.py @@ -0,0 +1,137 @@ +"""Lay down the backend's data dir before it boots, so the run is ready on the first tick. + +Everything here is written pre-boot on purpose: the workflow store and the settings +store both load from disk once at startup, so seeding files is cheaper and more +deterministic than replaying create/PATCH calls over HTTP (no aux LLM naming call, +no schedule normalization, no chance of the container inventing a second workflow). +""" + +import json +import os +import tempfile +from typing import Any, Dict + +from typeguard import typechecked + +from backend.apps.dashboards.models import Dashboard +from backend.apps.settings.models import AppSettings +from runner.run_spec import CLOUD_RUN_DASHBOARD_ID, RunSpec + +# 9Router provider id -> the AppSettings field the backend reads a raw key from. +API_KEY_SETTINGS_FIELD: Dict[str, str] = { + "anthropic": "anthropic_api_key", + "openai": "openai_api_key", + "gemini": "google_api_key", + "google": "google_api_key", + "openrouter": "openrouter_api_key", +} + +# One synthetic id for every cloud run. Without it each ephemeral container mints a fresh uuid and analytics sees a brand-new "install" per run. +CLOUD_RUNNER_INSTALLATION_ID = "openswarm-cloud-runner" + +# Told to the agent in as many words, because it cannot find this out any other way and the +# consequence of not knowing is a report written to a folder that is deleted minutes later. +DELIVERY_NOTE = ( + "You are running in the OpenSwarm cloud on a throwaway machine. Your working directory is " + "the ONLY place that survives: every file you save there is delivered back to the user, and " + "everything else on this machine is destroyed the moment this run ends. So when a task asks " + "for a document, spreadsheet, image or archive, write it to a plainly named file in your " + "working directory rather than only pasting it into your reply. Do not write deliverables to " + "/tmp or to your home directory; they will not come back." +) + + +@typechecked +def p_write_json(path: str, payload: Any) -> None: + """Atomic, owner-only write; these files hold API keys.""" + directory = os.path.dirname(path) + os.makedirs(directory, mode=0o700, exist_ok=True) + handle, temp_path = tempfile.mkstemp(dir=directory, prefix=".seed-", suffix=".json") + try: + with os.fdopen(handle, "w", encoding="utf-8") as stream: + json.dump(payload, stream, indent=2) + os.chmod(temp_path, 0o600) + os.replace(temp_path, path) + except BaseException: + if os.path.exists(temp_path): + os.unlink(temp_path) + raise + + +@typechecked +def unavailable_apps_note(spec: RunSpec) -> str: + """Name the user's connected apps this run cannot reach, so silence is not mistaken for absence. + + Their MCP credentials never leave the laptop, so the servers are not here and never will be + mid-run. Without this sentence the agent has no way to know the app exists, and "update my + Notion" comes back as a confident paragraph about Notion rather than an admission. + """ + names = [server.name for server in spec.unavailable_mcp_servers] + if not names: + return "" + return ( + "These apps are connected on the user's own computer but NOT reachable from this cloud " + f"run, because their sign-in details stay on that computer: {', '.join(sorted(names))}. " + "If a task needs one of them, say plainly that it cannot be done from a cloud run and " + "that it has to run on their machine. Never guess at, invent, or describe from memory " + "what one of those apps contains." + ) + + +@typechecked +def settings_for_run(spec: RunSpec, workspace: str) -> AppSettings: + """The AppSettings a cloud run needs: this workflow's model, this run's keys, no telemetry. + + `default_folder` is what makes the run's files findable afterwards. Left unset, the agent + falls back to $HOME and the launcher reroutes it into a per-session scratch directory whose + name nothing outside the backend can predict, so the harvest would have nowhere to look. + """ + settings = AppSettings() + settings.default_model = spec.workflow.model + settings.connection_mode = "own_key" + settings.analytics_opt_in = False + settings.installation_id = CLOUD_RUNNER_INSTALLATION_ID + settings.default_folder = workspace + additions = [DELIVERY_NOTE, unavailable_apps_note(spec)] + settings.default_system_prompt = "\n\n".join( + part for part in [settings.default_system_prompt or "", *additions] if part + ).strip() + for credential in spec.credentials: + if credential.auth_type != "api_key": + continue + field = API_KEY_SETTINGS_FIELD.get(credential.provider) + if field is None: + raise ValueError( + f"no settings field for api_key provider {credential.provider!r}; " + f"supported: {', '.join(sorted(set(API_KEY_SETTINGS_FIELD)))}" + ) + setattr(settings, field, credential.api_key) + return settings + + +@typechecked +def seed_data_root(data_root: str, workspace: str, spec: RunSpec) -> None: + """Write the workflow, settings and dashboard records the backend will read at boot. + + The dashboard exists so the Electron window has somewhere to land and browser cards have + somewhere to render. Writing it here rather than letting the backend's first-boot migration + invent one keeps its id knowable before anything has started. + + The workspace sits OUTSIDE the data root deliberately: it is the agent's own folder, and a + Glob or Grep run inside it should not sweep up the settings file its API keys live in. + """ + os.makedirs(workspace, mode=0o700, exist_ok=True) + workflow = spec.workflow_for_disk() + p_write_json( + os.path.join(data_root, "workflows", f"{workflow.id}.json"), + workflow.model_dump(mode="json"), + ) + p_write_json( + os.path.join(data_root, "settings", "settings.json"), + settings_for_run(spec, workspace).model_dump(mode="json"), + ) + dashboard = Dashboard(id=CLOUD_RUN_DASHBOARD_ID, name=spec.workflow.title or "Cloud run") + p_write_json( + os.path.join(data_root, "dashboards", f"{dashboard.id}.json"), + dashboard.model_dump(mode="json"), + ) diff --git a/openswarm-runner/runner/seed/router_credentials.py b/openswarm-runner/runner/seed/router_credentials.py new file mode 100644 index 00000000..5d2d719f --- /dev/null +++ b/openswarm-runner/runner/seed/router_credentials.py @@ -0,0 +1,128 @@ +"""Write 9Router's credential db for one cloud run, without a refresh token. Ever. + +This container must be structurally incapable of rotating the user's OAuth grant. +9Router's refresh dispatcher bails on `if (!b || !b.refreshToken) return null`, so a +providerConnections entry with no such field is one it can spend and never rotate. +If the runner did rotate, the user's laptop would be left replaying a dead token and +the provider would revoke their entire grant family. That is the whole safety +property of this file, not a style preference. + +Two independent walls hold it up: + 1. ProviderCredential forbids extra fields, so a payload carrying `refreshToken` + never parses into the process at all. + 2. assert_no_refresh_token re-reads the assembled payload just before the write + and refuses anything whose key names a refresh token, however it got there. +""" + +import json +import os +import tempfile +from datetime import datetime, timezone +from typing import Any, Dict, List +from uuid import uuid4 + +from typeguard import typechecked + +from runner.run_spec import ProviderCredential + +DB_FILENAME = "db.json" + +# Normalized key fragment that must never appear anywhere in the db we write. +FORBIDDEN_KEY_FRAGMENT = "refreshtoken" + + +class RefreshTokenLeak(RuntimeError): + """A refresh token reached the credential writer. Fail the run rather than write it.""" + + +@typechecked +def p_normalize_key(key: str) -> str: + return "".join(char for char in key.lower() if char.isalnum()) + + +@typechecked +def p_iso(moment: datetime) -> str: + """9Router timestamps are ISO-8601 UTC with a Z suffix; match it exactly.""" + return moment.astimezone(timezone.utc).isoformat(timespec="milliseconds").replace("+00:00", "Z") + + +@typechecked +def assert_no_refresh_token(payload: Any, path: str = "$") -> None: + """Raise if any key anywhere under `payload` names a refresh token.""" + if isinstance(payload, dict): + for key, value in payload.items(): + if FORBIDDEN_KEY_FRAGMENT in p_normalize_key(str(key)): + raise RefreshTokenLeak( + f"refusing to write a 9Router db containing a refresh token at {path}.{key}" + ) + assert_no_refresh_token(value, f"{path}.{key}") + elif isinstance(payload, list): + for index, value in enumerate(payload): + assert_no_refresh_token(value, f"{path}[{index}]") + + +@typechecked +def router_connection(credential: ProviderCredential, now: datetime) -> Dict[str, Any]: + """Build one providerConnections entry from an allow-list of keys, never a passthrough.""" + if credential.auth_type != "oauth": + raise ValueError(f"credential for {credential.provider!r} is not an oauth connection") + entry: Dict[str, Any] = { + "id": str(uuid4()), + "provider": credential.provider, + "authType": "oauth", + "name": credential.label, + "priority": 1, + "isActive": True, + "createdAt": p_iso(now), + "updatedAt": p_iso(now), + "accessToken": credential.access_token, + "testStatus": "active", + } + if credential.expires_at is not None: + entry["expiresAt"] = p_iso(credential.expires_at) + if credential.scope: + entry["scope"] = credential.scope + return entry + + +@typechecked +def router_db_payload(credentials: List[ProviderCredential], now: datetime) -> Dict[str, Any]: + """A complete 9Router db seeded with this run's subscription connections and nothing else.""" + return { + "providerConnections": [ + router_connection(credential, now) + for credential in credentials + if credential.auth_type == "oauth" + ], + "providerNodes": [], + "proxyPools": [], + "modelAliases": {}, + "mitmAlias": {}, + "combos": [], + "apiKeys": [], + "customModels": [], + "pricing": {}, + "settings": {}, + } + + +@typechecked +def write_router_db(data_dir: str, credentials: List[ProviderCredential], now: datetime) -> str: + """Write $DATA_DIR/db.json owner-only and return its path.""" + payload = router_db_payload(credentials, now) + assert_no_refresh_token(payload) + + os.makedirs(data_dir, mode=0o700, exist_ok=True) + os.chmod(data_dir, 0o700) + path = os.path.join(data_dir, DB_FILENAME) + handle, temp_path = tempfile.mkstemp(dir=data_dir, prefix=".db-", suffix=".json") + try: + with os.fdopen(handle, "w", encoding="utf-8") as stream: + json.dump(payload, stream, indent=2) + os.chmod(temp_path, 0o600) + os.replace(temp_path, path) + except BaseException: + if os.path.exists(temp_path): + os.unlink(temp_path) + raise + return path diff --git a/openswarm-runner/runner/seed/skills.py b/openswarm-runner/runner/seed/skills.py new file mode 100644 index 00000000..e0399b01 --- /dev/null +++ b/openswarm-runner/runner/seed/skills.py @@ -0,0 +1,52 @@ +"""Lay the user's skills down where the backend looks for them, before it boots. + +backend/apps/skills/skills.py hardwires SKILLS_DIR to ~/.claude/skills and gates the whole +Skill tool on at least one non-built-in skill existing there. A container that ships without +them does not get a degraded Skill tool, it gets no Skill tool at all, and the agent then +answers from general knowledge in a voice that sounds exactly as confident as the real thing. + +Written pre-boot for the same reason the workflow is: the skill index is read once at startup. +""" + +import logging +import os +from typing import List + +from typeguard import typechecked + +from runner.run_spec import SkillPayload + +logger = logging.getLogger(__name__) + +SKILLS_DIRNAME = os.path.join(".claude", "skills") + + +@typechecked +def skills_dir(home: str) -> str: + return os.path.join(home, SKILLS_DIRNAME) + + +@typechecked +def write_skills(home: str, skills: List[SkillPayload]) -> int: + """Write every skill folder. Returns how many landed. + + Paths were already proven relative and non-escaping by SkillFile's validator; this re-checks + the joined result anyway, because the one place a path traversal is worth catching twice is + the line that actually opens the file. + """ + root = skills_dir(home) + os.makedirs(root, mode=0o700, exist_ok=True) + written = 0 + for skill in skills: + folder = os.path.join(root, skill.id) + for file in skill.files: + target = os.path.abspath(os.path.join(folder, file.path)) + if not target.startswith(os.path.abspath(folder) + os.sep): + raise ValueError(f"skill {skill.id!r} file {file.path!r} resolves outside its own folder") + os.makedirs(os.path.dirname(target), mode=0o700, exist_ok=True) + with open(target, "w", encoding="utf-8") as handle: + handle.write(file.text) + written += 1 + if written: + logger.info("seeded %d skill(s) into %s", written, root) + return written diff --git a/openswarm-runner/runner/workflow_run.py b/openswarm-runner/runner/workflow_run.py new file mode 100644 index 00000000..2e516f90 --- /dev/null +++ b/openswarm-runner/runner/workflow_run.py @@ -0,0 +1,273 @@ +"""Drive one workflow through the backend's own HTTP surface and collect its result. + +Deliberately no shortcuts into agent_manager: the cloud run fires the same route the +Run button fires, so the MCP gate, action filtering, provider routing and history all +behave exactly as they do on a laptop. +""" + +import json +import logging +import time +from typing import Any, Callable, Dict, List, Optional + +import httpx +from pydantic import BaseModel, ConfigDict, Field +from typeguard import typechecked + +from runner.boot.backend_process import BackendProcess + +logger = logging.getLogger(__name__) + +TERMINAL_STATUSES = ("success", "failure", "ran_late", "skipped") +POLL_INTERVAL_SECONDS = 1.0 +TRANSCRIPT_MAX_CHARS = 14000 +# The backend is one process on one event loop, and a heavy tool call can hold it: MCP registry +# work has been measured keeping it from answering for well over a minute. A reply that is late +# therefore means busy, not dead, so the budget is generous and lateness is never fatal. The only +# thing that ends a run early is the process actually exiting. +REQUEST_TIMEOUT_SECONDS = 60.0 +TRIGGER_RETRY_SECONDS = 5.0 + + +class WorkflowRunFailed(RuntimeError): + """The backend refused to start the run at all.""" + + +class RunProgress(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + run_id: str + status: str + active_step_idx: Optional[int] = None + last_tool_label: Optional[str] = None + + +class RunOutcome(BaseModel): + model_config = ConfigDict(validate_assignment=True) + + run_id: str + status: str + error: Optional[str] = None + cost_usd: float = 0.0 + session_id: Optional[str] = None + transcript: str = "" + answer: str = "" + system_notices: List[str] = Field(default_factory=list) + + +@typechecked +def p_block_text(block: Dict[str, Any]) -> str: + kind = block.get("type") + if kind == "text": + return str(block.get("text") or "") + if kind == "tool_use": + return f"[tool {block.get('name')}] {json.dumps(block.get('input') or {})[:300]}" + if kind == "tool_result": + inner = block.get("content") + return f"[result] {inner if isinstance(inner, str) else json.dumps(inner)[:300]}" + return "" + + +@typechecked +def p_message_text(message: Dict[str, Any]) -> str: + content = message.get("content") + if isinstance(content, str): + return content + if isinstance(content, list): + parts = [p_block_text(block) for block in content if isinstance(block, dict)] + return "\n".join(part for part in parts if part) + return "" + + +@typechecked +def render_transcript(messages: List[Dict[str, Any]]) -> str: + """Role-tagged flatten, tail-biased so the end of a long run always survives the cap.""" + lines: List[str] = [] + for message in messages: + if message.get("hidden"): + continue + text = p_message_text(message).strip() + if text: + lines.append(f"{str(message.get('role') or '?').upper()}: {text}") + joined = "\n\n".join(lines) + if len(joined) > TRANSCRIPT_MAX_CHARS: + return "...(earlier turns trimmed)...\n\n" + joined[-TRANSCRIPT_MAX_CHARS:] + return joined + + +@typechecked +def final_answer(messages: List[Dict[str, Any]]) -> str: + """Last visible assistant text: the thing a user actually asked the workflow for.""" + for message in reversed(messages): + if message.get("hidden") or message.get("role") != "assistant": + continue + text = p_message_text(message).strip() + if text: + return text + return "" + + +@typechecked +def system_notices(messages: List[Dict[str, Any]]) -> List[str]: + """Every system-role bubble in the session. + + The backend appends a system message only when something went wrong (a dead + provider token, a run error, a blocked tool), and it does NOT fail the run for + those, so a workflow whose credential was rejected still comes back "success". + Keyed on the typed role, not on the prose, and reported rather than judged: the + control plane decides what a notice means for billing and retries. + """ + notices: List[str] = [] + for message in messages: + if message.get("role") != "system" or message.get("hidden"): + continue + text = p_message_text(message).strip() + if text: + notices.append(text) + return notices + + +@typechecked +def p_get_json(client: httpx.Client, backend: BackendProcess, path: str) -> Dict[str, Any]: + response = client.get(f"{backend.base_url}{path}", headers=backend.headers()) + response.raise_for_status() + payload = response.json() + return payload if isinstance(payload, dict) else {} + + +@typechecked +def p_started_run_id(client: httpx.Client, backend: BackendProcess, workflow_id: str) -> Optional[str]: + """The newest run on record, or None. Safe to adopt: the runs file ships empty in every + container, so anything in this list was started by the POST we just made.""" + try: + body = p_get_json(client, backend, f"/api/workflows/{workflow_id}/runs?limit=1") + except httpx.HTTPError: + return None + for record in body.get("runs") or []: + if isinstance(record, dict) and record.get("id"): + return str(record["id"]) + return None + + +@typechecked +def trigger_run(client: httpx.Client, backend: BackendProcess, workflow_id: str, deadline: float) -> str: + """Start the run, waiting out a backend too busy to answer instead of failing the job. + + A POST whose reply never arrived may still have started the run, so a retry looks for that + run before firing again. Posting blind would either execute the workflow twice or come back + "Previous run still active", and both are worse than waiting. + """ + while True: + try: + response = client.post( + f"{backend.base_url}/api/workflows/{workflow_id}/run", + headers=backend.headers(), + json={}, + ) + response.raise_for_status() + body = response.json() + break + except httpx.HTTPError as exc: + if not backend.is_alive(): + raise WorkflowRunFailed(f"backend died before the run could start: {exc}") from exc + adopted = p_started_run_id(client, backend, workflow_id) + if adopted: + logger.warning("trigger reply never arrived (%s); adopting the run it started", exc) + return adopted + if time.monotonic() >= deadline: + raise WorkflowRunFailed(f"backend never accepted the run trigger: {exc}") from exc + logger.warning("trigger did not answer (%s); backend is busy, retrying", exc) + time.sleep(TRIGGER_RETRY_SECONDS) + + run_id = str(body.get("run_id") or "") + if not run_id: + raise WorkflowRunFailed( + f"backend accepted the trigger but never created a run for workflow {workflow_id}" + ) + if body.get("status") == "failure": + raise WorkflowRunFailed(str(body.get("error") or "run failed immediately")) + return run_id + + +@typechecked +def p_find_run(client: httpx.Client, backend: BackendProcess, workflow_id: str, run_id: str) -> Dict[str, Any]: + body = p_get_json(client, backend, f"/api/workflows/{workflow_id}/runs?limit=50") + for record in body.get("runs") or []: + if isinstance(record, dict) and record.get("id") == run_id: + return record + return {} + + +@typechecked +def p_stop_run(client: httpx.Client, backend: BackendProcess, run_id: str) -> None: + try: + client.post(f"{backend.base_url}/api/workflows/runs/{run_id}/stop", headers=backend.headers()) + except httpx.HTTPError: + pass + + +@typechecked +def p_collect_session(client: httpx.Client, backend: BackendProcess, session_id: str) -> List[Dict[str, Any]]: + try: + body = p_get_json(client, backend, f"/api/agents/sessions/{session_id}") + except httpx.HTTPError: + return [] + messages = body.get("messages") + return [m for m in messages if isinstance(m, dict)] if isinstance(messages, list) else [] + + +@typechecked +def execute_workflow( + backend: BackendProcess, + workflow_id: str, + deadline: float, + on_progress: Optional[Callable[[RunProgress], None]] = None, +) -> RunOutcome: + """Fire the workflow, poll it to a terminal state, and pull the transcript back. + + Blowing the deadline stops the run and reports `timed_out`; the caller still gets + whatever the agent produced before the wall came down. + """ + with httpx.Client(timeout=REQUEST_TIMEOUT_SECONDS) as client: + run_id = trigger_run(client, backend, workflow_id, deadline) + record: Dict[str, Any] = {} + timed_out = False + + while True: + try: + record = p_find_run(client, backend, workflow_id, run_id) or record + except httpx.HTTPError as exc: + # A poll that goes unanswered says the backend is busy, and the run it is busy + # with is this one. Crashing here used to throw away a run that then finished fine. + logger.warning("poll for run %s went unanswered (%s); still waiting", run_id, exc) + status = str(record.get("status") or "running") + if on_progress is not None: + on_progress(RunProgress( + run_id=run_id, + status=status, + active_step_idx=record.get("active_step_idx"), + last_tool_label=record.get("last_tool_label"), + )) + if status in TERMINAL_STATUSES: + break + if not backend.is_alive(): + raise WorkflowRunFailed("backend died while the workflow was running") + if time.monotonic() >= deadline: + timed_out = True + p_stop_run(client, backend, run_id) + record = p_find_run(client, backend, workflow_id, run_id) or record + break + time.sleep(POLL_INTERVAL_SECONDS) + + session_id = record.get("session_id") + messages = p_collect_session(client, backend, str(session_id)) if session_id else [] + return RunOutcome( + run_id=run_id, + status="timed_out" if timed_out else str(record.get("status") or "failure"), + error=("wall-clock cap reached before the workflow finished" if timed_out else record.get("error")), + cost_usd=float(record.get("cost_usd") or 0.0), + session_id=str(session_id) if session_id else None, + transcript=render_transcript(messages), + answer=final_answer(messages), + system_notices=system_notices(messages), + ) diff --git a/openswarm-runner/tests/test_deliverables.py b/openswarm-runner/tests/test_deliverables.py new file mode 100644 index 00000000..becaf220 --- /dev/null +++ b/openswarm-runner/tests/test_deliverables.py @@ -0,0 +1,174 @@ +"""What the run made, what comes home, and what is refused out loud instead of silently. + +The caps are the point. A truncated file is worse than a refused one, and a file that +vanishes with no sentence attached is the failure the whole list exists to prevent. +""" + +import os + +import pytest + +from runner.results.deliverables import ( + MAX_FILE_BYTES, + MAX_FILES, + MAX_TOTAL_BYTES, + collect, + human_bytes, +) +from runner.results.report import deliver_files +from runner.run_spec import CallbackTarget + + +def write(root, relative: str, payload: bytes) -> str: + path = os.path.join(str(root), relative) + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as handle: + handle.write(payload) + return path + + +def test_a_missing_workspace_is_an_empty_harvest_not_a_crash(tmp_path) -> None: + assert collect(str(tmp_path / "never-made")).files == [] + + +def test_ordinary_files_are_collected_with_their_digest(tmp_path) -> None: + write(tmp_path, "report.md", b"# Digest\n") + write(tmp_path, "data/rows.csv", b"a,b\n1,2\n") + + # Walk order, and it is fixed: this directory's own files first, then subdirectories in name + # order, so a run's file list does not shuffle between two identical runs. + harvest = collect(str(tmp_path)) + assert [f.path for f in harvest.files] == ["report.md", "data/rows.csv"] + assert harvest.files[0].size_bytes == len(b"# Digest\n") + # A digest travels with every file, so a corrupted upload is detectable rather than assumed fine. + assert len(harvest.files[0].sha256) == 64 + assert harvest.refused == [] + + +def test_machinery_is_not_a_deliverable(tmp_path) -> None: + write(tmp_path, "report.md", b"keep me") + write(tmp_path, ".git/config", b"[core]") + write(tmp_path, "node_modules/left-pad/index.js", b"module.exports=1") + write(tmp_path, "__pycache__/x.pyc", b"\x00") + write(tmp_path, ".claude/worktrees/probe/README", b"scratch") + + assert [f.path for f in collect(str(tmp_path)).files] == ["report.md"] + + +def test_an_empty_file_is_neither_delivered_nor_complained_about(tmp_path) -> None: + write(tmp_path, "touched.txt", b"") + harvest = collect(str(tmp_path)) + assert harvest.files == [] + assert harvest.refused == [] + + +def test_a_file_over_the_cap_is_refused_whole_and_says_why(tmp_path) -> None: + write(tmp_path, "render.mp4", b"x" * (MAX_FILE_BYTES + 1)) + write(tmp_path, "notes.md", b"still fine") + + harvest = collect(str(tmp_path)) + assert [f.path for f in harvest.files] == ["notes.md"] + assert len(harvest.refused) == 1 + assert harvest.refused[0].path == "render.mp4" + assert "cannot exceed" in harvest.refused[0].reason + # Never a fragment: an over-sized file is absent, not shortened. + assert all(f.path != "render.mp4" for f in harvest.files) + + +def test_the_run_total_stops_collecting_but_keeps_what_already_fit(tmp_path) -> None: + chunk = b"x" * MAX_FILE_BYTES + for index in range(MAX_TOTAL_BYTES // MAX_FILE_BYTES + 1): + write(tmp_path, f"blob-{index}.bin", chunk) + + harvest = collect(str(tmp_path)) + assert harvest.total_bytes() <= MAX_TOTAL_BYTES + assert len(harvest.files) >= 1 + assert harvest.refused, "the file that blew the budget must be named, not dropped" + assert "limit is" in harvest.refused[0].reason + + +def test_too_many_files_refuses_the_extras_by_name(tmp_path) -> None: + for index in range(MAX_FILES + 3): + write(tmp_path, f"note-{index:03d}.txt", b"hi") + + harvest = collect(str(tmp_path)) + assert len(harvest.files) == MAX_FILES + assert len(harvest.refused) == 3 + assert all("maximum of" in item.reason for item in harvest.refused) + + +def test_a_symlink_out_of_the_workspace_is_never_followed(tmp_path) -> None: + secret = tmp_path / "outside" / "id_rsa" + os.makedirs(secret.parent, exist_ok=True) + secret.write_text("PRIVATE KEY") + workspace = tmp_path / "ws" + os.makedirs(workspace, exist_ok=True) + os.symlink(str(secret), str(workspace / "borrowed.pem")) + + assert collect(str(workspace)).files == [] + + +def test_with_nowhere_to_send_files_the_run_says_so_per_file(tmp_path) -> None: + write(tmp_path, "report.md", b"the answer") + reported = deliver_files(None, str(tmp_path), collect(str(tmp_path))) + + assert len(reported) == 1 + assert reported[0].delivered is False + assert "nowhere to send files" in (reported[0].reason or "") + + +def test_a_control_plane_with_no_file_route_is_reported_not_guessed(tmp_path) -> None: + write(tmp_path, "report.md", b"the answer") + callback = CallbackTarget(url="https://cloud.test/report", token="two-party") + assert callback.artifacts_url is None + + reported = deliver_files(callback, str(tmp_path), collect(str(tmp_path))) + assert reported[0].delivered is False + + +def test_refusals_reach_the_report_even_when_nothing_was_delivered(tmp_path) -> None: + write(tmp_path, "render.mp4", b"x" * (MAX_FILE_BYTES + 1)) + reported = deliver_files(None, str(tmp_path), collect(str(tmp_path))) + + assert len(reported) == 1 + assert reported[0].path == "render.mp4" + assert reported[0].delivered is False + assert "cannot exceed" in (reported[0].reason or "") + + +@pytest.mark.parametrize( + "count,expected", + [(512, "512 B"), (2048, "2 KB"), (5 * 1024 * 1024, "5.0 MB"), (3 * 1024**3, "3.0 GB")], +) +def test_sizes_are_written_the_way_a_person_reads_them(count: int, expected: str) -> None: + assert human_bytes(count) == expected + + +def test_a_failed_run_still_hands_over_what_it_managed_to_make(tmp_path, monkeypatch) -> None: + """A workflow that dies on step 3 may have written a perfectly good report on step 1.""" + from runner import main + from runner.run_spec import RunSpec + + write(tmp_path, "partial-report.md", b"# What I got through\n") + spec = RunSpec.model_validate({ + "run_id": "cr-fail", + "workflow": {"id": "wf-1", "steps": [{"id": "s1", "text": "go"}]}, + "credentials": [{"provider": "anthropic", "auth_type": "api_key", "api_key": "sk-test"}], + }) + + sent: list = [] + monkeypatch.setattr(main, "send_report", lambda callback, report: sent.append(report) or True) + code = main.p_fail(spec, "failure", "step 3 blew up", main.EXIT_WORKFLOW_FAILED, str(tmp_path)) + + assert code == main.EXIT_WORKFLOW_FAILED + assert [f.path for f in sent[0].files] == ["partial-report.md"] + + +def test_a_failure_with_no_workspace_reports_no_files_rather_than_guessing(monkeypatch) -> None: + from runner import main + + sent: list = [] + monkeypatch.setattr(main, "send_report", lambda callback, report: sent.append(report) or True) + main.p_fail(None, "failure", "bad spec", main.EXIT_BAD_SPEC) + + assert sent[0].files == [] diff --git a/openswarm-runner/tests/test_renderer_process.py b/openswarm-runner/tests/test_renderer_process.py new file mode 100644 index 00000000..f3eb79e2 --- /dev/null +++ b/openswarm-runner/tests/test_renderer_process.py @@ -0,0 +1,99 @@ +"""The renderer half: a run that asked for a browser must never quietly proceed without one. + +The Electron boot itself needs a Linux container and a real display, so what is pinned here is +the contract around it: the deep link the window opens on, the bundle check, and the three ways +"no window" is allowed to end (loudly, every time). +""" + +import subprocess + +import pytest + +from runner.boot import renderer_process +from runner.boot.renderer_process import ( + CONTAINER_CHROMIUM_FLAGS, + RendererUnavailable, + SANDBOX_FLAGS, + await_registration, + dashboard_url, + serve_frontend, + start_electron, +) +from runner.run_spec import CLOUD_RUN_DASHBOARD_ID + + +@pytest.fixture +def dead_electron(): + """A real Popen that has already exited; await_registration is typechecked on Popen.""" + process = subprocess.Popen(["/bin/sh", "-c", "exit 9"]) + process.wait() + return process + + +@pytest.fixture +def live_electron(): + """A real Popen that stays up long enough for a poll loop to run against it.""" + process = subprocess.Popen(["/bin/sh", "-c", "sleep 30"]) + yield process + process.kill() + process.wait() + + +def test_the_window_opens_on_the_run_dashboard_not_the_picker() -> None: + # HashRouter, so the route has to be a fragment or the static server 404s on it. + assert dashboard_url(4173, CLOUD_RUN_DASHBOARD_ID) == ( + "http://127.0.0.1:4173/index.html#/dashboard/cloud-run" + ) + + +def test_a_missing_bundle_says_so_instead_of_serving_an_empty_dir(tmp_path) -> None: + with pytest.raises(RendererUnavailable, match="no frontend bundle"): + serve_frontend(str(tmp_path)) + + +def test_a_missing_electron_binary_fails_the_run_rather_than_the_workflow(tmp_path, monkeypatch) -> None: + monkeypatch.setenv("ELECTRON_BIN", str(tmp_path / "nope")) + with pytest.raises(RendererUnavailable, match="built without a renderer"): + start_electron(str(tmp_path), 8324, "http://127.0.0.1:4173/index.html") + + +def test_chromium_is_launched_unsandboxed_on_purpose_and_out_of_shared_memory() -> None: + # Dropping Chromium's own sandbox is a real tradeoff (the Firecracker VM is the wall that's + # left), so it lives in a named constant a reviewer trips over, not inside a launch string. + assert SANDBOX_FLAGS == ["--no-sandbox"] + assert "--disable-dev-shm-usage" in CONTAINER_CHROMIUM_FLAGS + + +def test_a_dead_electron_is_reported_as_dead_not_waited_out(monkeypatch, dead_electron) -> None: + monkeypatch.setattr(renderer_process, "RENDERER_TIMEOUT_SECONDS", 30.0) + with pytest.raises(RendererUnavailable, match="exited with code 9"): + await_registration("http://127.0.0.1:1", {}, dead_electron, deadline=1e9) + + +def test_a_window_that_never_registers_times_out_loudly(monkeypatch, live_electron) -> None: + monkeypatch.setattr(renderer_process, "RENDERER_TIMEOUT_SECONDS", 0.5) + with pytest.raises(RendererUnavailable, match="no renderer ever registered"): + await_registration("http://127.0.0.1:1", {}, live_electron, deadline=1e9) + + +def test_registration_is_believed_only_when_the_backend_says_a_socket_is_attached(monkeypatch, live_electron) -> None: + replies = [{"attached": False, "ever_attached": False, "connections": 0}, + {"attached": True, "ever_attached": True, "connections": 1}] + + class p_Response: + def json(self): + return replies.pop(0) + + class p_Client: + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def get(self, url, headers=None): + return p_Response() + + monkeypatch.setattr(renderer_process.httpx, "Client", lambda **kw: p_Client()) + await_registration("http://127.0.0.1:1", {}, live_electron, deadline=1e9) + assert replies == [] diff --git a/openswarm-runner/tests/test_router_credentials.py b/openswarm-runner/tests/test_router_credentials.py new file mode 100644 index 00000000..2dcfb232 --- /dev/null +++ b/openswarm-runner/tests/test_router_credentials.py @@ -0,0 +1,111 @@ +"""The runner must be structurally unable to rotate a user's OAuth grant. + +Every test here exists to make one class of bug unwritable: a refresh token reaching +9Router's db.json. Delete either wall in runner/seed/router_credentials.py and these go red. +""" + +import json +import os +import stat +from datetime import datetime, timedelta, timezone + +import pytest + +from runner.run_spec import InvalidRunSpec, ProviderCredential, RunSpec, load_run_spec +from runner.seed.router_credentials import ( + RefreshTokenLeak, + assert_no_refresh_token, + router_db_payload, + write_router_db, +) + +NOW = datetime(2026, 7, 31, 12, 0, 0, tzinfo=timezone.utc) +ACCESS_TOKEN = "at-test-value-not-a-real-token" +REFRESH_TOKEN = "rt-test-value-not-a-real-token" + + +def spec_json(credential: dict) -> str: + return json.dumps({ + "run_id": "run-1", + "workflow": {"id": "wf-1", "title": "Test", "steps": [{"text": "say hi"}]}, + "credentials": [credential], + }) + + +def oauth_credential() -> ProviderCredential: + return ProviderCredential( + provider="claude", + auth_type="oauth", + access_token=ACCESS_TOKEN, + expires_at=NOW + timedelta(hours=8), + ) + + +@pytest.mark.parametrize("key", ["refreshToken", "refresh_token", "Refresh-Token", "oauthRefreshToken"]) +def test_a_spec_carrying_a_refresh_token_never_parses(key: str, tmp_path, monkeypatch) -> None: + payload = { + "provider": "claude", + "auth_type": "oauth", + "access_token": ACCESS_TOKEN, + "expires_at": (NOW + timedelta(hours=8)).isoformat(), + key: REFRESH_TOKEN, + } + monkeypatch.setenv("OPENSWARM_RUN_SPEC", spec_json(payload)) + with pytest.raises(InvalidRunSpec) as caught: + load_run_spec() + assert key in str(caught.value) + assert not list(tmp_path.iterdir()), "a rejected spec must not leave anything on disk" + + +@pytest.mark.parametrize("key", ["refreshToken", "refresh_token", "Refresh-Token", "oauthRefreshToken"]) +def test_the_writer_guard_rejects_a_poisoned_payload(key: str) -> None: + payload = {"providerConnections": [{"provider": "claude", key: REFRESH_TOKEN}]} + with pytest.raises(RefreshTokenLeak): + assert_no_refresh_token(payload) + + +def test_the_guard_passes_a_clean_payload() -> None: + assert_no_refresh_token(router_db_payload([oauth_credential()], NOW)) + + +def test_written_db_carries_the_access_token_and_no_refresh_token(tmp_path) -> None: + path = write_router_db(str(tmp_path / "9router"), [oauth_credential()], NOW) + raw = open(path, "r", encoding="utf-8").read() + + # Without this the "no refresh token" assertion below would also pass on an empty file. + assert ACCESS_TOKEN in raw + connection = json.loads(raw)["providerConnections"][0] + assert connection["provider"] == "claude" + assert connection["isActive"] is True + assert connection["expiresAt"] == "2026-07-31T20:00:00.000Z" + + assert "refresh" not in raw.lower() + assert not any("refresh" in key.lower() for key in connection) + + +def test_the_db_and_its_directory_are_owner_only(tmp_path) -> None: + path = write_router_db(str(tmp_path / "9router"), [oauth_credential()], NOW) + assert stat.S_IMODE(os.stat(path).st_mode) == 0o600 + assert stat.S_IMODE(os.stat(os.path.dirname(path)).st_mode) == 0o700 + + +def test_an_api_key_credential_never_reaches_the_router_db(tmp_path) -> None: + credential = ProviderCredential(provider="anthropic", auth_type="api_key", api_key="sk-test-not-real") + path = write_router_db(str(tmp_path / "9router"), [credential], NOW) + assert json.loads(open(path, encoding="utf-8").read())["providerConnections"] == [] + + +def test_an_oauth_credential_without_an_access_token_is_rejected() -> None: + with pytest.raises(ValueError, match="no access_token"): + ProviderCredential(provider="claude", auth_type="oauth", expires_at=NOW) + + +def test_an_expired_access_token_is_fatal_not_refreshable() -> None: + spec = RunSpec.model_validate_json(spec_json({ + "provider": "claude", + "auth_type": "oauth", + "access_token": ACCESS_TOKEN, + "expires_at": (NOW + timedelta(seconds=30)).isoformat(), + })) + assert [credential.provider for credential in spec.expired_credentials(NOW)] == ["claude"] + assert spec.expired_credentials(NOW - timedelta(hours=1)) == [] diff --git a/openswarm-runner/tests/test_run_spec.py b/openswarm-runner/tests/test_run_spec.py new file mode 100644 index 00000000..1733f5a5 --- /dev/null +++ b/openswarm-runner/tests/test_run_spec.py @@ -0,0 +1,94 @@ +"""The run spec is the only thing the control plane can say to this container.""" + +import json +import os +import stat + +import pytest + +from runner.run_spec import InvalidRunSpec, RunSpec, load_run_spec +from runner.seed.data_root import seed_data_root, settings_for_run + +VALID_CREDENTIAL = {"provider": "anthropic", "auth_type": "api_key", "api_key": "sk-test-not-real"} + + +def spec_body(**overrides) -> dict: + body = { + "run_id": "run-1", + "workflow": { + "id": "wf-1", + "title": "Daily digest", + "model": "opus-5", + "steps": [{"text": "summarize the inbox"}], + "schedule": {"enabled": True, "repeat_unit": "day", "hour": 9}, + }, + "credentials": [VALID_CREDENTIAL], + } + body.update(overrides) + return body + + +def test_a_missing_spec_names_both_env_vars(monkeypatch) -> None: + monkeypatch.delenv("OPENSWARM_RUN_SPEC", raising=False) + monkeypatch.delenv("OPENSWARM_RUN_SPEC_FILE", raising=False) + with pytest.raises(InvalidRunSpec, match="OPENSWARM_RUN_SPEC_FILE"): + load_run_spec() + + +def test_a_spec_file_is_accepted(tmp_path, monkeypatch) -> None: + path = tmp_path / "spec.json" + path.write_text(json.dumps(spec_body()), encoding="utf-8") + monkeypatch.delenv("OPENSWARM_RUN_SPEC", raising=False) + monkeypatch.setenv("OPENSWARM_RUN_SPEC_FILE", str(path)) + assert load_run_spec().workflow.title == "Daily digest" + + +def test_unknown_top_level_fields_are_rejected(monkeypatch) -> None: + monkeypatch.setenv("OPENSWARM_RUN_SPEC", json.dumps(spec_body(surprise="hello"))) + with pytest.raises(InvalidRunSpec, match="surprise"): + load_run_spec() + + +def test_a_run_needs_at_least_one_credential(monkeypatch) -> None: + monkeypatch.setenv("OPENSWARM_RUN_SPEC", json.dumps(spec_body(credentials=[]))) + with pytest.raises(InvalidRunSpec): + load_run_spec() + + +def test_the_container_never_inherits_the_schedule() -> None: + spec = RunSpec.model_validate(spec_body()) + assert spec.workflow.schedule.enabled is True + assert spec.workflow_for_disk().schedule.enabled is False + + +def test_seeding_writes_the_workflow_and_owner_only_settings(tmp_path) -> None: + spec = RunSpec.model_validate(spec_body()) + workspace = str(tmp_path / "workspace") + seed_data_root(str(tmp_path / "data"), workspace, spec) + + workflow_path = tmp_path / "data" / "workflows" / "wf-1.json" + settings_path = tmp_path / "data" / "settings" / "settings.json" + assert json.loads(workflow_path.read_text())["schedule"]["enabled"] is False + assert stat.S_IMODE(os.stat(settings_path).st_mode) == 0o600 + + settings = json.loads(settings_path.read_text()) + assert settings["anthropic_api_key"] == "sk-test-not-real" + assert settings["default_model"] == "opus-5" + assert settings["analytics_opt_in"] is False + # The agent's folder is the deliverable folder, and it exists before the backend boots. + assert settings["default_folder"] == workspace + assert os.path.isdir(workspace) + + +def test_the_agent_is_told_its_files_only_survive_from_the_workspace(tmp_path) -> None: + spec = RunSpec.model_validate(spec_body()) + prompt = settings_for_run(spec, str(tmp_path)).default_system_prompt or "" + assert "delivered back to the user" in prompt + + +def test_an_unmappable_api_key_provider_fails_loudly(tmp_path) -> None: + spec = RunSpec.model_validate(spec_body( + credentials=[{"provider": "wat", "auth_type": "api_key", "api_key": "x"}] + )) + with pytest.raises(ValueError, match="no settings field"): + settings_for_run(spec, str(tmp_path)) diff --git a/openswarm-runner/tests/test_seed_skills.py b/openswarm-runner/tests/test_seed_skills.py new file mode 100644 index 00000000..a046ca34 --- /dev/null +++ b/openswarm-runner/tests/test_seed_skills.py @@ -0,0 +1,83 @@ +"""The user's know-how travelling up, and their app credentials not travelling at all.""" + +import os + +import pytest +from pydantic import ValidationError + +from runner.run_spec import McpServerNote, RunSpec, SkillPayload +from runner.seed.data_root import settings_for_run, unavailable_apps_note +from runner.seed.skills import skills_dir, write_skills + +COFFEE = { + "id": "coffee-ratio", + "files": [ + {"path": "SKILL.md", "text": "---\nname: coffee-ratio\ndescription: house ratio\n---\n\n1:16.5\n"}, + {"path": "scripts/brew.py", "text": "print('brew')\n"}, + ], +} + + +def p_spec(**overrides) -> RunSpec: + body = { + "run_id": "cr-1", + "workflow": {"id": "wf-1", "title": "Test", "steps": [{"id": "s1", "text": "go"}]}, + "credentials": [{"provider": "anthropic", "auth_type": "api_key", "api_key": "sk-test"}], + } + body.update(overrides) + return RunSpec.model_validate(body) + + +def test_a_skill_folder_lands_where_the_backend_looks_for_it(tmp_path) -> None: + written = write_skills(str(tmp_path), [SkillPayload.model_validate(COFFEE)]) + + assert written == 1 + root = skills_dir(str(tmp_path)) + assert root.endswith(os.path.join(".claude", "skills")) + with open(os.path.join(root, "coffee-ratio", "SKILL.md"), encoding="utf-8") as handle: + assert "house ratio" in handle.read() + assert os.path.isfile(os.path.join(root, "coffee-ratio", "scripts", "brew.py")) + + +def test_no_skills_is_an_empty_directory_not_a_failure(tmp_path) -> None: + assert write_skills(str(tmp_path), []) == 0 + assert os.path.isdir(skills_dir(str(tmp_path))) + + +def test_a_skill_with_no_skill_md_is_refused_before_a_machine_boots() -> None: + with pytest.raises(ValidationError, match="no SKILL.md"): + SkillPayload.model_validate({"id": "x", "files": [{"path": "README.md", "text": "hi"}]}) + + +@pytest.mark.parametrize("bad", ["../escape.md", "/etc/passwd", "a/../../b.md", "a\\b.md", ""]) +def test_a_skill_path_that_could_climb_out_is_refused(bad: str) -> None: + with pytest.raises(ValidationError): + SkillPayload.model_validate({"id": "x", "files": [{"path": bad, "text": "hi"}]}) + + +@pytest.mark.parametrize("bad", ["../evil", "a/b", ".hidden", "has space", ""]) +def test_a_skill_id_that_is_not_a_plain_folder_name_is_refused(bad: str) -> None: + with pytest.raises(ValidationError): + SkillPayload.model_validate({"id": bad, "files": [{"path": "SKILL.md", "text": "hi"}]}) + + +def test_the_spec_cannot_carry_an_mcp_secret_at_all() -> None: + # extra="forbid" is the wall: there is no field for a token, so a payload with one dies here + # rather than landing in a container that runs the user's own prose with Bash. + with pytest.raises(ValidationError): + McpServerNote.model_validate({"name": "Notion", "access_token": "secret_abc"}) + with pytest.raises(ValidationError): + McpServerNote.model_validate({"name": "Slack", "env": {"SLACK_MCP_XOXC_TOKEN": "xoxc-1"}}) + + +def test_unreachable_apps_are_named_in_the_prompt_so_silence_is_not_mistaken_for_absence(tmp_path) -> None: + spec = p_spec(unavailable_mcp_servers=[{"name": "Notion"}, {"name": "Google Workspace"}]) + note = unavailable_apps_note(spec) + + assert "Notion" in note and "Google Workspace" in note + assert "cannot be done from a cloud run" in note + assert note in (settings_for_run(spec, str(tmp_path)).default_system_prompt or "") + + +def test_with_no_connected_apps_nothing_is_added_to_the_prompt(tmp_path) -> None: + assert unavailable_apps_note(p_spec()) == "" diff --git a/openswarm-runner/tests/test_workflow_run.py b/openswarm-runner/tests/test_workflow_run.py new file mode 100644 index 00000000..ce1f00eb --- /dev/null +++ b/openswarm-runner/tests/test_workflow_run.py @@ -0,0 +1,123 @@ +"""Reading a finished session correctly, including the failures the backend calls success.""" + +import subprocess +import sys +import time +from typing import Any, Dict, Iterator + +import httpx +import pytest + +from runner import workflow_run +from runner.boot.backend_process import BackendProcess +from runner.workflow_run import ( + WorkflowRunFailed, + execute_workflow, + final_answer, + render_transcript, + system_notices, + trigger_run, +) + +# Shape taken verbatim from a real container run whose provider token was rejected. +REJECTED_TOKEN_SESSION = [ + {"role": "user", "content": "Reply with exactly the word PONG and nothing else."}, + {"role": "system", "content": "Provider authentication expired. Open Settings, Models and reconnect, then send your message again."}, +] + +ANSWERED_SESSION = [ + {"role": "user", "content": "ping"}, + {"role": "assistant", "content": [{"type": "tool_use", "name": "Bash", "input": {"command": "echo hi"}}]}, + {"role": "assistant", "content": [{"type": "text", "text": "PONG"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "draft"}], "hidden": True}, +] + + +def test_a_rejected_credential_surfaces_as_a_system_notice() -> None: + assert system_notices(REJECTED_TOKEN_SESSION) == [REJECTED_TOKEN_SESSION[1]["content"]] + assert final_answer(REJECTED_TOKEN_SESSION) == "" + + +def test_a_healthy_run_raises_no_notices() -> None: + assert system_notices(ANSWERED_SESSION) == [] + + +def test_the_answer_is_the_last_visible_assistant_text() -> None: + assert final_answer(ANSWERED_SESSION) == "PONG" + + +def test_the_transcript_keeps_tool_calls_and_drops_hidden_turns() -> None: + transcript = render_transcript(ANSWERED_SESSION) + assert "[tool Bash]" in transcript + assert "PONG" in transcript + assert "draft" not in transcript + + +# A backend busy enough to miss a reply is the normal case, not a broken one: a single MCP +# registry call has been measured holding its event loop past a minute. These pin that a late +# reply never costs the user the run. + +RUN_ROW = {"id": "run_1", "status": "success", "session_id": "sess_1", "cost_usd": 0.0} + + +@pytest.fixture +def backend() -> Iterator[BackendProcess]: + """A real BackendProcess around a process that just sits there, so is_alive() is honest.""" + process = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(60)"]) + try: + yield BackendProcess(process=process, base_url="http://backend.test", token="t") + finally: + process.kill() + process.wait(timeout=10) + + +def p_client(stalls: int) -> httpx.Client: + """A client whose first `stalls` requests time out, exactly like a starved event loop.""" + state = {"left": stalls} + + def handle(request: httpx.Request) -> httpx.Response: + if state["left"] > 0: + state["left"] -= 1 + raise httpx.ReadTimeout("timed out", request=request) + if request.url.path.endswith("/run"): + return httpx.Response(200, json={"run_id": "run_1", "status": "running"}) + if request.url.path.endswith("/runs"): + return httpx.Response(200, json={"runs": [RUN_ROW]}) + return httpx.Response(404, json={}) + + return httpx.Client(transport=httpx.MockTransport(handle), timeout=1.0) + + +def test_a_stalled_trigger_adopts_the_run_it_already_started(backend: BackendProcess) -> None: + # One unanswered POST, then the run it started is visible. Posting again would either run the + # workflow twice or come back "Previous run still active". + assert trigger_run(p_client(stalls=1), backend, "wf_1", time.monotonic() + 5.0) == "run_1" + + +def test_a_dead_backend_fails_the_trigger_instead_of_waiting(backend: BackendProcess) -> None: + backend.process.kill() + backend.process.wait(timeout=10) + with pytest.raises(WorkflowRunFailed, match="backend died"): + trigger_run(p_client(stalls=99), backend, "wf_1", time.monotonic() + 5.0) + + +def test_a_stalled_poll_does_not_throw_away_a_run_that_finishes( + backend: BackendProcess, monkeypatch: pytest.MonkeyPatch +) -> None: + polls = {"left": 2} + + def p_find(*_args: Any, **_kwargs: Any) -> Dict[str, Any]: + if polls["left"] > 0: + polls["left"] -= 1 + raise httpx.ReadTimeout("timed out") + return RUN_ROW + + monkeypatch.setattr(workflow_run, "trigger_run", lambda *_a, **_k: "run_1") + monkeypatch.setattr(workflow_run, "p_find_run", p_find) + monkeypatch.setattr(workflow_run, "p_collect_session", lambda *_a, **_k: ANSWERED_SESSION) + monkeypatch.setattr(workflow_run, "POLL_INTERVAL_SECONDS", 0.01) + + outcome = execute_workflow(backend, "wf_1", time.monotonic() + 10.0) + assert outcome.status == "success" + assert outcome.answer == "PONG" + assert polls["left"] == 0 diff --git a/scripts/ci/verify-all.js b/scripts/ci/verify-all.js index 20a6ef6b..28504e12 100644 --- a/scripts/ci/verify-all.js +++ b/scripts/ci/verify-all.js @@ -36,6 +36,7 @@ function main() { ['deps fully pinned (reproducible backend builds)', 'verify-deps-pinned.js', []], ['no build-host paths leaked into the artifact', 'verify-host-leakage.js', appArg], ['locale paks shipped (empty --lang -> Blink null-deref crash)', 'verify-locale-paks.js', appArg], + ['native modules match the target arch (no x86_64 .node in an arm64 app)', 'verify-native-arch.js', appArg], ['9router deps shipped (else subscription service hangs)', 'verify-router-deps.js', appArg], ['bundled python runs (--version + import smoke)', 'verify-python-health.js', appArg], ['MCP bundles answer initialize over stdio', 'verify-mcp-bundles.js', appArg], diff --git a/scripts/ci/verify-native-arch.js b/scripts/ci/verify-native-arch.js new file mode 100644 index 00000000..2828c5dd --- /dev/null +++ b/scripts/ci/verify-native-arch.js @@ -0,0 +1,82 @@ +#!/usr/bin/env node +// Guards against shipping a native module built for the WRONG architecture. +// prebuildify packages (uiohook-napi, the dictation hotkey tap) ship one .node per platform+arch +// they support: seven dirs, of which exactly one is ours. Six are dead weight, and on macOS an +// x86_64 Mach-O inside an arm64 bundle is what makes the OS raise its Intel-deprecation dialog at +// the user. electron/build/after-pack.js prunes them; this asserts the prune actually happened, +// because an afterPack hook that silently no-ops (layout changed, module moved into the asar) would +// otherwise ship the same bundle it always did with nobody the wiser. + +'use strict'; +const fs = require('fs'); +const path = require('path'); +const { execFileSync } = require('child_process'); +const h = require('./lib/app-harness'); + +function parseArgs(argv) { + const out = { app: null }; + for (let i = 0; i < argv.length; i++) if (argv[i] === '--app') out.app = argv[++i]; + return out; +} + +function findNodeFiles(root) { + const found = []; + (function walk(dir, depth) { + if (depth > 14) return; + let ents = []; + try { ents = fs.readdirSync(dir, { withFileTypes: true }); } catch { return; } + for (const e of ents) { + const full = path.join(dir, e.name); + if (e.isDirectory()) walk(full, depth + 1); + else if (e.isFile() && e.name.endsWith('.node')) found.push(full); + } + })(root, 0); + return found; +} + +// `file` names every slice in a Mach-O, so a fat binary reports both and a wrong-arch one reports +// only the wrong slice. On Windows/Linux there is no equivalent worth the dependency, so the gate +// there just asserts no foreign prebuild DIRS survived. +function machoArches(file) { + try { + return execFileSync('file', [file], { encoding: 'utf8' }).trim(); + } catch (err) { + return `file failed: ${err && err.message}`; + } +} + +function main() { + const args = parseArgs(process.argv.slice(2)); + const exe = h.packagedAppPath(args.app); + const root = process.platform === 'darwin' ? exe.slice(0, exe.indexOf('.app') + 4) : path.dirname(exe); + const want = process.arch === 'arm64' ? 'arm64' : 'x86_64'; + + const nodes = findNodeFiles(root); + process.stdout.write(` ${nodes.length} .node file(s) under ${path.basename(root)}\n`); + + const offenders = []; + for (const n of nodes) { + // A prebuilds dir for another OS is wrong no matter what `file` says about it. + const tuple = path.basename(path.dirname(n)); + const platformOk = !tuple.includes('-') || tuple.startsWith(process.platform); + const desc = process.platform === 'darwin' ? machoArches(n) : ''; + const archOk = process.platform !== 'darwin' || desc.includes(want); + if (!platformOk || !archOk) offenders.push(`${path.relative(root, n)} [${tuple}] ${desc.split(':').slice(1).join(':').trim()}`); + } + + if (!offenders.length) { + process.stdout.write(`PASS every bundled .node targets ${process.platform}/${want}\n`); + process.exit(0); + } + process.stderr.write( + `FAIL packaged build ships ${offenders.length} native module(s) for the WRONG target:\n` + + offenders.map((o) => ` ${o}\n`).join('') + + ` An x86_64 .node inside an arm64 bundle makes macOS show its Intel-deprecation\n` + + ` dialog, and every foreign prebuild is dead weight the user downloads.\n` + + ` Fix: electron/build/after-pack.js pruneForeignPrebuilds() should have deleted these.\n` + + ` If they now live inside app.asar rather than app.asar.unpacked, the prune needs to\n` + + ` run before the asar is built (beforePack) instead.\n`); + process.exit(1); +} + +main();