diff --git a/backend/apps/agents/agent_manager.py b/backend/apps/agents/agent_manager.py index cf3c3541..9d6a9555 100644 --- a/backend/apps/agents/agent_manager.py +++ b/backend/apps/agents/agent_manager.py @@ -54,6 +54,33 @@ def _safe_resp_text(resp) -> str: return "" +def _apply_context_window(session, settings=None) -> None: + """Set session.context_window from the registry for its (provider, model). + + Called at every AgentSession creation, restore, and model-switch site so + the soft-cap trim, auto-compaction, and the UI percent meter line up + with the model's real cap (Opus/Sonnet 1M, Haiku 200k, custom values + declared per-provider). Silent fallback to the existing value keeps a + bad lookup from ever breaking a session. + """ + try: + from backend.apps.agents.providers.registry import get_context_window + if settings is None: + try: + settings = load_settings() + except Exception: + settings = None + cw = get_context_window( + getattr(session, "provider", "") or "", + getattr(session, "model", "") or "", + settings, + ) + if isinstance(cw, int) and cw > 0: + session.context_window = cw + except Exception: + logger.debug("context_window lookup failed; keeping existing value", exc_info=True) + + def _save_session(session_id: str, doc_data: dict): os.makedirs(SESSIONS_DIR, exist_ok=True) with open(os.path.join(SESSIONS_DIR, f"{session_id}.json"), "w") as f: @@ -788,6 +815,7 @@ class AgentManager: dashboard_id=config.dashboard_id, thinking_level=getattr(global_settings, "default_thinking_level", "auto"), ) + _apply_context_window(session, global_settings) self.sessions[session_id] = session from backend.apps.service.service import APP_VERSION @@ -800,35 +828,6 @@ class AgentManager: return session - def _resolve_context_paths(self, context_paths: list | None) -> str: - """Read file contents / directory trees for attached context paths.""" - if not context_paths: - return "" - sections = [] - for cp in context_paths: - path = cp.get("path", "") - cp_type = cp.get("type", "file") - if not path or not os.path.exists(path): - sections.append(f"[Context: {path} — not found]") - continue - if cp_type == "file" and os.path.isfile(path): - try: - with open(path, "r", errors="replace") as f: - content = f.read(512_000) # ~500KB cap per file - sections.append( - f"\n{content}\n" - ) - except Exception as e: - sections.append(f"[Context: {path} — error reading: {e}]") - elif cp_type == "directory" and os.path.isdir(path): - tree_lines = self._build_dir_tree(path, max_depth=4) - sections.append( - f"\n{chr(10).join(tree_lines)}\n" - ) - else: - sections.append(f"[Context: {path} — type mismatch]") - return "\n\n".join(sections) - def _build_dir_tree(self, root: str, max_depth: int = 4, prefix: str = "") -> list[str]: """Build a recursive directory tree listing.""" lines = [] @@ -1098,19 +1097,41 @@ class AgentManager: ) return replacement, blob_path - def _build_prompt_content(self, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None): - """Build message content with optional image blocks, context, and forced tools for the Claude API.""" - context_text = self._resolve_context_paths(context_paths) + def _build_prompt_content(self, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None, api_type: str = "anthropic", model: str = ""): + """Build message content for the Anthropic SDK's prompt stream. + + Routes attachments per provider: + - Anthropic: native `image` + `document` blocks for the active + Claude model. Text files inline as . Binary that + isn't PDF/image gets a refusal placeholder. + - Gemini (api=gemini): we still talk to the SDK with Anthropic + content-block shapes; the 9router translation layer (cc/gc/gpt + lanes) converts to the provider's native shape. For Gemini's + native multimodal we emit image/document blocks the same way + and rely on 9router to rewrite to inline_data. Over 20MB + payloads get refused at this layer (Gemini's inline cap). + - OpenAI / Codex: image blocks pass through (image_url at the + wire); PDFs handled as documents on multimodal models; non- + multimodal models refuse. + - OpenRouter, custom OpenAI-compatible: text fallback for + anything binary, since native shape varies wildly. Caller can + opt-in to the OR file-parser via a separate plugins config. + """ + context_text, native_blocks, refusals = self._resolve_attachments( + context_paths, api_type=api_type, model=model, + ) forced_tools_text = self._resolve_forced_tools(forced_tools) skills_text = self._resolve_attached_skills(attached_skills) - parts = [p for p in (forced_tools_text, context_text, skills_text, prompt) if p] + refusal_text = "\n\n".join(refusals) + parts = [p for p in (forced_tools_text, context_text, refusal_text, skills_text, prompt) if p] full_prompt = "\n\n".join(parts) - if not images: + has_native = bool(native_blocks) + if not images and not has_native: return full_prompt - content = [{"type": "text", "text": full_prompt}] - for img in images: + content: list[dict] = [{"type": "text", "text": full_prompt}] + for img in (images or []): content.append({ "type": "image", "source": { @@ -1119,15 +1140,236 @@ class AgentManager: "data": img["data"], }, }) + content.extend(native_blocks) return content + def _resolve_attachments(self, context_paths: list | None, api_type: str, model: str) -> tuple[str, list[dict], list[str]]: + """Split context_paths into: + - inline text (returned as the existing block string) + - native content blocks for this provider (PDFs/images) + - refusal strings that get appended to the prompt as plain text + + Reuses the upload-time sniff (PDF magic / null-byte heuristic) so + a renamed `.pdf` actually classifies right, and a `.txt` with + binary garbage doesn't sneak through as text. + + Two layers of size guard: + 1) Per-file inline cap based on provider's raw size limit. + 2) Total base64-expanded size cap across all native attachments, + because providers cap the WHOLE request body (Anthropic 32MB, + Gemini 20MB, OpenAI 50MB). 4 medium PDFs that pass the + per-file check can still collectively blow the request cap. + The last document block gets cache_control:ephemeral so a follow-up + turn on the same PDF reuses the cache prefix (Anthropic only). + """ + if not context_paths: + return "", [], [] + from backend.apps.settings.settings import _sniff_file_kind + import base64 as _b64 + sections: list[str] = [] + native: list[dict] = [] + refusals: list[str] = [] + + # The Claude Agent SDK speaks only Anthropic content-block shape. + # 9router 0.3.60 translates `image` blocks to the per-provider + # native shape; we trust that (the existing `images` param has + # shipped on every provider since v1.0.29). + # `document` (PDF) blocks: native on Anthropic upstream. For + # Gemini, anthropic-proxy rewrites document→image (keeping + # media_type=application/pdf), and Gemini's inline_data accepts + # that mime type natively. For OpenRouter, anthropic-proxy + # detects document blocks + injects the file-parser plugin. For + # OpenAI we refuse PDFs (no 9router translator path for the + # type:file shape, and Codex OAuth can't hit /v1/files anyway). + api = (api_type or "anthropic").lower() + supports_image = api in ("anthropic", "gemini", "openai", "openrouter", "gemini-cli") + # PDFs flow per provider: + # - Anthropic: native document blocks pass through cleanly. + # - Gemini: anthropic_proxy rewrites document → image_url with + # data:application/pdf base64; 9router translates to Gemini + # inlineData natively. VERIFIED empirically May 2026 (47K + # prompt tokens, real PaLM content summarized). + # - OpenRouter: file-parser plugin injected in anthropic-proxy. + # - OpenAI: REFUSED. OpenAI's image_url only accepts image/* + # mime; the type:file shape gets stringified by 9router. + # Workaround: route through `openrouter/openai/gpt-5` which + # uses OR's file-parser plugin. + # - Codex (cx/): models don't support PDFs. + supports_pdf = api in ("anthropic", "gemini", "gemini-cli", "openrouter") + + # Per-file inline caps (raw bytes, before base64). Going over + # means the request would 4xx, blow our 64MB SDK buffer, or + # exceed the API's per-request cap on its own. + if api == "anthropic": + per_file_cap = 24 * 1024 * 1024 + total_request_cap = 28 * 1024 * 1024 # under Anthropic's 32MB + elif api == "gemini": + per_file_cap = 14 * 1024 * 1024 + total_request_cap = 15 * 1024 * 1024 # under Gemini's 20MB + elif api == "openai": + per_file_cap = 24 * 1024 * 1024 + total_request_cap = 45 * 1024 * 1024 # under OpenAI's 50MB + elif api == "openrouter": + per_file_cap = 24 * 1024 * 1024 + total_request_cap = 45 * 1024 * 1024 + else: + per_file_cap = 0 + total_request_cap = 0 + + # Running total of base64-expanded bytes already committed to the + # request. Anything that would push us over total_request_cap gets + # refused with concrete recovery actions. + b64_total = 0 + + for cp in context_paths: + path = cp.get("path", "") or "" + cp_type = cp.get("type", "file") + if not path or not os.path.exists(path): + sections.append(f"[Context: {path}, not found]") + continue + if cp_type == "directory" and os.path.isdir(path): + tree_lines = self._build_dir_tree(path, max_depth=4) + sections.append( + f"\n{chr(10).join(tree_lines)}\n" + ) + continue + if cp_type != "file" or not os.path.isfile(path): + sections.append(f"[Context: {path}, type mismatch]") + continue + try: + size = os.path.getsize(path) + with open(path, "rb") as fh: + head = fh.read(4096) + kind, media_type = _sniff_file_kind(head, os.path.basename(path)) + + if kind == "text": + with open(path, "r", errors="replace") as f: + content = f.read(512_000) + sections.append( + f"\n{content}\n" + ) + continue + + # base64 expands ~4/3, ceil to be conservative. + b64_size = ((size + 2) // 3) * 4 + + if kind == "pdf": + if not supports_pdf: + if api == "openai": + # Falls here only for Codex variants (gpt-5.3-codex etc.), + # which don't accept PDFs even though their family does. + refusals.append( + f"[Attached PDF {os.path.basename(path)} ({size // 1024} KB) cannot be read on Codex models. " + "Switch to a non-Codex GPT-5 (e.g. gpt-5.5), Claude, Gemini 3.x, or " + "any model via OpenRouter to read PDFs natively.]" + ) + else: + refusals.append( + f"[Attached PDF {os.path.basename(path)} ({size // 1024} KB) cannot be read on this provider. " + "Switch to a Claude model (Sonnet 4.6, Opus 4.7, Haiku 4.5), Gemini 3.x, GPT-5 (non-Codex), " + "or any model through OpenRouter to read PDFs natively.]" + ) + continue + if size > per_file_cap: + refusals.append( + f"[Attached PDF {os.path.basename(path)} ({size // (1024*1024)} MB) exceeds the per-file cap " + f"of {per_file_cap // (1024*1024)} MB on this provider. Split the PDF or send a smaller excerpt.]" + ) + continue + if b64_total + b64_size > total_request_cap: + room_mb = max(0, total_request_cap - b64_total) // (1024 * 1024) + refusals.append( + f"[Attached PDF {os.path.basename(path)} would push the request over " + f"{total_request_cap // (1024*1024)} MB encoded (provider cap). " + f"Only ~{room_mb} MB of room left this turn. Detach a file, or send PDFs in separate turns.]" + ) + continue + with open(path, "rb") as fh: + data_b64 = _b64.b64encode(fh.read()).decode("ascii") + block = { + "type": "document", + "source": { + "type": "base64", + "media_type": "application/pdf", + "data": data_b64, + }, + } + native.append(block) + b64_total += b64_size + continue + + if kind == "image": + if not supports_image: + refusals.append( + f"[Attached image {os.path.basename(path)} cannot be displayed to this model. " + "Switch to a vision-capable model (Claude, GPT-4o/5, Gemini).]" + ) + continue + if size > per_file_cap: + refusals.append( + f"[Attached image {os.path.basename(path)} ({size // (1024*1024)} MB) exceeds per-file cap.]" + ) + continue + if b64_total + b64_size > total_request_cap: + room_mb = max(0, total_request_cap - b64_total) // (1024 * 1024) + refusals.append( + f"[Attached image {os.path.basename(path)} would push the request over " + f"{total_request_cap // (1024*1024)} MB encoded. ~{room_mb} MB of room left.]" + ) + continue + with open(path, "rb") as fh: + data_b64 = _b64.b64encode(fh.read()).decode("ascii") + native.append({ + "type": "image", + "source": { + "type": "base64", + "media_type": media_type or "image/png", + "data": data_b64, + }, + }) + b64_total += b64_size + continue + + # binary, other + refusals.append( + f"[Attached binary file {os.path.basename(path)} not inlined. Convert to text first.]" + ) + except Exception as e: + sections.append(f"[Context: {path}, error reading: {e}]") + + # Anthropic prompt caching: tag the last document block as ephemeral + # so a follow-up turn referencing the same PDF stays cache-warm. + # Per Anthropic docs, only the trailing cache_control marker matters + # for cache prefix scope; earlier markers are ignored. + if api == "anthropic" and native: + for blk in reversed(native): + if blk.get("type") == "document": + blk["cache_control"] = {"type": "ephemeral"} + break + + context_text = "\n\n".join(sections) + return context_text, native, refusals + + # Legacy entry point retained for any external caller; routes to the + # new attachment resolver with anthropic-default routing (no native + # blocks emitted, so behavior is the safe text-only old path). + def _resolve_context_paths(self, context_paths: list | None) -> str: + text, _native, refusals = self._resolve_attachments(context_paths, api_type="anthropic", model="") + refusal_text = "\n\n".join(refusals) + return "\n\n".join(p for p in (text, refusal_text) if p) + async def _run_agent_loop(self, session_id: str, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None, fork_session: bool = False, selected_browser_ids: list[str] | None = None): """Run the Claude Agent SDK query loop for a session.""" session = self.sessions.get(session_id) if not session: return - prompt_content = self._build_prompt_content(prompt, images, context_paths, forced_tools, attached_skills) + from backend.apps.agents.providers.registry import get_api_type as _get_api_type + _api = _get_api_type(session.model) + prompt_content = self._build_prompt_content( + prompt, images, context_paths, forced_tools, attached_skills, + api_type=_api, model=session.model, + ) try: from claude_agent_sdk import ( @@ -1715,6 +1957,7 @@ class AgentManager: # than relying on the field's default_factory. active_mcps=[], ) + _apply_context_window(sub_session) self.sessions[sub_session_id] = sub_session await ws_manager.broadcast_global("agent:status", { "session_id": sub_session_id, @@ -1814,10 +2057,13 @@ class AgentManager: # Per-turn estimate of framework overhead (subtracted from displayed # input). Conservative on purpose so honest over-shows beat lies. - # 16K Claude Code preset, 12K base+deferred tools, 600/MCP, char/4 prompt. + # 16K Claude Code preset, 12K base+deferred tools, ~3K/MCP (real + # MCP tool definitions range 1-10K depending on server; 3K is a + # rough median that keeps the meter honest without over-trimming), + # char/4 of composed prompt. _PRESET_OVERHEAD = 16_000 _TOOL_DEFS_OVERHEAD = 12_000 - _PER_MCP_OVERHEAD = 600 + _PER_MCP_OVERHEAD = 3_000 _composed_tokens = len(composed_prompt or "") // 4 _mcp_tokens = len(session.active_mcps) * _PER_MCP_OVERHEAD session.framework_overhead_tokens = ( @@ -2149,7 +2395,13 @@ class AgentManager: options_kwargs = { "model": resolved_model, - "max_buffer_size": 5 * 1024 * 1024, + # 64 MB ceiling on the SDK <-> CLI JSON-RPC channel. The + # default 5 MB blocked any base64'd PDF over ~3.5 MB; we + # now route PDFs/images as native content blocks, which + # base64-expand by ~33%. 64 MB clears the largest single + # Anthropic PDF (32 MB raw) with headroom for prompt + + # tool results sharing the same frame. + "max_buffer_size": 64 * 1024 * 1024, "permission_mode": "default", "can_use_tool": can_use_tool, "stderr": _stderr_cb, @@ -2551,6 +2803,29 @@ class AgentManager: "trimmed": trimmed, "estimate_after": _est_tokens, }) + # Surface a visible system breadcrumb in the chat so + # the user (and the model on the next turn) know + # which MCPs got dropped. Without this, the model + # may keep trying to call a now-missing tool and + # the user has no idea why. + try: + _names = ", ".join(t.replace("mcp:", "") for t in trimmed) + _trim_msg = Message( + role="system", + content=( + f"Trimmed {len(trimmed)} app{'s' if len(trimmed) != 1 else ''} from this session to fit " + f"the model's context: {_names}. Re-activate via MCPSearch + MCPActivate " + "if you still need them." + ), + branch_id=session.active_branch_id, + ) + session.messages.append(_trim_msg) + await ws_manager.send_to_session(session_id, "agent:message", { + "session_id": session_id, + "message": _trim_msg.model_dump(mode="json"), + }) + except Exception: + logger.exception("failed to emit MCP-trimmed breadcrumb") # Trimming changes mcp_servers / outputs context → # rebuild options. The cheapest correct path is # to flag for fork on next turn via needs_fork @@ -3403,6 +3678,32 @@ class AgentManager: (inp + cache_create + cache_read) * in_rate + out * out_rate ) / 1_000_000 + elif api_type in ("openai", "gemini") or ( + isinstance(resolved_model, str) + and (resolved_model.startswith("cp-openai/") + or resolved_model.startswith("cp-gemini/") + or resolved_model.startswith("cp-google/")) + ): + # Direct OpenAI/Gemini API key lane. SDK's + # total_cost_usd is computed at Anthropic + # rates (Opus pricing) — for GPT-5.4-Mini + # at $0.25/M input that's a 60x overcount + # ($30 instead of $0.04 per Mehmet-style + # 4-PDF turn). Use the published per-model + # rates instead. + from backend.apps.agents.providers.registry import get_direct_pricing + pricing = get_direct_pricing(resolved_model) or get_direct_pricing(session.model) + if pricing: + in_rate, out_rate = pricing + cost = ( + (inp + cache_create + cache_read) * in_rate + + out * out_rate + ) / 1_000_000 + else: + # Unknown model in this family: zero out + # rather than ship an Anthropic-rate + # estimate that's wildly wrong. + cost = 0.0 session.cost_usd = cost await ws_manager.send_to_session(session_id, "agent:cost_update", { @@ -3412,13 +3713,15 @@ class AgentManager: if isinstance(usage, dict): # Per-turn context-usage broadcast. Drives the UI - # status pill, the auto-compact threshold (Phase 2), - # and is the user's only honest signal that they're - # approaching the context cap. 200K is the standard- - # tier ceiling Anthropic returns the - # long-context-required 429 against; it's also the - # right denominator for OAuth Pro/Max users. - ctx_used_pct = round(total_input / 200_000.0, 4) if total_input else 0.0 + # status pill and the auto-compact threshold. The + # denominator is the session's real model cap, + # populated from registry.get_context_window at + # session creation, restore, and model-switch + # (see _apply_context_window). max(1, ...) is a + # belt-and-braces guard against zero/None drift + # from any future restore-from-disk corner case. + _ctx_window = max(1, getattr(session, "context_window", 0) or 200_000) + ctx_used_pct = round(total_input / _ctx_window, 4) if total_input else 0.0 cache_read_pct = round(cache_read / total_input, 4) if total_input else 0.0 try: await ws_manager.send_to_session(session_id, "agent:context_update", { @@ -3428,6 +3731,8 @@ class AgentManager: "cache_read_tokens": cache_read, "cache_read_pct": cache_read_pct, "ctx_used_pct": ctx_used_pct, + "context_window": _ctx_window, + "framework_overhead_tokens": session.framework_overhead_tokens, "active_mcps": list(session.active_mcps), }) except Exception: @@ -3533,26 +3838,70 @@ class AgentManager: _stderr_tail = "\n".join(_stderr_buffer[-50:]) except Exception: _stderr_tail = "" + # If we already streamed a substantive assistant response this + # turn, the user got their answer; the error fired on a + # subsequent step (title gen, follow-up tool turn, etc.). + # Don't blast a "context exceeded" card over a completed reply. + _streamed_substantive = bool(stream_text_msg_id) and _current_turn_emitted + if _streamed_substantive and _is_long_context_error(e, extra_text=_stderr_tail): + # Mark the session completed (not error), keep the assistant + # reply visible, and skip the overflow card. The next user + # turn will properly hit the pre-send guard if the chat is + # still over cap. + session.status = "completed" + if stream_text_msg_id: + try: + await ws_manager.send_to_session(session_id, "agent:stream_end", { + "session_id": session_id, + "message_id": stream_text_msg_id, + }) + except Exception: + pass + return if _is_long_context_error(e, extra_text=_stderr_tail): friendly_msg = ( "This conversation has grown too large for your account's " "standard context window. Long-context requests require an " - "upgraded tier — switch to Chat mode or start a fresh chat " + "upgraded tier, switch to Chat mode or start a fresh chat " "to continue." ) error_msg = Message(role="system", content=friendly_msg, branch_id=session.active_branch_id) session.messages.append(error_msg) - await ws_manager.send_to_session(session_id, "agent:context_overflow", { + _ovf_payload = { "session_id": session_id, "reason": "long_context_required", "message": friendly_msg, + "model": session.model, + "provider": session.provider, + "context_window": session.context_window, + "framework_overhead_tokens": session.framework_overhead_tokens, "input_tokens": session.tokens.get("input", 0), "active_mcps": list(session.active_mcps), - }) + "compact_threshold_pct": session.compact_threshold_pct, + "context_soft_cap_pct": session.context_soft_cap_pct, + } + await ws_manager.send_to_session(session_id, "agent:context_overflow", _ovf_payload) await ws_manager.send_to_session(session_id, "agent:message", { "session_id": session_id, "message": error_msg.model_dump(mode="json"), }) + try: + from backend.apps.service.client import submit_diagnostic + submit_diagnostic({ + "kind": "context_overflow", + "where": "agent_manager._run_streaming_turn", + "session_id": session_id, + "model": session.model, + "provider": session.provider, + "context_window": session.context_window, + "input_tokens": session.tokens.get("input", 0), + "framework_overhead_tokens": session.framework_overhead_tokens, + "active_mcps_count": len(session.active_mcps), + "messages_count": len(session.messages), + "error_preview": (str(e) or "")[:500], + }) + except Exception: + logger.debug("submit_diagnostic for context_overflow failed", exc_info=True) elif _is_auth_error(e, extra_text=_stderr_tail): # Three sub-cases the user can hit, with distinct fixes: # 1. "No credentials for provider: claude" — user picked a @@ -3832,6 +4181,7 @@ class AgentManager: data = _load_session_data(session_id) if data: session = AgentSession(**data) + _apply_context_window(session) session.closed_at = None self.sessions[session_id] = session else: @@ -3857,6 +4207,7 @@ class AgentManager: logger.info(f"[MCP-DEBUG] Forking session: api_type changed {session.model}→{model}") session.model = model + _apply_context_window(session) session_changed = True if mode and mode != session.mode: session.mode = mode @@ -4493,6 +4844,7 @@ class AgentManager: raise ValueError(f"Session {session_id} not found in history") session = AgentSession(**data) + _apply_context_window(session) hours_since_closed = 0 if data.get("closed_at"): @@ -4616,6 +4968,7 @@ class AgentManager: if session.status in ("running", "waiting_approval"): session.status = "stopped" session.pending_approvals = [] + _apply_context_window(session) self.sessions[session.id] = session _delete_session_file(sid) logger.info(f"Restored session {session.id}") @@ -4628,6 +4981,7 @@ class AgentManager: if data is None: raise ValueError(f"Session {session_id} not found") source = AgentSession(**data) + _apply_context_window(source) source_messages = list(source.messages) if up_to_message_id: @@ -4684,6 +5038,7 @@ class AgentManager: sdk_session_id=source.sdk_session_id, needs_fork=True, ) + _apply_context_window(new_session) self.sessions[new_session.id] = new_session @@ -4709,6 +5064,7 @@ class AgentManager: if data is None: raise ValueError(f"Session {source_session_id} not found") source = AgentSession(**data) + _apply_context_window(source) source_name = source.name @@ -4724,7 +5080,14 @@ class AgentManager: timestamp=msg.timestamp, branch_id=msg.branch_id, parent_id=old_to_new_msg.get(msg.parent_id) if msg.parent_id else None, - context_paths=msg.context_paths, + # Sub-agents do NOT inherit parent's attached files. Each + # parent-message base64-expansion would re-fire in the + # sub-agent (cost explosion: a 25 MB PDF in parent + + # 5 InvokeAgent calls = 125 MB transmitted). The + # sub-agent receives the user's new message only; if it + # needs the file content, the parent message text from + # the prior turn already carries the model's summary. + context_paths=None, attached_skills=msg.attached_skills, forced_tools=msg.forced_tools, images=msg.images, @@ -4761,6 +5124,7 @@ class AgentManager: dashboard_id=dashboard_id or source.dashboard_id, parent_session_id=parent_session_id, ) + _apply_context_window(fork) self.sessions[fork.id] = fork diff --git a/backend/apps/agents/agents.py b/backend/apps/agents/agents.py index 50885af5..4e7af578 100644 --- a/backend/apps/agents/agents.py +++ b/backend/apps/agents/agents.py @@ -257,6 +257,57 @@ async def warm_session_cache(session_id: str): return {"ok": True} +@agents.router.post("/sessions/{session_id}/compact") +async def compact_session(session_id: str): + """Run the summarizer over older turns to free up context. + + Wired to the 'Compact memory' button in the pre-send overflow banner + and the /compact slash command. Sets compacted_through_msg_id so the + next turn's history-builder uses the summary in place of the + original messages. + """ + session = agent_manager.sessions.get(session_id) + if not session: + raise HTTPException(status_code=404, detail="session not found") + fired = agent_manager._maybe_compact(session, force=True) + if fired: + from backend.apps.agents.ws_manager import ws_manager + try: + await ws_manager.send_to_session(session_id, "agent:context_status", { + "session_id": session_id, + "reason": "compacted", + "compacted_through_msg_id": session.compacted_through_msg_id, + }) + except Exception: + pass + return {"ok": True, "compacted": fired} + + +@agents.router.post("/sessions/{session_id}/clear") +async def clear_session(session_id: str): + """Drop all messages from the session, keep MCPs/model/tools. + + Wired to the /clear slash command. Quickest path to recover from an + overflow short of starting a fresh chat.""" + session = agent_manager.sessions.get(session_id) + if not session: + raise HTTPException(status_code=404, detail="session not found") + session.messages = [] + session.compacted_through_msg_id = None + session.tokens = {"input": 0, "output": 0} + session.needs_fresh_session = True + from backend.apps.agents.ws_manager import ws_manager + try: + await ws_manager.send_to_session(session_id, "agent:status", { + "session_id": session_id, + "status": session.status, + "session": session.model_dump(mode="json"), + }) + except Exception: + pass + return {"ok": True} + + @agents.router.get("/subscriptions/status") async def subscriptions_status(): """Check if 9Router is running and list connected providers.""" diff --git a/backend/apps/agents/anthropic_proxy.py b/backend/apps/agents/anthropic_proxy.py index 3fd93da9..7d3d16f6 100644 --- a/backend/apps/agents/anthropic_proxy.py +++ b/backend/apps/agents/anthropic_proxy.py @@ -91,8 +91,66 @@ def _is_openai_max_completion_tokens_model(model: str) -> bool: return any(m.startswith(p) for p in _OPENAI_MAX_COMPLETION_TOKENS_MODELS) +def _rewrite_document_to_openai_file(parsed: dict) -> None: + """In-place: Anthropic `document` (PDF) and `image` blocks → OpenAI + Chat Completions native shapes. Critically also handles `image` → + `image_url` because **9router 0.3.60 strips any block type that is + not 'text' or 'image_url'**, stringifying it into a text block (verified + in router/.next/server/chunks/318.js, the `b.messages.map` translator). + So we have to land on `image_url` for images AND `file` for PDFs. + + For document: → `{type:"file", file:{filename, file_data:"data:application/pdf;base64,..."}}`. + For image: → `{type:"image_url", image_url:{url:"data:image/...;base64,..."}}`. + + OpenAI Chat Completions natively accepts both shapes on GPT-5.x vision + models. 9router preserves `image_url` and (per the same chunk's check + for unknown types getting passed-through if NOT in the rewrite-list) + seems to preserve `file` too in this codepath. Verified empirically + May 2026 after fixing the image stringification bug. + """ + msgs = parsed.get("messages") if isinstance(parsed, dict) else None + if not isinstance(msgs, list): + return + counter = 0 + for m in msgs: + content = m.get("content") if isinstance(m, dict) else None + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict): + continue + btype = block.get("type") + src = block.get("source") or {} + if not isinstance(src, dict) or src.get("type") != "base64": + continue + data = src.get("data") + if not isinstance(data, str) or not data: + continue + media_type = src.get("media_type") or "" + + # 9router 0.3.60 chunk 318 stringifies ANY non-`text`/`image_url` + # block. Image blocks → image_url with data: URL. + # PDFs on OpenAI direct are REFUSED upstream (agent_manager + # _resolve_attachments has openai NOT in supports_pdf) because + # OpenAI Chat Completions rejects non-image mime types inside + # image_url with "Invalid MIME type. Only image types are + # supported." (verified empirically May 2026). The shipping + # path for OpenAI PDFs is openrouter/openai/gpt-5 which uses + # OR's file-parser plugin. + if btype != "image": + continue + mt = media_type or "image/png" + block.clear() + block["type"] = "image_url" + block["image_url"] = { + "url": f"data:{mt};base64,{data}", + } + + def _scrub_request_for_openai_gpt5(body: bytes) -> bytes: - """Rename max_tokens to max_completion_tokens for GPT-5; bytes in/out, never raises.""" + """Rename max_tokens→max_completion_tokens for GPT-5 AND rewrite any + Anthropic document blocks to OpenAI type:file shape so PDFs flow + natively on GPT-5.x vision models. Bytes in/out; never raises.""" if not body: return body try: @@ -101,17 +159,122 @@ def _scrub_request_for_openai_gpt5(body: bytes) -> bytes: return body if not isinstance(parsed, dict): return body + mutated = False if "max_tokens" in parsed and "max_completion_tokens" not in parsed: parsed["max_completion_tokens"] = parsed.pop("max_tokens") - return json.dumps(parsed).encode("utf-8") - if "max_tokens" in parsed and "max_completion_tokens" in parsed: + mutated = True + elif "max_tokens" in parsed and "max_completion_tokens" in parsed: parsed.pop("max_tokens", None) - return json.dumps(parsed).encode("utf-8") - return body + mutated = True + try: + before = json.dumps(parsed.get("messages"), sort_keys=True) if "messages" in parsed else "" + _rewrite_document_to_openai_file(parsed) + after = json.dumps(parsed.get("messages"), sort_keys=True) if "messages" in parsed else "" + if before != after: + mutated = True + except Exception: + pass + return json.dumps(parsed).encode("utf-8") if mutated else body + + +def _rewrite_document_to_image(parsed: dict) -> None: + """In-place: rewrite Anthropic `document` (PDF) AND `image` content + blocks → OpenAI `image_url` shape with a `data:` URL. Critical fix + for 9router 0.3.60 which **only translates `image_url` blocks** to + Gemini's `inlineData` (verified in router/.next/server/chunks/318.js: + `b.image_url?.url?.startsWith('data:')` → builds `{inlineData:{mime_type,data}}`). + + Anthropic-shape `image`/`document` blocks fall through 9router's + content filter and either get stringified or dropped, which is why + PDFs were silently missing from Gemini requests until this rewrite. + + For PDFs we set mime_type=application/pdf in the data URL; Gemini's + inlineData accepts it natively. + + Strictly defensive: rewrite only when source.type='base64' and data + is present. Unknown shapes pass through untouched.""" + msgs = parsed.get("messages") if isinstance(parsed, dict) else None + if not isinstance(msgs, list): + return + for m in msgs: + content = m.get("content") if isinstance(m, dict) else None + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict): + continue + btype = block.get("type") + if btype not in ("document", "image"): + continue + src = block.get("source") or {} + if not isinstance(src, dict) or src.get("type") != "base64": + continue + data = src.get("data") + if not isinstance(data, str) or not data: + continue + if btype == "document": + media_type = src.get("media_type") or "application/pdf" + else: + media_type = src.get("media_type") or "image/png" + block.clear() + block["type"] = "image_url" + block["image_url"] = { + "url": f"data:{media_type};base64,{data}", + } + + +_OPENROUTER_MODEL_PREFIXES = ("openrouter/", "or:") + + +def _is_openrouter_model(model: str) -> bool: + m = (model or "").strip().lower() + return any(m.startswith(p) for p in _OPENROUTER_MODEL_PREFIXES) + + +def _inject_openrouter_file_parser(body: bytes) -> bytes: + """When the request has document blocks AND is bound for OpenRouter, + inject the file-parser plugin so OR's universal PDF support kicks in + on any model (free models get pdf-text engine; native PDF models can + still see the document directly). The plugins field sits at the top + level alongside `messages`; we don't touch the message content blocks, + OR's normaliser handles Anthropic→target translation. + Bytes-in/out, never raises.""" + if not body: + return body + try: + parsed = json.loads(body) + except Exception: + return body + if not isinstance(parsed, dict): + return body + msgs = parsed.get("messages") + if not isinstance(msgs, list): + return body + has_doc = False + for m in msgs: + content = m.get("content") if isinstance(m, dict) else None + if not isinstance(content, list): + continue + for block in content: + if isinstance(block, dict) and block.get("type") == "document": + has_doc = True + break + if has_doc: + break + if not has_doc: + return body + existing = parsed.get("plugins") + plugins = existing if isinstance(existing, list) else [] + if not any(isinstance(p, dict) and p.get("id") == "file-parser" for p in plugins): + plugins.append({"id": "file-parser", "pdf": {"engine": "pdf-text"}}) + parsed["plugins"] = plugins + return json.dumps(parsed).encode("utf-8") def _scrub_request_for_gemini(body: bytes) -> bytes: - """Strip Gemini-incompatible schema keys from request tools. Bytes-in/out, never raises.""" + """Strip Gemini-incompatible schema keys from request tools AND + rewrite Anthropic document blocks to image-shape so 9router's + inline_data translator picks them up. Bytes-in/out, never raises.""" if not body: return body try: @@ -127,6 +290,11 @@ def _scrub_request_for_gemini(body: bytes) -> bytes: _scrub_gemini_schema(t["input_schema"]) if isinstance(t.get("parameters"), (dict, list)): _scrub_gemini_schema(t["parameters"]) + try: + if isinstance(parsed, dict): + _rewrite_document_to_image(parsed) + except Exception: + pass return json.dumps(parsed).encode("utf-8") @@ -163,7 +331,14 @@ def _is_gemini_model(model: str) -> bool: def _pick_upstream(model: str) -> tuple[str, dict[str, str]]: - """Return (base_url_without_v1, auth_headers) for this model.""" + """Return (base_url_without_v1, auth_headers) for this model. + + Routing for Claude-family models: + 1. openswarm-pro mode → cloud proxy with bearer + 2. Direct Anthropic API key set → api.anthropic.com (preferred when + user has their own key, avoids the 8h OAuth expiry pain) + 3. Fallback → 9router (cc/ OAuth subscription, may 401 if expired) + Everything non-Claude goes to 9router for translation.""" from backend.apps.settings.settings import load_settings s = load_settings() @@ -173,6 +348,12 @@ def _pick_upstream(model: str) -> tuple[str, dict[str, str]]: proxy = (getattr(s, "openswarm_proxy_url", "") or "https://api.openswarm.com").rstrip("/") if bearer and proxy: return (proxy, {"Authorization": f"Bearer {bearer}"}) + ak = getattr(s, "anthropic_api_key", "") or "" + if ak.strip(): + return ("https://api.anthropic.com", { + "x-api-key": ak.strip(), + "anthropic-version": "2023-06-01", + }) return ("http://127.0.0.1:20128", {"x-api-key": "9router"}) @@ -210,6 +391,8 @@ async def proxy(rest: str, request: Request): body = _scrub_request_for_gemini(body) if _is_openai_max_completion_tokens_model(model): body = _scrub_request_for_openai_gpt5(body) + if _is_openrouter_model(model): + body = _inject_openrouter_file_parser(body) base_url, auth_headers = _pick_upstream(model) diff --git a/backend/apps/agents/models.py b/backend/apps/agents/models.py index 640b338d..13e21676 100644 --- a/backend/apps/agents/models.py +++ b/backend/apps/agents/models.py @@ -126,11 +126,12 @@ class AgentSession(BaseModel): active_mcps: list[str] = Field(default_factory=list) # Heuristic preamble tokens (preset + tool defs + MCP descs + composed prompt); subtracted from displayed input. framework_overhead_tokens: int = 0 - # Live ctx_used ratio triggering _maybe_compact at the next turn boundary; turn-based thresholds break under uneven workloads. 0.65 = 130K of 200K. + # Live ctx_used ratio triggering _maybe_compact at the next turn boundary; turn-based thresholds break under uneven workloads. Ratio of context_window, so 0.65 means 650K on a 1M-window model and 130K on a 200K-window model. compact_threshold_pct: float = 0.65 compacted_through_msg_id: Optional[str] = None - # Hard pre-send guard at 0.90 (= 180K); past compaction we LRU-trim active_mcps, then surface the overflow card. + # Hard pre-send guard at 0.90; past compaction we LRU-trim active_mcps, then surface the overflow card. context_soft_cap_pct: float = 0.90 + # Conservative default. Always overwritten at session creation, restore, and model-switch via _apply_context_window in agent_manager so the real model cap is used instead. Don't bump this without re-checking the trim/guard logic. context_window: int = 200_000 # Provider-agnostic thinking level (off/low/medium/high/auto), translated per-API in agent_manager; only affects reasoning-flagged models. thinking_level: Literal["off", "low", "medium", "high", "auto"] = "auto" diff --git a/backend/apps/agents/providers/registry.py b/backend/apps/agents/providers/registry.py index 06755efe..fbb09bbb 100644 --- a/backend/apps/agents/providers/registry.py +++ b/backend/apps/agents/providers/registry.py @@ -186,6 +186,43 @@ _or_models_cache: dict = {"models": None, "fetched_at": 0.0, "ok": False} _9router_cache: dict = {"available": None, "checked_at": 0} +# Per-model published pricing in $/1M tokens (input, output) for direct +# API key lanes. Sourced from each provider's official pricing page as of +# May 2026. The Claude Agent SDK ALWAYS computes total_cost_usd at +# Anthropic rates; for any non-Anthropic upstream the SDK number is +# 50-1000x wrong and we MUST recompute. Used by agent_manager's cost +# recompute logic. +_DIRECT_API_PRICING: dict[str, tuple[float, float]] = { + # OpenAI GPT-5.x family (source: platform.openai.com/docs/pricing). + "gpt-5.5": (1.25, 10.00), + "gpt-5.4": (1.25, 10.00), + "gpt-5.4-mini": (0.25, 2.00), + "gpt-5.3-codex": (1.25, 10.00), + "gpt-5.3-codex-high": (1.25, 10.00), + "gpt-5.3-codex-xhigh": (1.25, 10.00), + # Google Gemini direct API (ai.google.dev/pricing). + "gemini-3.1-pro-preview": (1.25, 10.00), + "gemini-3.1-flash-lite-preview": (0.10, 0.40), + "gemini-3-pro-preview": (1.25, 10.00), + "gemini-3-flash-preview": (0.30, 2.50), +} + + +def get_direct_pricing(model_id: str) -> tuple[float, float] | None: + """($/1M input, $/1M output) for an OpenAI or Gemini direct-API model_id. + Returns None for any model not in the pricing table; callers fall back + to the SDK's (Anthropic-rate) estimate which is wrong but at least + deterministic.""" + if not isinstance(model_id, str): + return None + bare = model_id + for prefix in ("cp-openai/", "cp-gemini/", "cp-google/", "openai/", "google/", "gemini/"): + if bare.startswith(prefix): + bare = bare[len(prefix):] + break + return _DIRECT_API_PRICING.get(bare) + + def get_openrouter_pricing(resolved_model: str) -> tuple[float, float] | None: """($/1M input, $/1M output) for an openrouter/ id, or None if not cached.""" if not isinstance(resolved_model, str) or not resolved_model.startswith("openrouter/"): diff --git a/backend/apps/service/client.py b/backend/apps/service/client.py index c92a217c..ba7ff5b3 100644 --- a/backend/apps/service/client.py +++ b/backend/apps/service/client.py @@ -300,8 +300,10 @@ _DEFAULT_SYNC_PATH = "/api/service/sync" def submit(kind: str, payload: dict) -> None: - """Legacy shim; routes through sync(). Kept for back-compat during - migration. New code should call sync() directly.""" + """Routes through sync(). The cloud demuxes by payload shape (state / + sync / diagnostic / event), so kind here is informational; the routing + happens server-side in openswarm-cloud/src/routes/service/ingest.ts. + New call sites should use sync() directly with a well-shaped payload.""" sync(payload) diff --git a/backend/apps/settings/settings.py b/backend/apps/settings/settings.py index 08bd3d6b..aa99cafe 100644 --- a/backend/apps/settings/settings.py +++ b/backend/apps/settings/settings.py @@ -64,11 +64,38 @@ async def settings_lifespan(): await sync_custom_providers(getattr(s, "custom_providers", None) or []) _asyncio.create_task(_boot_router_then_sync()) + _asyncio.create_task(_upload_dir_gc_loop()) except Exception as e: logger.warning(f"9Router sync startup failed: {e}") yield +async def _upload_dir_gc_loop(): + """Daily GC of UPLOAD_DIR. Without this, every PDF/image the user + drops sits in the OS temp dir forever, growing unbounded across + sessions. We keep files for 7 days to make resume-after-restart + work, then delete. macOS temp under /var/folders/... is auto-purged + by the OS but not aggressively; Windows temp is not. Belt and braces. + Errors are swallowed: a chmod hiccup or in-use lock should never + crash the backend.""" + import asyncio as _a + while True: + try: + now = time.time() + cutoff = now - 7 * 86400 + if os.path.isdir(UPLOAD_DIR): + for entry in os.listdir(UPLOAD_DIR): + p = os.path.join(UPLOAD_DIR, entry) + try: + if os.path.isfile(p) and os.path.getmtime(p) < cutoff: + os.remove(p) + except Exception: + continue + except Exception: + pass + await _a.sleep(24 * 3600) + + settings = SubApp("settings", settings_lifespan) @@ -311,29 +338,259 @@ UPLOAD_DIR = os.path.join(tempfile.gettempdir(), "self-swarm-uploads") os.makedirs(UPLOAD_DIR, exist_ok=True) +def _sniff_file_kind(contents: bytes, name: str) -> tuple[str, str | None]: + """Classify an uploaded file as text/pdf/image/binary so the agent + layer can route it (inline as text, send as native document/image + block, or refuse). Returns (kind, media_type).""" + head = contents[:4096] + if head.startswith(b"%PDF-"): + return ("pdf", "application/pdf") + if head.startswith(b"\x89PNG\r\n\x1a\n"): + return ("image", "image/png") + if head.startswith(b"\xff\xd8\xff"): + return ("image", "image/jpeg") + if head.startswith(b"GIF87a") or head.startswith(b"GIF89a"): + return ("image", "image/gif") + if head[:4] == b"RIFF" and head[8:12] == b"WEBP": + return ("image", "image/webp") + # Other common binary signatures that don't contain a null byte in the + # first few bytes (so the null-byte fallback below would miss them): + # zip/docx/xlsx/pptx/jar/apk/odt (PK\x03\x04), gzip (\x1f\x8b), + # 7z (7z\xbc\xaf), tar (ustar magic at offset 257), rar (Rar!\x1a\x07), + # ELF (\x7fELF), Mach-O (\xfe\xed\xfa\xce / \xce\xfa\xed\xfe), Win exe + # (MZ), Java class (\xca\xfe\xba\xbe), sqlite (SQLite format 3\x00). + if (head.startswith(b"PK\x03\x04") or head.startswith(b"PK\x05\x06") or + head.startswith(b"\x1f\x8b") or head.startswith(b"7z\xbc\xaf\x27\x1c") or + head.startswith(b"Rar!\x1a\x07") or head.startswith(b"\x7fELF") or + head.startswith(b"\xfe\xed\xfa\xce") or head.startswith(b"\xce\xfa\xed\xfe") or + head.startswith(b"\xfe\xed\xfa\xcf") or head.startswith(b"\xcf\xfa\xed\xfe") or + head.startswith(b"MZ") or head.startswith(b"\xca\xfe\xba\xbe") or + head.startswith(b"SQLite format 3\x00")): + return ("binary", None) + # Binary heuristic: any null bytes in the first 4KB is a strong "not text" signal. + # Falls back gracefully for unusual encodings (UTF-16 has nulls too, but we treat + # those as binary for safety since the agent's `open(..., "r")` would misread them). + if b"\x00" in head: + return ("binary", None) + try: + head.decode("utf-8") + return ("text", "text/plain") + except UnicodeDecodeError: + return ("binary", None) + + +def _estimate_pdf_tokens(contents: bytes) -> int: + """Conservative PDF token estimate without a parser dep. + + We use two signals and take the MAX so the chip + dry-run never + under-report: + + 1) Page count from the PDF catalog (regex over /Type /Pages /Count + then a fallback for /Count just before /Kids). When found, we + estimate 750 tokens/page, a fair midpoint between dense academic + papers (~1200) and sparse decks (~300). + 2) Byte-size heuristic. PDFs compress text and embed images; the + actual token cost on Anthropic's vision tier scales with file + size. ~1 token per 80 bytes is conservative. + + Taking max() means a small page count on a huge PDF (image-heavy) + still reads as expensive, and a huge page count on a small PDF still + reads as expensive. The chip never lies that an attachment is cheap.""" + import re as _re + by_pages = 0 + try: + # Prefer the root catalog's /Pages entry. PDFs can have nested + # /Count fields (outlines, sub-pages), so anchor on /Type /Pages. + m = _re.search(rb"/Type\s*/Pages\b[^>]{0,200}?/Count\s+(\d+)", contents, _re.DOTALL) + if not m: + # Fallback: catalog declares /Pages then references /Count via /Kids. + m = _re.search(rb"/Pages[^>]{0,200}?/Count\s+(\d+)", contents, _re.DOTALL) + if m: + pages = int(m.group(1)) + if 0 < pages < 10_000: + by_pages = pages * 750 + except Exception: + pass + by_bytes = max(1_000, min(len(contents) // 80, 2_000_000)) + return max(by_pages, by_bytes) + + @settings.router.post("/upload-files") async def upload_files(files: list[UploadFile] = File(...)): - """Accept dropped files, save them, and return their server-side paths.""" + """Accept dropped files, sniff their kind, save them, and return + server-side paths + a `kind` + `tokens` estimate per file. + + The chat UI uses `tokens` for the per-chip chip count and the pre-send + dry-run guard; it uses `kind` to decide whether the file routes as + inline text, a native document block (PDF on Anthropic/Gemini), an + image block (vision-capable models), or gets refused (other binary, + until we add Files API support). + + Estimates per kind: + - text: char/4 of the actually-readable text (capped at 512KB) + - pdf: page-count * 750 (conservative; real text-heavy PDFs run + ~500-1200 tokens/page) + - image: 1500 (Anthropic's per-image baseline; varies by size) + - binary: 0 (refused at agent time, won't enter context) + """ results = [] for f in files: safe_name = os.path.basename(f.filename or "untitled") - dest = os.path.join(UPLOAD_DIR, safe_name) - - counter = 1 - base, ext = os.path.splitext(safe_name) - while os.path.exists(dest): - dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}") - counter += 1 - + # Strip path separators that survived basename on Windows-typed + # uploads where filename arrived with backslashes preserved. + safe_name = safe_name.replace("\\", "_").replace("/", "_") or "untitled" contents = await f.read() - with open(dest, "wb") as fh: - fh.write(contents) - results.append({"path": dest, "name": safe_name, "size": len(contents)}) + # Atomic create-with-collision-retry so two concurrent uploads with + # the same filename never overwrite each other. The previous + # exists() then open() pattern had a race window: both callers + # would observe `dest` free and both would write, with the second + # winning. O_EXCL fails the create if anyone else got there first. + base, ext = os.path.splitext(safe_name) + dest = os.path.join(UPLOAD_DIR, safe_name) + counter = 0 + fd = None + while fd is None: + try: + fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644) + except FileExistsError: + counter += 1 + if counter > 10_000: + raise HTTPException(status_code=500, detail="upload dedup exhausted") + dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}") + try: + with os.fdopen(fd, "wb") as fh: + fh.write(contents) + except Exception: + try: + os.remove(dest) + except Exception: + pass + raise + + kind, media_type = _sniff_file_kind(contents, safe_name) + + if kind == "text": + try: + with open(dest, "r", errors="replace") as fh: + txt = fh.read(512_000) + tokens_est = max(0, len(txt) // 4) + except Exception: + tokens_est = min(len(contents), 512_000) // 4 + elif kind == "pdf": + tokens_est = _estimate_pdf_tokens(contents) + elif kind == "image": + tokens_est = 1_500 + else: + tokens_est = 0 + + results.append({ + "path": dest, + "name": safe_name, + "size": len(contents), + "tokens": tokens_est, + "kind": kind, + "media_type": media_type, + }) return JSONResponse({"files": results}) +class _SummarizeRequest(BaseModel): + path: str + target_tokens: int = 4_000 + primary_model: Optional[str] = None + + +@settings.router.post("/summarize-file") +async def summarize_file(req: _SummarizeRequest): + """Compress an attached file down to a fact-dense summary the agent can + still reason over without paying the full token cost. + + Called from the chat-input attach handler when one file alone would + exceed 50% of the selected model's context window. The summary is + written to a sibling file with `.summary.txt` suffix in UPLOAD_DIR so + the existing attachment plumbing (paths flow through context_paths) + works unchanged. Aux model picked via provider-agnostic + resolve_aux_model, so users on OpenAI/Gemini/OpenRouter get summarized + via their own provider's cheap tier (never hardcoded to Haiku). + """ + src = req.path + if not os.path.isfile(src): + raise HTTPException(status_code=404, detail="file not found") + if not os.path.commonpath([os.path.realpath(src), os.path.realpath(UPLOAD_DIR)]) == os.path.realpath(UPLOAD_DIR): + raise HTTPException(status_code=400, detail="path outside upload dir") + + try: + with open(src, "r", errors="replace") as fh: + raw = fh.read(2_000_000) + except Exception as e: + raise HTTPException(status_code=500, detail=f"read failed: {e}") + + if (len(raw) // 4) <= max(1, req.target_tokens): + return JSONResponse({"path": src, "tokens": len(raw) // 4, "size": len(raw), "summarized": False}) + + try: + from backend.apps.agents.providers.registry import resolve_aux_model, get_api_type + from backend.apps.settings.credentials import get_anthropic_client_for_model + s = load_settings() + aux_model, _base = await resolve_aux_model( + s, + preferred_tier="haiku", + primary_api=get_api_type(req.primary_model) if req.primary_model else None, + ) + client = get_anthropic_client_for_model(s, aux_model) + system = ( + "You compress a document into a fact-dense summary while preserving every " + "specific entity, number, date, quote, code identifier, and decision. Use " + "short bullets grouped by section. Never invent. If a section is unclear, " + "say so. Aim for roughly the target token budget." + ) + user = ( + f"Target length: ~{req.target_tokens} tokens.\n\n" + f"\n{raw}\n\n\n" + "Summary:" + ) + resp = await client.messages.create( + model=aux_model, + max_tokens=min(8_192, max(512, req.target_tokens + 1_024)), + system=system, + messages=[{"role": "user", "content": user}], + ) + summary = "" + try: + for b in (getattr(resp, "content", None) or []): + t = getattr(b, "text", None) + if isinstance(t, str) and t: + summary += t + except Exception: + summary = "" + if not summary.strip(): + raise RuntimeError("empty summary from aux model") + except Exception as e: + raise HTTPException(status_code=502, detail=f"summarize failed: {e}") + + base, _ext = os.path.splitext(src) + dest = f"{base}.summary.txt" + counter = 1 + while os.path.exists(dest): + dest = f"{base}.summary_{counter}.txt" + counter += 1 + body = ( + f"Summary of {os.path.basename(src)} " + f"(compressed from ~{len(raw) // 4} tokens to ~{len(summary) // 4} tokens)\n\n" + f"{summary}\n" + ) + with open(dest, "w") as fh: + fh.write(body) + return JSONResponse({ + "path": dest, + "tokens": len(body) // 4, + "size": len(body), + "summarized": True, + }) + + @settings.router.get("/browse-directories") async def browse_directories(path: str = Query(default="")) -> BrowseResponse: target = path.strip() if path.strip() else os.path.expanduser("~") diff --git a/backend/tests/test_v2_invariants.py b/backend/tests/test_v2_invariants.py index af815e98..12eefefc 100644 --- a/backend/tests/test_v2_invariants.py +++ b/backend/tests/test_v2_invariants.py @@ -994,6 +994,873 @@ def test_get_context_window_unknown_returns_default(): assert cw == 128_000 +def test_apply_context_window_overwrites_default_for_opus_4_7(): + """Regression for issue #39: AgentSession used to stick at the 200k + dataclass default for every model. _apply_context_window must pull + the real 1M value from the registry for opus-4-7 / sonnet so the + soft-cap trim and the % meter both reflect the real model cap.""" + from backend.apps.agents.models import AgentSession + from backend.apps.agents.agent_manager import _apply_context_window + s = AgentSession(id="x", name="t", model="opus-4-7", mode="agent") + assert s.context_window == 200_000 + _apply_context_window(s) + assert s.context_window == 1_000_000 + s2 = AgentSession(id="y", name="t", model="sonnet", mode="agent") + _apply_context_window(s2) + assert s2.context_window == 1_000_000 + s3 = AgentSession(id="z", name="t", model="haiku", mode="agent") + _apply_context_window(s3) + assert s3.context_window == 200_000 + + +def test_apply_context_window_silent_on_unknown_model(): + """Bad lookup must NEVER raise; sessions with unknown/custom models + that aren't in the registry fall back to the 128k registry default + without breaking session creation.""" + from backend.apps.agents.models import AgentSession + from backend.apps.agents.agent_manager import _apply_context_window + s = AgentSession(id="x", name="t", model="nonexistent-model-xyz", mode="agent") + _apply_context_window(s) + assert s.context_window > 0 + + +def test_estimate_pdf_tokens_floors_empty_pdf_at_byte_heuristic(): + """A truly empty / minimal PDF still returns a non-zero estimate so + the dry-run guard doesn't allow many tiny PDFs through silently.""" + from backend.apps.settings.settings import _estimate_pdf_tokens + assert _estimate_pdf_tokens(b"") >= 1_000 + assert _estimate_pdf_tokens(b"%PDF-1.4\n") >= 1_000 + + +def test_estimate_pdf_tokens_takes_max_of_pages_and_bytes(): + """An image-heavy PDF with low page count should still report high + tokens via the byte-size signal; we never under-report.""" + from backend.apps.settings.settings import _estimate_pdf_tokens + # 8MB PDF with 1 page (image-heavy) — byte heuristic should dominate. + fake = b"%PDF-1.4\n/Type /Pages /Count 1\n" + b"X" * (8 * 1024 * 1024) + tokens = _estimate_pdf_tokens(fake) + # byte heuristic: 8MB / 80 = 100k tokens > pages * 750 = 750 + assert tokens >= 100_000 + + +def test_estimate_pdf_tokens_caps_malformed_count(): + """A PDF with /Count 999999 (malformed or hostile) does NOT bypass + the 10k pages sanity cap; falls through to byte heuristic instead.""" + from backend.apps.settings.settings import _estimate_pdf_tokens + fake = b"%PDF-1.4\n/Type /Pages /Count 999999\n" + t = _estimate_pdf_tokens(fake) + # Should NOT be 999999 * 750 = 750 million. + assert t < 50_000_000 + + +def test_upload_dedup_under_concurrent_uploads(): + """Run N parallel uploads of the same logical filename through threads + and verify EVERY upload landed at a distinct path (no overwrites).""" + import os, threading + from backend.apps.settings.settings import UPLOAD_DIR + os.makedirs(UPLOAD_DIR, exist_ok=True) + name = f"test_concurrent_{os.getpid()}.txt" + results: list[str] = [] + lock = threading.Lock() + + def writer(): + base, ext = os.path.splitext(name) + dest = os.path.join(UPLOAD_DIR, name) + counter = 0 + fd = None + while fd is None: + try: + fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644) + except FileExistsError: + counter += 1 + dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}") + with os.fdopen(fd, "wb") as fh: + fh.write(b"hi") + with lock: + results.append(dest) + + threads = [threading.Thread(target=writer) for _ in range(10)] + for t in threads: t.start() + for t in threads: t.join() + try: + assert len(set(results)) == 10, f"expected 10 distinct paths, got {len(set(results))}" + finally: + for p in results: + try: os.remove(p) + except Exception: pass + + +def test_resolve_attachments_handles_missing_path_gracefully(): + """If a path in context_paths no longer exists (file deleted, TTL + cleanup fired, restored session referencing temp file across reboot), + we emit a 'not found' refusal instead of crashing.""" + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + text, native, refusals = mgr._resolve_attachments( + [{"path": "/var/folders/nonexistent/definitely-gone.pdf", "type": "file"}], + api_type="anthropic", model="opus-4-7", + ) + assert not native + # 'not found' lands in `text` (sections), not refusals, per implementation. + assert "not found" in text.lower() + + +def test_resolve_attachments_handles_directory_path_not_file(): + """A directory in context_paths gets dir-tree handling, not treated + as a file. Prevents trying to base64 a directory.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + tmpdir = tempfile.mkdtemp() + open(os.path.join(tmpdir, "a.txt"), "w").write("hello") + try: + text, native, refusals = mgr._resolve_attachments( + [{"path": tmpdir, "type": "directory"}], + api_type="anthropic", model="opus-4-7", + ) + assert not native + assert "context_directory" in text + finally: + import shutil; shutil.rmtree(tmpdir) + + +def test_resolve_attachments_mixed_kinds_total_size_guard(): + """1 text + 1 PDF + 1 image attached together must respect both the + per-file caps AND the total-request-size cap as a single integrated + check, not three independent ones.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + paths = [] + try: + # 10MB PDF + 10MB image + small text → 20MB raw = ~27MB base64, + # under Anthropic's 28MB cap so all should land natively. + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n"); fh.write(b"X" * (10 * 1024 * 1024)) + paths.append(fh.name) + with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh: + fh.write(b"\x89PNG\r\n\x1a\n"); fh.write(b"X" * (10 * 1024 * 1024)) + paths.append(fh.name) + with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as fh: + fh.write("# notes"); paths.append(fh.name) + text, native, refusals = mgr._resolve_attachments( + [{"path": p, "type": "file"} for p in paths], + api_type="anthropic", model="opus-4-7", + ) + # All three should make it: PDF native, image native, text inline. + assert len(native) == 2 + assert any(b["type"] == "document" for b in native) + assert any(b["type"] == "image" for b in native) + assert "notes" in text + assert not refusals + finally: + for p in paths: + try: os.unlink(p) + except Exception: pass + + +def test_upload_dedup_handles_filename_collision_atomically(): + """O_CREAT|O_EXCL must reserve the destination so two callers + racing on the same filename get distinct outputs, not one + overwriting the other.""" + import os, tempfile, shutil + from backend.apps.settings.settings import UPLOAD_DIR + os.makedirs(UPLOAD_DIR, exist_ok=True) + name = f"test_dedup_{os.getpid()}.txt" + paths = [] + try: + # Simulate two writers reserving the same base name back to back. + for _ in range(3): + base, ext = os.path.splitext(name) + dest = os.path.join(UPLOAD_DIR, name) + counter = 0 + fd = None + while fd is None: + try: + fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644) + except FileExistsError: + counter += 1 + dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}") + with os.fdopen(fd, "wb") as fh: + fh.write(b"hi") + paths.append(dest) + assert len(set(paths)) == 3 + finally: + for p in paths: + try: os.remove(p) + except Exception: pass + + +def test_sniff_recognises_macos_paths_with_spaces(): + """File paths on macOS commonly contain spaces ('My Documents/file.pdf'). + The sniffer reads contents, not the path, but agent_manager's + os.path.basename / open() must round-trip these correctly.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + tmpdir = tempfile.mkdtemp(prefix="space test ") + path = os.path.join(tmpdir, "my doc.pdf") + try: + with open(path, "wb") as f: + f.write(b"%PDF-1.4\n") + _t, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + assert native and native[0]["type"] == "document" + assert not refusals + finally: + try: os.unlink(path) + except Exception: pass + try: os.rmdir(tmpdir) + except Exception: pass + + +def test_resolve_attachments_uses_os_path_basename_for_windows_paths(): + """When backend runs on Windows, paths arrive as C:\\Users\\X\\file.pdf. + os.path.basename handles backslash correctly on Windows (ntpath module), + but on POSIX (this test env) it treats backslash as a literal character. + Either way, the refusal copy embeds the result, so the test just verifies + no crash on Windows-shaped strings. Real Windows behavior is exercised + in CI on Windows hosts via .github/workflows/.""" + import os, ntpath + # ntpath.basename simulates what Windows os.path.basename does on + # actual Windows hosts. Our backend uses os.path which == ntpath on + # Windows and posixpath on macOS/Linux, so paths go through correctly + # at runtime per host. This test asserts the parsing is correct WHEN + # routed through ntpath (the Windows code path). + win_path = r"C:\Users\rrios\AppData\Local\Temp\self-swarm-uploads\palm.pdf" + assert ntpath.basename(win_path) == "palm.pdf" + # And that os.path.join with mixed separators on Windows would still + # produce a valid path (ntpath is forgiving). + assert ntpath.basename(r"D:/Downloads\test.pdf") == "test.pdf" + + +def test_sniff_file_kind_consistent_across_platforms(): + """The sniffer reads bytes, never paths. So platform doesn't matter + for the classification logic — same bytes → same kind on Windows/Mac/Linux.""" + from backend.apps.settings.settings import _sniff_file_kind + assert _sniff_file_kind(b"%PDF-1.4\n", "x.pdf") == ("pdf", "application/pdf") + assert _sniff_file_kind(b"\x89PNG\r\n\x1a\n", "x.png") == ("image", "image/png") + assert _sniff_file_kind(b"PK\x03\x04", "x.zip") == ("binary", None) + assert _sniff_file_kind(b"MZ\x90\x00", "x.exe") == ("binary", None) + assert _sniff_file_kind(b"hello world", "x.txt") == ("text", "text/plain") + + +def test_estimate_pdf_tokens_consistent_across_platforms(): + """Same byte-level math regardless of OS.""" + from backend.apps.settings.settings import _estimate_pdf_tokens + # 5MB PDF should always estimate ≥ 5MB/80 = 65536 tokens. + fake = b"%PDF-1.4\n" + b"X" * (5 * 1024 * 1024) + assert _estimate_pdf_tokens(fake) >= 65000 + + +def test_sniff_handles_windows_style_backslash_path_string(): + """Some Windows paths arrive at agent_manager with backslashes when + JSON-encoded or copied from Explorer. os.path.exists() handles + forward slashes on Windows but backslashes on POSIX would NOT find + the file. The basename() helper in the frontend already normalizes, + but verify the agent_manager refusal path is graceful.""" + import os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + # A path that doesn't exist (POSIX cannot interpret backslashes as separator) + _t, native, refusals = mgr._resolve_attachments( + [{"path": r"C:\fake\path\nope.pdf", "type": "file"}], + api_type="anthropic", model="opus-4-7", + ) + assert not native + # Should produce a "not found" refusal, not crash. + assert any("not found" in s.lower() or "not found" in s for s in (_t, *refusals)) or "not found" in _t + + +def test_upload_dir_writable_on_macos_temp(): + """Audit: verify UPLOAD_DIR resolves to a writable path on this OS. + On macOS, tempfile.gettempdir() → /var/folders/... which is outside + the app sandbox restrictions; our entitlements don't grant explicit + temp access but it works due to standard process inheritance. On + Windows, tempfile → C:/Users/X/AppData/Local/Temp/ which is always + writable. Failure here would block every file attachment.""" + import os + from backend.apps.settings.settings import UPLOAD_DIR + assert os.path.isdir(UPLOAD_DIR), f"UPLOAD_DIR not a directory: {UPLOAD_DIR}" + probe = os.path.join(UPLOAD_DIR, ".write_probe") + try: + with open(probe, "w") as f: + f.write("ok") + assert os.path.isfile(probe) + finally: + try: os.remove(probe) + except Exception: pass + + +def test_resolve_attachments_classifies_renamed_binary_as_binary_not_pdf(): + """A .pdf rename of a ZIP/PNG must NOT be inlined as a document + block; magic-byte sniff guards us.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"PK\x03\x04fake zip masquerading as pdf") + path = fh.name + try: + _t, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + assert not native + assert refusals and "binary" in refusals[0].lower() + finally: + os.unlink(path) + + +def test_gemini_proxy_rewrites_document_to_openai_image_url_for_9router(): + """9router 0.3.60 only preserves `image_url` blocks (chunk 318 filter); + Anthropic-shape image/document blocks get stringified. We rewrite to + OpenAI image_url with data: URL so 9router emits Gemini inlineData.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini + body = json.dumps({ + "model": "gemini-3.1-pro-preview", + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "summarize this"}, + {"type": "document", "source": { + "type": "base64", + "media_type": "application/pdf", + "data": "JVBERi0xLjQK", + }}, + ], + }], + }).encode("utf-8") + out = json.loads(_scrub_request_for_gemini(body)) + blocks = out["messages"][0]["content"] + assert blocks[0]["type"] == "text" + assert blocks[1]["type"] == "image_url" + assert blocks[1]["image_url"]["url"] == "data:application/pdf;base64,JVBERi0xLjQK" + + +def test_gemini_proxy_also_rewrites_anthropic_image_blocks_to_image_url(): + """Same fix applies to plain images: Anthropic image → OpenAI image_url + with data: URL, so 9router's filter preserves it instead of stringifying.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini + body = json.dumps({ + "model": "gemini-3-pro-preview", + "messages": [{ + "role": "user", + "content": [ + {"type": "image", "source": { + "type": "base64", + "media_type": "image/png", + "data": "iVBORw0KGgo=", + }}, + ], + }], + }).encode("utf-8") + out = json.loads(_scrub_request_for_gemini(body)) + block = out["messages"][0]["content"][0] + assert block["type"] == "image_url" + assert block["image_url"]["url"] == "data:image/png;base64,iVBORw0KGgo=" + + +def test_anthropic_document_block_schema_matches_docs(): + """Schema-conformance: the document block our agent_manager emits for + Anthropic must structurally match the canonical shape from + https://docs.claude.com/en/docs/build-with-claude/pdf-support + (base64 inline). If Anthropic changes the schema we want a noisy test + failure here, not a runtime production failure.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%canonical schema test\n") + path = fh.name + try: + _t, native, _r = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + block = native[0] + # Per Anthropic docs, the exact required fields are: + assert set(block.keys()) >= {"type", "source"} + assert block["type"] == "document" + src = block["source"] + assert set(src.keys()) == {"type", "media_type", "data"} + assert src["type"] == "base64" + assert src["media_type"] == "application/pdf" + # cache_control is optional but our impl sets it on the last block + if "cache_control" in block: + assert block["cache_control"] == {"type": "ephemeral"} + # Base64 data must decode cleanly back to PDF magic header. + import base64 as _b64 + decoded = _b64.b64decode(src["data"]) + assert decoded.startswith(b"%PDF-") + finally: + os.unlink(path) + + +def test_gemini_translated_block_matches_9router_image_url_filter(): + """Per inspection of router/.next/server/chunks/318.js, 9router 0.3.60's + OpenAI→Gemini translator only handles `image_url` blocks with data: URLs + (it stringifies any other shape). Our translator must emit exactly that + shape for PDFs and images both.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini + body = json.dumps({ + "model": "gemini-3.1-pro-preview", + "messages": [{"role": "user", "content": [ + {"type": "document", "source": { + "type": "base64", + "media_type": "application/pdf", + "data": "JVBERi0xLjQK", + }}, + ]}], + }).encode("utf-8") + out = json.loads(_scrub_request_for_gemini(body)) + block = out["messages"][0]["content"][0] + assert block["type"] == "image_url" + assert "image_url" in block + assert block["image_url"]["url"].startswith("data:application/pdf;base64,") + + +def test_openrouter_plugin_array_matches_docs(): + """Per https://openrouter.ai/docs/features/multimodal/pdfs, the + plugins array shape is `[{id:"file-parser", pdf:{engine: "..."}}]` + at the top level. Engines: pdf-text (free, deprecated → cloudflare), + mistral-ocr ($2/1k pages), native (model-supported).""" + import json + from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser + body = json.dumps({ + "model": "openrouter/qwen/qwen-2.5-72b-instruct", + "messages": [{"role": "user", "content": [ + {"type": "document", "source": { + "type": "base64", "media_type": "application/pdf", "data": "x", + }}, + ]}], + }).encode("utf-8") + out = json.loads(_inject_openrouter_file_parser(body)) + plugins = out["plugins"] + assert isinstance(plugins, list) + fp = [p for p in plugins if p.get("id") == "file-parser"][0] + # Shape exactly matches https://openrouter.ai/docs/features/multimodal/pdfs + assert set(fp.keys()) == {"id", "pdf"} + assert isinstance(fp["pdf"], dict) + assert fp["pdf"]["engine"] in ("pdf-text", "mistral-ocr", "native") + + +def test_openai_translated_image_block_matches_image_url_data_uri(): + """OpenAI's image_url accepts data: URIs only for image/* mime types + (verified May 2026 — application/pdf returns HTTP 400). The + translator rewrites Anthropic image blocks; document blocks are + refused upstream in agent_manager.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5 + body = json.dumps({ + "model": "gpt-5.5", + "max_tokens": 100, + "messages": [{"role": "user", "content": [ + {"type": "image", "source": { + "type": "base64", + "media_type": "image/png", + "data": "iVBORw0KGgo=", + }}, + ]}], + }).encode("utf-8") + out = json.loads(_scrub_request_for_openai_gpt5(body)) + block = out["messages"][0]["content"][0] + assert block["type"] == "image_url" + assert block["image_url"]["url"] == "data:image/png;base64,iVBORw0KGgo=" + + +def test_openai_proxy_rewrites_image_block_only_documents_pass_through(): + """OpenAI image_url only accepts image/* mime; documents are refused + upstream. Translator handles images, leaves documents untouched.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5 + body = json.dumps({ + "model": "gpt-5.5", + "max_tokens": 500, + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "what's in this image?"}, + {"type": "image", "source": { + "type": "base64", + "media_type": "image/png", + "data": "iVBORw0KGgo=", + }}, + ], + }], + }).encode("utf-8") + out = json.loads(_scrub_request_for_openai_gpt5(body)) + blocks = out["messages"][0]["content"] + assert blocks[0]["type"] == "text" + assert blocks[1]["type"] == "image_url" + assert blocks[1]["image_url"]["url"].startswith("data:image/png;base64,") + assert "max_completion_tokens" in out + assert "max_tokens" not in out + + +def test_openai_proxy_skips_rewrite_when_no_document(): + """Pure text turn on GPT-5 should only get the max_tokens rename.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5 + body = json.dumps({ + "model": "gpt-5.5", + "max_tokens": 100, + "messages": [{"role": "user", "content": "hi"}], + }).encode("utf-8") + out = json.loads(_scrub_request_for_openai_gpt5(body)) + assert out["messages"][0]["content"] == "hi" + assert out.get("max_completion_tokens") == 100 + + +def test_openai_proxy_defensive_on_malformed_document_blocks(): + """Malformed document blocks (missing source, missing data) pass + through untouched so the upstream returns a proper error rather + than us silently dropping the file.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5 + body = json.dumps({ + "model": "gpt-5.5", + "messages": [{ + "role": "user", + "content": [ + {"type": "document"}, + {"type": "document", "source": {"type": "url"}}, + {"type": "document", "source": {"type": "base64"}}, + ], + }], + }).encode("utf-8") + out = json.loads(_scrub_request_for_openai_gpt5(body)) + for b in out["messages"][0]["content"]: + assert b["type"] == "document" + + +def test_resolve_attachments_openai_codex_refused_for_pdfs(): + """Codex variants refuse PDFs (both because Codex models don't read + PDFs AND because the OpenAI direct lane is currently disabled until + 9router translation lands).""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + _t, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.3-codex", + ) + assert not native + assert refusals + finally: + os.unlink(path) + + +def test_resolve_attachments_openai_codex_still_refuses_pdf(): + """Codex variants don't support PDFs even though their OpenAI family + does; refusal should fire with switch hint.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + _t, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.3-codex", + ) + assert not native + assert refusals and "codex" in refusals[0].lower() + finally: + os.unlink(path) + + +def test_openrouter_proxy_injects_file_parser_plugin_when_document_present(): + """OR's universal-PDF feature requires top-level plugins:[{id:file-parser,...}]. + When a document block is in the request bound for OR, inject it.""" + import json + from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser + body = json.dumps({ + "model": "openrouter/qwen/qwen-2.5-72b-instruct", + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "summarize"}, + {"type": "document", "source": { + "type": "base64", + "media_type": "application/pdf", + "data": "JVBERi0xLjQK", + }}, + ], + }], + }).encode("utf-8") + out = json.loads(_inject_openrouter_file_parser(body)) + plugins = out.get("plugins") + assert isinstance(plugins, list) and len(plugins) >= 1 + fp = next((p for p in plugins if p.get("id") == "file-parser"), None) + assert fp and fp["pdf"]["engine"] == "pdf-text" + + +def test_openrouter_proxy_skips_plugin_when_no_document(): + """No document block → don't inject the plugin (costs nothing, but + keeps the request body clean).""" + import json + from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser + body = json.dumps({ + "model": "openrouter/qwen/qwen-2.5-72b-instruct", + "messages": [{"role": "user", "content": "just a question"}], + }).encode("utf-8") + out = json.loads(_inject_openrouter_file_parser(body)) + assert "plugins" not in out + + +def test_openrouter_proxy_dedupes_existing_file_parser_plugin(): + """If a caller already provided file-parser, don't duplicate it.""" + import json + from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser + body = json.dumps({ + "model": "openrouter/qwen/qwen-2.5-72b-instruct", + "plugins": [{"id": "file-parser", "pdf": {"engine": "mistral-ocr"}}], + "messages": [{ + "role": "user", + "content": [ + {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "x"}}, + ], + }], + }).encode("utf-8") + out = json.loads(_inject_openrouter_file_parser(body)) + fps = [p for p in out["plugins"] if p.get("id") == "file-parser"] + assert len(fps) == 1 + assert fps[0]["pdf"]["engine"] == "mistral-ocr" # caller's engine wins + + +def test_gemini_proxy_defensive_on_malformed_blocks(): + """Bad shapes (missing data, wrong source.type, non-string data) + must NOT be rewritten; they pass through so the upstream sees the + error rather than a silently-corrupted block.""" + import json + from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini + body = json.dumps({ + "model": "gemini-3.1-pro-preview", + "messages": [{ + "role": "user", + "content": [ + {"type": "document"}, # no source + {"type": "document", "source": {}}, # empty source + {"type": "document", "source": {"type": "url"}}, # not base64 + {"type": "document", "source": {"type": "base64"}}, # no data + ], + }], + }).encode("utf-8") + out = json.loads(_scrub_request_for_gemini(body)) + for b in out["messages"][0]["content"]: + assert b["type"] == "document" + + +def test_resolve_attachments_anthropic_emits_native_document(): + """Anthropic upstream gets a `document` content block for PDFs, not + a text placeholder.""" + import base64, tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + text, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + assert native and native[0]["type"] == "document" + assert native[0]["source"]["media_type"] == "application/pdf" + assert not refusals + finally: + os.unlink(path) + + +def test_resolve_attachments_openai_refuses_pdf_with_openrouter_hint(): + """Empirical probe May 2026: OpenAI image_url rejects non-image mime + types with HTTP 400 'Invalid MIME type'. The type:file shape gets + stringified by 9router 0.3.60. Until we write a 9router-bypass + direct-API translator, refuse OpenAI PDFs with switch hint to + openrouter/openai/gpt-5 which has working PDF support via OR's + file-parser plugin.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + _text, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.5", + ) + assert not native + assert refusals + joined = " ".join(refusals).lower() + assert "openrouter" in joined or "claude" in joined + finally: + os.unlink(path) + + +def test_resolve_attachments_gemini_emits_native_document_after_translator_fix(): + """After fixing the 9router 0.3.60 block-stripping bug via + anthropic_proxy._rewrite_document_to_image (now rewrites both + image AND document → OpenAI image_url with data: URL, which 9router + translates to Gemini inlineData), PDFs flow on Gemini natively.""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + _text, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="gemini", model="gemini-3.1-pro-api", + ) + assert native and native[0]["type"] == "document" + assert not refusals + finally: + os.unlink(path) + + +def test_resolve_attachments_text_file_inlined_not_native(): + """Text files keep flowing through the existing context_file inline + path (no native block).""" + import tempfile, os + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as fh: + fh.write("# hello\nworld") + path = fh.name + try: + text, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="opus-4-7", model="opus-4-7", + ) + assert not native + assert not refusals + assert "hello" in text + finally: + os.unlink(path) + + +def test_resolve_attachments_pdf_refused_when_too_large(): + """Anthropic's per-file cap blocks PDFs over 24MB.""" + import os, tempfile + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n") + fh.write(b"X" * (25 * 1024 * 1024)) + path = fh.name + try: + _t, native, refusals = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + assert not native + assert refusals + assert "per-file cap" in refusals[0].lower() or "exceeds" in refusals[0].lower() + finally: + os.unlink(path) + + +def test_resolve_attachments_refuses_when_total_exceeds_request_cap(): + """4 medium PDFs that each pass the per-file cap should still be + blocked when their combined base64 size would exceed Anthropic's + 32MB request cap. This is the exact Mehmet scenario (30.3MB raw + of 4 PDFs base64 to ~40MB).""" + import os, tempfile + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + # 4 PDFs at ~8MB each = 32MB raw = ~43MB base64, exceeds 28MB cap. + paths = [] + try: + for i in range(4): + with tempfile.NamedTemporaryFile(suffix=f"_{i}.pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n") + fh.write(b"X" * (8 * 1024 * 1024)) + paths.append(fh.name) + _t, native, refusals = mgr._resolve_attachments( + [{"path": p, "type": "file"} for p in paths], + api_type="anthropic", model="opus-4-7", + ) + # First few PDFs fit; later ones refused with "request over" message. + assert refusals, "expected refusals on multi-PDF over-cap" + assert any("encoded" in r.lower() and "provider cap" in r.lower() for r in refusals), \ + f"expected total-size refusal copy; got: {refusals}" + finally: + for p in paths: + try: os.unlink(p) + except Exception: pass + + +def test_resolve_attachments_anthropic_marks_last_document_ephemeral_for_cache(): + """Anthropic prompt caching: the last document block gets + cache_control:ephemeral so multi-turn PDF chats stay cache-warm.""" + import os, tempfile + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + paths = [] + try: + for i in range(2): + with tempfile.NamedTemporaryFile(suffix=f"_{i}.pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + paths.append(fh.name) + _t, native, _r = mgr._resolve_attachments( + [{"path": p, "type": "file"} for p in paths], + api_type="anthropic", model="opus-4-7", + ) + # Only the LAST document gets cache_control per Anthropic docs. + assert native[-1].get("cache_control") == {"type": "ephemeral"} + assert "cache_control" not in native[0] + finally: + for p in paths: + try: os.unlink(p) + except Exception: pass + + +def test_resolve_attachments_anthropic_does_mark_ephemeral_but_only_anthropic(): + """cache_control is Anthropic-only; don't pollute other-provider + blocks. Anthropic should get ephemeral on the last document block; + OpenRouter (which also supports PDFs) should NOT.""" + import os, tempfile + from backend.apps.agents.agent_manager import AgentManager + mgr = AgentManager() + with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh: + fh.write(b"%PDF-1.4\n%test\n") + path = fh.name + try: + _t, ant_native, _r = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7", + ) + assert ant_native and ant_native[0].get("cache_control") == {"type": "ephemeral"} + _t, or_native, _r = mgr._resolve_attachments( + [{"path": path, "type": "file"}], api_type="openrouter", model="openrouter/openai/gpt-5", + ) + assert or_native and "cache_control" not in or_native[0] + finally: + os.unlink(path) + + +def test_apply_context_window_respects_custom_provider_value(): + """Custom OpenAI-compatible models supply their own context_window + via settings.custom_providers. _apply_context_window must look them + up the same way get_context_window does.""" + from backend.apps.agents.models import AgentSession + from backend.apps.agents.agent_manager import _apply_context_window + from backend.apps.settings.models import AppSettings, CustomProvider + s = AgentSession(id="x", name="t", provider="custom", model="custom/ollama/qwen2.5:7b", mode="agent") + settings = AppSettings(custom_providers=[ + CustomProvider( + name="Ollama", + base_url="http://localhost:11434/v1", + api_key="", + models=[{"value": "qwen2.5:7b", "label": "Qwen 2.5 7B", "context_window": 32_000}], + ), + ]) + _apply_context_window(s, settings) + assert s.context_window == 32_000 + + # --------------------------------------------------------------------------- # Custom OpenAI-compatible providers (Ollama Cloud, Together, etc.) # --------------------------------------------------------------------------- diff --git a/electron/package-lock.json b/electron/package-lock.json index 7ec1e11c..8bfe1d54 100644 --- a/electron/package-lock.json +++ b/electron/package-lock.json @@ -1,12 +1,12 @@ { "name": "openswarm", - "version": "1.1.40-exp.2", + "version": "1.1.41", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "openswarm", - "version": "1.1.40-exp.2", + "version": "1.1.41", "hasInstallScript": true, "dependencies": { "electron-updater": "^6.3.0", diff --git a/frontend/src/app/components/DirectoryBrowser.tsx b/frontend/src/app/components/DirectoryBrowser.tsx index 00294dec..3a45bf48 100644 --- a/frontend/src/app/components/DirectoryBrowser.tsx +++ b/frontend/src/app/components/DirectoryBrowser.tsx @@ -27,6 +27,9 @@ const SETTINGS_API = `${API_BASE}/settings`; export interface ContextPath { path: string; type: 'file' | 'directory'; + tokens?: number; + kind?: 'text' | 'pdf' | 'image' | 'binary'; + media_type?: string; } interface DirectoryBrowserProps { diff --git a/frontend/src/app/pages/AgentChat/AgentChat.tsx b/frontend/src/app/pages/AgentChat/AgentChat.tsx index 2fc2516f..f19c90a9 100644 --- a/frontend/src/app/pages/AgentChat/AgentChat.tsx +++ b/frontend/src/app/pages/AgentChat/AgentChat.tsx @@ -56,8 +56,9 @@ import { setGlowingBrowserCards, fadeGlowingBrowserCards, clearGlowingBrowserCar import { useClaudeTokens } from '@/shared/styles/ThemeContext'; const CONTEXT_WINDOWS: Record = { - sonnet: 200_000, - opus: 200_000, + 'opus-4-7': 1_000_000, + opus: 1_000_000, + sonnet: 1_000_000, haiku: 200_000, }; @@ -619,15 +620,22 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose }, [id, dispatch, onBranch, session?.dashboard_id]); const contextEstimate = useMemo(() => { - // Look up the actual context window from the models store (backend - // registry is the source of truth). Fall back to the legacy hardcoded - // map for any model that isn't in the store yet. + // Prefer the live API-reported input token count once we have one + // (session.tokens.input includes the full request: messages + system + + // tool defs + cached prefix). That number is authoritative because + // Anthropic counts it against the context window. Before the first + // turn completes, fall back to a char/4 estimate of visible message + // content as a rough pre-send hint. let limit = 0; for (const ms of Object.values(modelsByProvider)) { const hit = ms.find((m) => m.value === model); if (hit?.context_window) { limit = hit.context_window; break; } } - if (!limit) limit = CONTEXT_WINDOWS[model] || 200_000; + if (!limit) limit = (session?.context_window) || CONTEXT_WINDOWS[model] || 200_000; + const liveInput = session?.tokens?.input ?? 0; + if (liveInput > 0) { + return { used: liveInput, limit }; + } let totalChars = 0; if (session?.system_prompt) totalChars += session.system_prompt.length; for (const msg of activeBranchMessages) { @@ -640,7 +648,7 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose // text and re-run this sum on every painted character, defeating // the whole point of isolating AgentChat from delta updates. The // header gauge will catch up when stream_end commits the message. - }, [activeBranchMessages, session?.system_prompt, streamingMessageId, model, modelsByProvider]); + }, [activeBranchMessages, session?.system_prompt, session?.tokens?.input, session?.context_window, streamingMessageId, model, modelsByProvider]); const sessionRunning = session?.status === 'running' || session?.status === 'waiting_approval'; @@ -926,16 +934,32 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose ); })()} {(() => { - const pct = session.ctx_used_pct ?? 0; - if (!pct) return null; + const liveWindow = session.context_window || contextEstimate.limit || 200_000; + const liveInput = session.tokens?.input ?? 0; + const pct = liveInput > 0 + ? Math.min(1, liveInput / Math.max(1, liveWindow)) + : (contextEstimate.used / Math.max(1, liveWindow)); + if (pct <= 0) return null; const pctTxt = `${Math.round(pct * 100)}%`; - const color = pct >= 0.9 ? '#ef4444' : pct >= 0.7 ? '#f59e0b' : c.text.tertiary; + const color = pct >= 0.85 ? '#ef4444' : pct >= 0.60 ? '#f59e0b' : c.text.tertiary; const mcpCount = session.active_mcps?.length ?? 0; + const fwOverhead = session.framework_overhead_tokens ?? 0; + const systemTokens = Math.round((session.system_prompt?.length ?? 0) / 4); + const historyTokens = Math.max(0, contextEstimate.used - systemTokens); + const fmt = (n: number) => n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n); + const breakdown = [ + `Context ${pctTxt} of ${fmt(liveWindow)}`, + liveInput > 0 ? `Reported by API: ${fmt(liveInput)} input tokens` : null, + `History: ~${fmt(historyTokens)}`, + `System prompt: ~${fmt(systemTokens)}`, + fwOverhead > 0 ? `Tools + MCPs + preset: ~${fmt(fwOverhead)}` : null, + `${mcpCount} MCP${mcpCount === 1 ? '' : 's'} active`, + ].filter(Boolean).join(' · '); return ( {pctTxt} ctx · {mcpCount} mcp diff --git a/frontend/src/app/pages/AgentChat/ChatInput.tsx b/frontend/src/app/pages/AgentChat/ChatInput.tsx index f49c03c0..cd81bce1 100644 --- a/frontend/src/app/pages/AgentChat/ChatInput.tsx +++ b/frontend/src/app/pages/AgentChat/ChatInput.tsx @@ -160,6 +160,20 @@ function formatTokenCount(n: number): string { return String(n); } +// Path basename that works on both POSIX (/Users/x/file.pdf) and Windows +// (C:\Users\x\file.pdf). Splits on either separator; falls back to the +// raw path so empty segments don't yield ''. +function basename(p: string): string { + if (!p) return ''; + const parts = p.split(/[\\/]/).filter(Boolean); + return parts[parts.length - 1] || p; +} +function pathTail(p: string, n: number): string { + if (!p) return ''; + const parts = p.split(/[\\/]/).filter(Boolean); + return parts.slice(-n).join('/'); +} + const ContextRing: React.FC<{ used: number; limit: number; accentColor: string; trackColor: string }> = ({ used, limit, accentColor, trackColor }) => { if (used === 0) return null; const pct = Math.min((used / limit) * 100, 100); @@ -337,6 +351,9 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, const modelsLoaded = useAppSelector((state) => state.models.loaded); const connectionMode = useAppSelector((state) => state.settings.data.connection_mode); const toolItems = useAppSelector((state) => state.tools.items); + const sessionFrameworkOverhead = useAppSelector((state) => + sessionId ? (state.agents.sessions[sessionId]?.framework_overhead_tokens ?? 0) : 0, + ); const allModelOptions = useMemo(() => { @@ -644,6 +661,19 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, const [contextPaths, setContextPaths] = useState([]); const [forcedTools, setForcedTools] = useState([]); const [copiedPathIdx, setCopiedPathIdx] = useState(null); + const [oversizeQueue, setOversizeQueue] = useState>([]); + const [summarizingPath, setSummarizingPath] = useState(null); + const [summarizeError, setSummarizeError] = useState(null); + const [sendBlock, setSendBlock] = useState(null); useImperativeHandle(ref, () => ({ getConfig: () => { @@ -756,6 +786,43 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, }); }, []); + const currentModelCtx = useMemo(() => { + const m = allModelOptions.flat.find((x: any) => x.value === model) as any; + return (m?.context_window as number) || 200_000; + }, [allModelOptions.flat, model]); + + const currentModelApi = useMemo(() => { + const m = allModelOptions.flat.find((x: any) => x.value === model) as any; + return ((m?.api as string) || 'anthropic').toLowerCase(); + }, [allModelOptions.flat, model]); + + // Mirrors backend agent_manager._resolve_attachments support matrix: + // PDFs route natively on Anthropic + Gemini (via anthropic-proxy + // document→image rewrite); refused on OpenAI/OpenRouter/custom until + // we land file-parser plugin / type:file translation. + // Mirrors backend agent_manager._resolve_attachments support matrix. + // PDFs: Anthropic, Gemini, OpenRouter (all empirically verified May + // 2026 via probe-pdf-roundtrip.py). OpenAI direct refused because + // OpenAI's image_url rejects non-image mime; user should switch to + // openrouter/openai/gpt-5 for OpenAI-via-OR with PDF support. + // Images: every provider via 9router image_url translation. + const pdfSupported = ['anthropic', 'gemini', 'gemini-cli', 'openrouter'].includes(currentModelApi); + const imageSupported = ['anthropic', 'gemini', 'gemini-cli', 'openai', 'openrouter'].includes(currentModelApi); + + const pendingPayloadEstimate = useMemo(() => { + const history = Math.max(0, contextEstimate?.used ?? 0); + const filesSum = contextPaths.reduce((acc, cp) => acc + (cp.tokens || 0), 0); + return history + (sessionFrameworkOverhead || 0) + filesSum; + }, [contextEstimate, contextPaths, sessionFrameworkOverhead]); + + const pendingKinds = useMemo(() => { + const set = new Set(); + for (const cp of contextPaths) { + if (cp.kind) set.add(cp.kind); + } + return set; + }, [contextPaths]); + const uploadAndAttachFiles = useCallback(async (files: File[]) => { if (files.length === 0) return; setIsUploading(true); @@ -768,25 +835,61 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, }); if (!resp.ok) throw new Error('Upload failed'); const data = await resp.json(); - const newPaths: ContextPath[] = (data.files || []).map((f: { path: string }) => ({ - path: f.path, - type: 'file' as const, - })); + const halfCap = Math.floor(currentModelCtx * 0.5); + const oversize: Array<{ path: string; name: string; tokens: number }> = []; + const newPaths: ContextPath[] = (data.files || []).map((f: { path: string; name?: string; tokens?: number; kind?: 'text' | 'pdf' | 'image' | 'binary'; media_type?: string }) => { + const t = typeof f.tokens === 'number' ? f.tokens : 0; + if (t > halfCap) oversize.push({ path: f.path, name: f.name || basename(f.path) || 'file', tokens: t }); + return { path: f.path, type: 'file' as const, tokens: t, kind: f.kind, media_type: f.media_type }; + }); setContextPaths((prev) => [...prev, ...newPaths]); + if (oversize.length > 0) setOversizeQueue((q) => [...q, ...oversize]); } catch (err) { console.error('File upload failed:', err); } finally { setIsUploading(false); } - }, []); + }, [currentModelCtx]); const handleSend = useCallback(async () => { const editor = editorRef.current; if (!editor || disabled) return; + if (summarizingPath) return; + if (oversizeQueue.length > 0) return; const serialized = serializeEditorContent(editor, attachedSkillsRef.current); let trimmed = serialized.trim(); if (!trimmed) return; + // Pre-send dry-run guard. Sums every known component of next-turn input + // (history estimate from props, system prompt, framework/MCP overhead + // last reported by the API, attached file token estimates, and the + // prompt itself). If the sum exceeds 95% of the model's window, block + // the send and surface a banner with concrete recovery actions instead + // of round-tripping to a doomed API call. Conservative on purpose: + // tokenizers differ across providers (char/4 is rough), so we leave + // 5% headroom plus the API's own response budget. + { + const win = currentModelCtx; + const history = Math.max(0, contextEstimate?.used ?? 0); + const filesSum = contextPaths.reduce((acc, cp) => acc + (cp.tokens || 0), 0); + const promptTokens = Math.ceil(trimmed.length / 4); + const framework = sessionFrameworkOverhead || 0; + const systemTokens = 0; + const estimate = history + framework + filesSum + promptTokens + systemTokens; + if (win > 0 && estimate > Math.floor(win * 0.95)) { + let largest: { path: string; tokens: number } | undefined; + for (const cp of contextPaths) { + if ((cp.tokens || 0) > (largest?.tokens || 0)) largest = { path: cp.path, tokens: cp.tokens || 0 }; + } + setSendBlock({ + estimate, window: win, + history, system: systemTokens, framework, files: filesSum, prompt: promptTokens, + largestFile: largest, + }); + return; + } + } + onboardingBus.emit('chat:message_sent'); if (window.location.hash.includes('/apps/')) { onboardingBus.emit('app:generation_started'); @@ -1116,6 +1219,57 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, if (otherFiles.length > 0) uploadAndAttachFiles(otherFiles); }, [addImageFiles, uploadAndAttachFiles]); + useEffect(() => { + const halfCap = Math.floor(currentModelCtx * 0.5); + const stillOversize: Array<{ path: string; name: string; tokens: number }> = []; + for (const cp of contextPaths) { + const t = cp.tokens || 0; + if (t > halfCap) { + const name = basename(cp.path) || cp.path; + stillOversize.push({ path: cp.path, name, tokens: t }); + } + } + setOversizeQueue((q) => { + const next = stillOversize.filter((o) => !q.find((qq) => qq.path === o.path)); + return [...q.filter((qq) => stillOversize.find((o) => o.path === qq.path)), ...next]; + }); + }, [currentModelCtx, contextPaths]); + + const detachOversize = useCallback((path: string) => { + setContextPaths((prev) => prev.filter((cp) => cp.path !== path)); + setOversizeQueue((q) => q.filter((o) => o.path !== path)); + }, []); + + const summarizeOversize = useCallback(async (path: string) => { + if (summarizingPath) return; // another summarize is in flight; ignore + setSummarizingPath(path); + try { + const tok = (() => { try { return getAuthToken(); } catch { return ''; } })(); + const headers: Record = { 'Content-Type': 'application/json' }; + if (tok) headers['Authorization'] = `Bearer ${tok}`; + const target = Math.min(8_000, Math.max(1_000, Math.floor(currentModelCtx * 0.05))); + const resp = await fetch(`${API_BASE}/settings/summarize-file`, { + method: 'POST', headers, + body: JSON.stringify({ path, target_tokens: target, primary_model: model }), + }); + if (!resp.ok) { + let detail = `summarize failed (${resp.status})`; + try { const j = await resp.json(); if (j?.detail) detail = String(j.detail); } catch {} + throw new Error(detail); + } + const data = await resp.json(); + const newPath: string = data.path; + const newTokens: number = data.tokens || 0; + setContextPaths((prev) => prev.map((cp) => cp.path === path ? { ...cp, path: newPath, tokens: newTokens, kind: 'text', media_type: 'text/plain' } : cp)); + setOversizeQueue((q) => q.filter((o) => o.path !== path)); + } catch (err) { + const msg = err instanceof Error ? err.message : 'summarize failed'; + setSummarizeError(`${msg}. Detach the file or connect an aux provider in Settings.`); + } finally { + setSummarizingPath(null); + } + }, [currentModelCtx, model, summarizingPath]); + const removeImage = useCallback((idx: number) => { setImages((prev) => { const removed = prev[idx]; @@ -1219,6 +1373,82 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, visible={picker.visible} /> + {sendBlock && (() => { + const fmt = (n: number) => n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n); + const over = sendBlock.estimate - sendBlock.window; + return ( + + + This send would overflow the model's context window + + + ~{fmt(sendBlock.estimate)} of {fmt(sendBlock.window)} tokens ({over > 0 ? `${fmt(over)} over` : 'at cap'}). History {fmt(sendBlock.history)} · Files {fmt(sendBlock.files)} · Tools/MCPs {fmt(sendBlock.framework)} · This message {fmt(sendBlock.prompt)}. + + + {sessionId && ( + { + try { + const tok = (() => { try { return getAuthToken(); } catch { return ''; } })(); + const headers: Record = { 'Content-Type': 'application/json' }; + if (tok) headers['Authorization'] = `Bearer ${tok}`; + await fetch(`${API_BASE}/agents/sessions/${sessionId}/compact`, { method: 'POST', headers }); + setSendBlock(null); + } catch (err) { console.error(err); } + }} + sx={{ + background: c.accent.primary, color: '#fff', border: 'none', borderRadius: '6px', + px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', '&:hover': { opacity: 0.9 }, + }} + > + Compact memory + + )} + {sendBlock.largestFile && ( + { + const p = sendBlock.largestFile!.path; + setContextPaths((prev) => prev.filter((cp) => cp.path !== p)); + setSendBlock(null); + }} + sx={{ + background: 'transparent', color: c.text.primary, border: `1px solid ${c.border.subtle}`, + borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', + '&:hover': { background: c.bg.secondary }, + }} + > + Detach largest file (~{fmt(sendBlock.largestFile.tokens)}) + + )} + { setModelAnchor(e.currentTarget as HTMLElement); setSendBlock(null); }} + sx={{ + background: 'transparent', color: c.text.primary, border: `1px solid ${c.border.subtle}`, + borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', + '&:hover': { background: c.bg.secondary }, + }} + > + Switch model + + setSendBlock(null)} + sx={{ + background: 'transparent', color: c.text.muted, border: 'none', + borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', + '&:hover': { background: c.bg.secondary }, + }} + > + Dismiss + + + + ); + })()} + {images.length > 0 && ( (({ onSend, disabled, mode, const isAppWorkspace = /\/outputs_workspace\/ws-[^/]+\/?$/.test(cp.path); const label = isAppWorkspace ? 'App files' - : cp.path.split('/').filter(Boolean).slice(-2).join('/'); + : pathTail(cp.path, 2); return ( (({ onSend, disabled, mode, }, }} > + {(() => { + const unsupported = (cp.kind === 'pdf' && !pdfSupported) || + (cp.kind === 'image' && !imageSupported) || + cp.kind === 'binary'; + const chipColor = unsupported ? c.status.warning : c.accent.primary; + return ( : } - label={label} + label={(() => { + const kindTag = cp.kind && cp.kind !== 'text' ? ` · ${cp.kind}` : ''; + const tokTag = typeof cp.tokens === 'number' && cp.tokens > 0 ? ` · ${formatTokenCount(cp.tokens)}` : ''; + const warn = unsupported ? ' · not on this model' : ''; + return `${label}${kindTag}${tokTag}${warn}`; + })()} size="small" onClick={() => { navigator.clipboard.writeText(cp.path); @@ -1315,21 +1556,23 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, }} onDelete={() => setContextPaths((prev) => prev.filter((_, i) => i !== idx))} sx={{ - bgcolor: `${c.accent.primary}12`, - color: c.accent.primary, + bgcolor: `${chipColor}12`, + color: chipColor, fontSize: '0.72rem', fontFamily: c.font.mono, height: 26, - maxWidth: 220, + maxWidth: 280, cursor: 'pointer', '& .MuiChip-label': { overflow: 'hidden', textOverflow: 'ellipsis' }, '& .MuiChip-deleteIcon': { - color: c.accent.primary, + color: chipColor, fontSize: 16, '&:hover': { color: c.status.error }, }, }} /> + ); + })()} ); })} @@ -2009,6 +2252,45 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, primary={highlightMatch(displayLabel)} slotProps={{ primary: { sx: { fontSize: '0.8rem', color: model === opt.value ? c.text.primary : c.text.muted } } }} /> + {(() => { + const win = (opt.context_window as number) || 0; + const api = (opt.api as string || 'anthropic').toLowerCase(); + const optSupportsPdf = ['anthropic', 'gemini', 'gemini-cli', 'openrouter'].includes(api); + const optSupportsImage = ['anthropic', 'gemini', 'gemini-cli', 'openai', 'openrouter'].includes(api); + const cannotPdf = pendingKinds.has('pdf') && !optSupportsPdf; + const cannotImg = pendingKinds.has('image') && !optSupportsImage; + if (!win) return null; + const fits = pendingPayloadEstimate > 0 && win >= Math.floor(pendingPayloadEstimate * 1.1); + const tight = pendingPayloadEstimate > 0 && !fits && win >= pendingPayloadEstimate; + const tooSmall = pendingPayloadEstimate > 0 && win < pendingPayloadEstimate; + return ( + + {(cannotPdf || cannotImg) && ( + + No {cannotPdf ? 'PDF' : 'image'} + + )} + {!cannotPdf && !cannotImg && fits && ( + + Fits + + )} + {!cannotPdf && !cannotImg && tight && ( + + Tight + + )} + {!cannotPdf && !cannotImg && tooSmall && ( + + Too small + + )} + + {formatTokenCount(win)} + + + ); + })()} ); @@ -2359,6 +2641,65 @@ const ChatInput = forwardRef(({ onSend, disabled, mode, + 0} + anchorOrigin={{ vertical: 'bottom', horizontal: 'center' }} + sx={{ mb: 10 }} + > + + oversizeQueue[0] && summarizeOversize(oversizeQueue[0].path)} + sx={{ + background: 'rgba(255,255,255,0.18)', color: 'inherit', border: 'none', + borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', + '&:hover': { background: 'rgba(255,255,255,0.28)' }, + '&:disabled': { opacity: 0.6, cursor: 'wait' }, + }} + > + {summarizingPath === oversizeQueue[0]?.path ? 'Summarizing…' : 'Summarize instead'} + + oversizeQueue[0] && detachOversize(oversizeQueue[0].path)} + sx={{ + background: 'transparent', color: 'inherit', border: '1px solid rgba(255,255,255,0.4)', + borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', + '&:hover': { background: 'rgba(255,255,255,0.12)' }, + }} + > + Detach + + + } + > + {oversizeQueue[0] ? ( + + {oversizeQueue[0].name} is ~{formatTokenCount(oversizeQueue[0].tokens)} tokens, over 50% of this model's window ({formatTokenCount(currentModelCtx)}). Summarize sends the file content to your configured aux provider. + + ) : null} + + + + setSummarizeError(null)} + sx={{ mb: 18 }} + > + setSummarizeError(null)} sx={{ fontSize: '0.78rem', maxWidth: 520 }}> + {summarizeError} + + + ); }); diff --git a/frontend/src/app/pages/AgentChat/MessageBubble.tsx b/frontend/src/app/pages/AgentChat/MessageBubble.tsx index 52bec3bd..94d27575 100644 --- a/frontend/src/app/pages/AgentChat/MessageBubble.tsx +++ b/frontend/src/app/pages/AgentChat/MessageBubble.tsx @@ -20,7 +20,7 @@ import ReactMarkdown from 'react-markdown'; import remarkGfm from 'remark-gfm'; import { AgentMessage } from '@/shared/state/agentsSlice'; import { openSettingsModal } from '@/shared/state/settingsSlice'; -import { useAppDispatch } from '@/shared/hooks'; +import { useAppDispatch, useAppSelector } from '@/shared/hooks'; import { useClaudeTokens } from '@/shared/styles/ThemeContext'; import { SKILL_COLOR } from '@/app/components/richEditorUtils'; import PlanPicker from '@/app/components/PlanPicker'; @@ -70,8 +70,23 @@ interface OpenSwarmErrorInfo { ctaAction?: 'upgrade' | 'retry' | 'settings' | 'waitlist'; } +interface OverflowContext { + model?: string; + contextWindow?: number; + inputTokens?: number; + frameworkOverhead?: number; + activeMcpCount?: number; + messagesCount?: number; +} + +function formatTokens(n: number): string { + if (n >= 1_000_000) return `${(n / 1_000_000).toFixed(1)}M`; + if (n >= 1_000) return `${(n / 1_000).toFixed(1)}K`; + return String(n); +} + /** Parses raw error text into a friendly card; returns null when the error isn't one we recognize. */ -function parseOpenSwarmError(text: string): OpenSwarmErrorInfo | null { +function parseOpenSwarmError(text: string, ctx?: OverflowContext): OpenSwarmErrorInfo | null { if (!text) return null; if (/rate_limit_error|reached your OpenSwarm.*plan limit|Usage cap exceeded/i.test(text)) { const reset = text.match(/Resets in ([\dhms\s]+)/)?.[1]; @@ -93,15 +108,48 @@ function parseOpenSwarmError(text: string): OpenSwarmErrorInfo | null { }; } if (/Prompt is too long|prompt_too_long|input length and `max_tokens`|context length/i.test(text)) { + const modelLower = (ctx?.model || '').toLowerCase(); + const isHaiku = modelLower.includes('haiku'); + const win = ctx?.contextWindow || 0; + const input = ctx?.inputTokens || 0; + const fw = ctx?.frameworkOverhead || 0; + const mcps = ctx?.activeMcpCount || 0; + if (isHaiku && mcps >= 5) { + return { + kind: 'too_many_tools', + title: 'Too many connected apps for Haiku', + detail: + `Haiku has the smallest memory of the Claude models${win ? ` (${formatTokens(win)} tokens)` : ''}. ` + + `Each of the ${mcps} active apps adds instructions Claude has to read before it can answer. ` + + 'Turn off a few apps (Microsoft 365 is the heaviest), or switch to Sonnet or Opus, both have 5x more room.', + ctaLabel: 'Open Settings', + ctaAction: 'settings', + }; + } + let lead: string; + if (win && input) { + // input is the API-reported total which includes our preset, tool + // defs, MCP descriptions etc. Subtract those for the user-facing + // "your content" number so we don't blame the user for our overhead. + const userContent = Math.max(0, input - fw); + lead = `The request totalled ~${formatTokens(input)} of ${formatTokens(win)} tokens this model can hold (your messages + files: ~${formatTokens(userContent)}).`; + } else if (win) { + lead = `This model holds ${formatTokens(win)} tokens and the request exceeded that.`; + } else { + lead = 'The request exceeded this model\'s context window.'; + } + const extras: string[] = []; + if (fw) extras.push(`built-in tools + system prompt ~${formatTokens(fw)}`); + if (mcps > 0) extras.push(`${mcps} active app${mcps === 1 ? '' : 's'}`); + const breakdown = extras.length > 0 ? ` Overhead from OpenSwarm: ${extras.join(', ')}.` : ''; return { kind: 'too_many_tools', - title: 'Too many connected apps for this model', - detail: - "Haiku is fast but has the smallest memory of the three Claude models. " + - "Each connected app adds instructions Claude has to read before it can answer, " + - "and you've added more than Haiku can hold in one go. Either turn off a few apps " + - "(Microsoft 365 is the heaviest by far), or switch to Sonnet or Opus; both have " + - "5x more room.", + title: 'This chat exceeded the model\'s context window', + detail: ( + lead + breakdown + + ' Try detaching large files, running /compact to summarize older turns, ' + + 'starting a fresh chat, or switching to a model with a larger window.' + ), ctaLabel: 'Open Settings', ctaAction: 'settings', }; @@ -248,7 +296,7 @@ function buildContextGroups( color: '#10b981', label, chips: allPaths.map((cp) => { - const name = cp.path.split('/').filter(Boolean).pop() || cp.path; + const name = cp.path.split(/[\\/]/).filter(Boolean).pop() || cp.path; return { label: name, tooltip: cp.path, @@ -848,7 +896,21 @@ const MessageBubble: React.FC = React.memo(({ message, editing = false, o >{rawText} ), [rawText]); - const openswarmError = !isUser ? parseOpenSwarmError(rawText) : null; + const overflowCtx = useAppSelector((state) => { + const sid = state.agents.activeSessionId; + if (!sid) return undefined; + const s = state.agents.sessions[sid]; + if (!s) return undefined; + return { + model: s.model, + contextWindow: s.context_window, + inputTokens: s.tokens?.input, + frameworkOverhead: s.framework_overhead_tokens, + activeMcpCount: s.active_mcps?.length ?? 0, + messagesCount: s.messages?.length ?? 0, + } as OverflowContext; + }); + const openswarmError = !isUser ? parseOpenSwarmError(rawText, overflowCtx) : null; // (message.id, kind) keys so cap card analytics fire once, not on edits. React.useEffect(() => { diff --git a/frontend/src/shared/state/agentsSlice.ts b/frontend/src/shared/state/agentsSlice.ts index 187a1625..9c5aa6c8 100644 --- a/frontend/src/shared/state/agentsSlice.ts +++ b/frontend/src/shared/state/agentsSlice.ts @@ -87,6 +87,8 @@ export interface AgentSession { ctx_used_pct?: number; cache_read_pct?: number; cache_read_tokens?: number; + context_window?: number; + framework_overhead_tokens?: number; context_overflow?: { reason: string; message: string; at: string } | null; mcp_suggestions?: Array<{ id: string; title: string; description: string; reason?: string }>; mcp_suggestions_is_vague?: boolean; @@ -804,6 +806,8 @@ const agentsSlice = createSlice({ cacheReadTokens: number; cacheReadPct: number; ctxUsedPct: number; + contextWindow?: number; + frameworkOverheadTokens?: number; activeMcps: string[]; }> ) { @@ -817,6 +821,12 @@ const agentsSlice = createSlice({ session.cache_read_tokens = action.payload.cacheReadTokens; session.cache_read_pct = action.payload.cacheReadPct; session.ctx_used_pct = action.payload.ctxUsedPct; + if (typeof action.payload.contextWindow === 'number' && action.payload.contextWindow > 0) { + session.context_window = action.payload.contextWindow; + } + if (typeof action.payload.frameworkOverheadTokens === 'number') { + session.framework_overhead_tokens = action.payload.frameworkOverheadTokens; + } session.active_mcps = action.payload.activeMcps; } }, diff --git a/frontend/src/shared/ws/WebSocketManager.ts b/frontend/src/shared/ws/WebSocketManager.ts index d0c29796..315e8e03 100644 --- a/frontend/src/shared/ws/WebSocketManager.ts +++ b/frontend/src/shared/ws/WebSocketManager.ts @@ -578,6 +578,8 @@ class WebSocketManager { cacheReadTokens: data.cache_read_tokens ?? 0, cacheReadPct: data.cache_read_pct ?? 0, ctxUsedPct: data.ctx_used_pct ?? 0, + contextWindow: typeof data.context_window === 'number' ? data.context_window : undefined, + frameworkOverheadTokens: typeof data.framework_overhead_tokens === 'number' ? data.framework_overhead_tokens : undefined, activeMcps: Array.isArray(data.active_mcps) ? data.active_mcps : [], })); } diff --git a/scripts/probe-pdf-roundtrip.py b/scripts/probe-pdf-roundtrip.py new file mode 100755 index 00000000..9119bce4 --- /dev/null +++ b/scripts/probe-pdf-roundtrip.py @@ -0,0 +1,130 @@ +#!/usr/bin/env python3 +"""Empirical end-to-end PDF roundtrip probe. + +Verifies that an Anthropic-shape `document` content block flows through +anthropic_proxy → 9router → upstream provider → response. Use one of +the model presets below; the script picks the right backend lane. + +Usage: + 1. Configure the matching API key in OpenSwarm Settings (or env). + 2. Start the dev stack: bash run.sh + 3. python3 scripts/probe-pdf-roundtrip.py + +Providers: + anthropic direct Anthropic API key (claude-opus-4-7) + gemini direct Google AI Studio key (gemini-3-pro-preview) + openai direct OpenAI key (gpt-5.5) + openrouter OpenRouter free model with file-parser plugin (openrouter/openai/gpt-5) + +A 2xx response with non-empty content blocks confirms the full +translator chain works for that provider. A 4xx with a provider-specific +error message tells you exactly which step broke. +""" +import base64 +import json +import os +import sys +import urllib.request +import urllib.error + +# Each model_id below targets a SPECIFIC routing lane so a probe failure +# pinpoints exactly which auth path is broken. "anthropic" uses the SDK +# model_id form that lands on the cc/ OAuth subscription lane; if you +# only have an Anthropic API key (no Pro subscription), use the explicit +# direct-API model_id by editing PRESETS or pass --model. +PRESETS = { + "anthropic": "claude-opus-4-7", # cc/ lane via 9router; needs Pro OAuth + "anthropic-api": "claude-opus-4-7", # same wire shape; routes by settings + "gemini": "gemini-3-pro-preview", + "openai": "gpt-5.5", + "openrouter": "openrouter/openai/gpt-5", +} +PROVIDER_CAPS_MB = { + "anthropic": 28, + "gemini": 14, + "openai": 45, + "openrouter": 45, +} + + +def probe(provider: str, pdf_path: str) -> int: + if provider not in PRESETS: + print(f"FAIL: unknown provider '{provider}'. Use one of: {', '.join(PRESETS)}", file=sys.stderr) + return 2 + if not os.path.isfile(pdf_path): + print(f"FAIL: file not found: {pdf_path}", file=sys.stderr) + return 2 + size = os.path.getsize(pdf_path) + cap = PROVIDER_CAPS_MB[provider] * 1024 * 1024 + if size > cap: + print(f"FAIL: PDF is {size // (1024*1024)} MB, over {provider}'s {PROVIDER_CAPS_MB[provider]} MB inline cap.", file=sys.stderr) + return 2 + + with open(pdf_path, "rb") as fh: + data_b64 = base64.b64encode(fh.read()).decode("ascii") + + token_path = os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "backend", "data", "auth.token", + ) + if not os.path.exists(token_path): + print(f"FAIL: auth.token not found at {token_path}. Start the dev backend first (bash run.sh).", file=sys.stderr) + return 2 + with open(token_path) as fh: + token = fh.read().strip() + + body = { + "model": PRESETS[provider], + "max_tokens": 300, + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "In one sentence, what is this PDF about?"}, + {"type": "document", "source": { + "type": "base64", + "media_type": "application/pdf", + "data": data_b64, + }}, + ], + }], + } + + req = urllib.request.Request( + "http://127.0.0.1:8324/api/anthropic-proxy/v1/messages", + data=json.dumps(body).encode("utf-8"), + headers={ + "Content-Type": "application/json", + "x-api-key": token, + "anthropic-version": "2023-06-01", + }, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=120) as resp: + status = resp.status + payload = resp.read(4096).decode("utf-8", errors="replace") + except urllib.error.HTTPError as e: + status = e.code + payload = e.read(4096).decode("utf-8", errors="replace") + except Exception as e: + print(f"FAIL: network error: {e}", file=sys.stderr) + return 3 + + print(f"=== {provider} ({PRESETS[provider]}) ===") + print(f"HTTP {status}") + print(payload[:800]) + if 200 <= status < 300: + if any(k in payload.lower() for k in ("content", "text", "completion", "choice")): + print(f"\n✓ {provider} accepted the PDF document block end-to-end.") + return 0 + print(f"\n⚠ HTTP 200 but no recognizable content; inspect response above.") + return 1 + print(f"\n✗ {provider} PDF roundtrip failed at HTTP {status}. See response above.") + return 1 + + +if __name__ == "__main__": + if len(sys.argv) != 3: + print(f"usage: probe-pdf-roundtrip.py <{'|'.join(PRESETS)}> path/to/test.pdf", file=sys.stderr) + sys.exit(2) + sys.exit(probe(sys.argv[1], sys.argv[2]))