diff --git a/backend/apps/agents/agent_manager.py b/backend/apps/agents/agent_manager.py
index cf3c3541..9d6a9555 100644
--- a/backend/apps/agents/agent_manager.py
+++ b/backend/apps/agents/agent_manager.py
@@ -54,6 +54,33 @@ def _safe_resp_text(resp) -> str:
return ""
+def _apply_context_window(session, settings=None) -> None:
+ """Set session.context_window from the registry for its (provider, model).
+
+ Called at every AgentSession creation, restore, and model-switch site so
+ the soft-cap trim, auto-compaction, and the UI percent meter line up
+ with the model's real cap (Opus/Sonnet 1M, Haiku 200k, custom values
+ declared per-provider). Silent fallback to the existing value keeps a
+ bad lookup from ever breaking a session.
+ """
+ try:
+ from backend.apps.agents.providers.registry import get_context_window
+ if settings is None:
+ try:
+ settings = load_settings()
+ except Exception:
+ settings = None
+ cw = get_context_window(
+ getattr(session, "provider", "") or "",
+ getattr(session, "model", "") or "",
+ settings,
+ )
+ if isinstance(cw, int) and cw > 0:
+ session.context_window = cw
+ except Exception:
+ logger.debug("context_window lookup failed; keeping existing value", exc_info=True)
+
+
def _save_session(session_id: str, doc_data: dict):
os.makedirs(SESSIONS_DIR, exist_ok=True)
with open(os.path.join(SESSIONS_DIR, f"{session_id}.json"), "w") as f:
@@ -788,6 +815,7 @@ class AgentManager:
dashboard_id=config.dashboard_id,
thinking_level=getattr(global_settings, "default_thinking_level", "auto"),
)
+ _apply_context_window(session, global_settings)
self.sessions[session_id] = session
from backend.apps.service.service import APP_VERSION
@@ -800,35 +828,6 @@ class AgentManager:
return session
- def _resolve_context_paths(self, context_paths: list | None) -> str:
- """Read file contents / directory trees for attached context paths."""
- if not context_paths:
- return ""
- sections = []
- for cp in context_paths:
- path = cp.get("path", "")
- cp_type = cp.get("type", "file")
- if not path or not os.path.exists(path):
- sections.append(f"[Context: {path} — not found]")
- continue
- if cp_type == "file" and os.path.isfile(path):
- try:
- with open(path, "r", errors="replace") as f:
- content = f.read(512_000) # ~500KB cap per file
- sections.append(
- f"\n{content}\n"
- )
- except Exception as e:
- sections.append(f"[Context: {path} — error reading: {e}]")
- elif cp_type == "directory" and os.path.isdir(path):
- tree_lines = self._build_dir_tree(path, max_depth=4)
- sections.append(
- f"\n{chr(10).join(tree_lines)}\n"
- )
- else:
- sections.append(f"[Context: {path} — type mismatch]")
- return "\n\n".join(sections)
-
def _build_dir_tree(self, root: str, max_depth: int = 4, prefix: str = "") -> list[str]:
"""Build a recursive directory tree listing."""
lines = []
@@ -1098,19 +1097,41 @@ class AgentManager:
)
return replacement, blob_path
- def _build_prompt_content(self, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None):
- """Build message content with optional image blocks, context, and forced tools for the Claude API."""
- context_text = self._resolve_context_paths(context_paths)
+ def _build_prompt_content(self, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None, api_type: str = "anthropic", model: str = ""):
+ """Build message content for the Anthropic SDK's prompt stream.
+
+ Routes attachments per provider:
+ - Anthropic: native `image` + `document` blocks for the active
+ Claude model. Text files inline as . Binary that
+ isn't PDF/image gets a refusal placeholder.
+ - Gemini (api=gemini): we still talk to the SDK with Anthropic
+ content-block shapes; the 9router translation layer (cc/gc/gpt
+ lanes) converts to the provider's native shape. For Gemini's
+ native multimodal we emit image/document blocks the same way
+ and rely on 9router to rewrite to inline_data. Over 20MB
+ payloads get refused at this layer (Gemini's inline cap).
+ - OpenAI / Codex: image blocks pass through (image_url at the
+ wire); PDFs handled as documents on multimodal models; non-
+ multimodal models refuse.
+ - OpenRouter, custom OpenAI-compatible: text fallback for
+ anything binary, since native shape varies wildly. Caller can
+ opt-in to the OR file-parser via a separate plugins config.
+ """
+ context_text, native_blocks, refusals = self._resolve_attachments(
+ context_paths, api_type=api_type, model=model,
+ )
forced_tools_text = self._resolve_forced_tools(forced_tools)
skills_text = self._resolve_attached_skills(attached_skills)
- parts = [p for p in (forced_tools_text, context_text, skills_text, prompt) if p]
+ refusal_text = "\n\n".join(refusals)
+ parts = [p for p in (forced_tools_text, context_text, refusal_text, skills_text, prompt) if p]
full_prompt = "\n\n".join(parts)
- if not images:
+ has_native = bool(native_blocks)
+ if not images and not has_native:
return full_prompt
- content = [{"type": "text", "text": full_prompt}]
- for img in images:
+ content: list[dict] = [{"type": "text", "text": full_prompt}]
+ for img in (images or []):
content.append({
"type": "image",
"source": {
@@ -1119,15 +1140,236 @@ class AgentManager:
"data": img["data"],
},
})
+ content.extend(native_blocks)
return content
+ def _resolve_attachments(self, context_paths: list | None, api_type: str, model: str) -> tuple[str, list[dict], list[str]]:
+ """Split context_paths into:
+ - inline text (returned as the existing block string)
+ - native content blocks for this provider (PDFs/images)
+ - refusal strings that get appended to the prompt as plain text
+
+ Reuses the upload-time sniff (PDF magic / null-byte heuristic) so
+ a renamed `.pdf` actually classifies right, and a `.txt` with
+ binary garbage doesn't sneak through as text.
+
+ Two layers of size guard:
+ 1) Per-file inline cap based on provider's raw size limit.
+ 2) Total base64-expanded size cap across all native attachments,
+ because providers cap the WHOLE request body (Anthropic 32MB,
+ Gemini 20MB, OpenAI 50MB). 4 medium PDFs that pass the
+ per-file check can still collectively blow the request cap.
+ The last document block gets cache_control:ephemeral so a follow-up
+ turn on the same PDF reuses the cache prefix (Anthropic only).
+ """
+ if not context_paths:
+ return "", [], []
+ from backend.apps.settings.settings import _sniff_file_kind
+ import base64 as _b64
+ sections: list[str] = []
+ native: list[dict] = []
+ refusals: list[str] = []
+
+ # The Claude Agent SDK speaks only Anthropic content-block shape.
+ # 9router 0.3.60 translates `image` blocks to the per-provider
+ # native shape; we trust that (the existing `images` param has
+ # shipped on every provider since v1.0.29).
+ # `document` (PDF) blocks: native on Anthropic upstream. For
+ # Gemini, anthropic-proxy rewrites document→image (keeping
+ # media_type=application/pdf), and Gemini's inline_data accepts
+ # that mime type natively. For OpenRouter, anthropic-proxy
+ # detects document blocks + injects the file-parser plugin. For
+ # OpenAI we refuse PDFs (no 9router translator path for the
+ # type:file shape, and Codex OAuth can't hit /v1/files anyway).
+ api = (api_type or "anthropic").lower()
+ supports_image = api in ("anthropic", "gemini", "openai", "openrouter", "gemini-cli")
+ # PDFs flow per provider:
+ # - Anthropic: native document blocks pass through cleanly.
+ # - Gemini: anthropic_proxy rewrites document → image_url with
+ # data:application/pdf base64; 9router translates to Gemini
+ # inlineData natively. VERIFIED empirically May 2026 (47K
+ # prompt tokens, real PaLM content summarized).
+ # - OpenRouter: file-parser plugin injected in anthropic-proxy.
+ # - OpenAI: REFUSED. OpenAI's image_url only accepts image/*
+ # mime; the type:file shape gets stringified by 9router.
+ # Workaround: route through `openrouter/openai/gpt-5` which
+ # uses OR's file-parser plugin.
+ # - Codex (cx/): models don't support PDFs.
+ supports_pdf = api in ("anthropic", "gemini", "gemini-cli", "openrouter")
+
+ # Per-file inline caps (raw bytes, before base64). Going over
+ # means the request would 4xx, blow our 64MB SDK buffer, or
+ # exceed the API's per-request cap on its own.
+ if api == "anthropic":
+ per_file_cap = 24 * 1024 * 1024
+ total_request_cap = 28 * 1024 * 1024 # under Anthropic's 32MB
+ elif api == "gemini":
+ per_file_cap = 14 * 1024 * 1024
+ total_request_cap = 15 * 1024 * 1024 # under Gemini's 20MB
+ elif api == "openai":
+ per_file_cap = 24 * 1024 * 1024
+ total_request_cap = 45 * 1024 * 1024 # under OpenAI's 50MB
+ elif api == "openrouter":
+ per_file_cap = 24 * 1024 * 1024
+ total_request_cap = 45 * 1024 * 1024
+ else:
+ per_file_cap = 0
+ total_request_cap = 0
+
+ # Running total of base64-expanded bytes already committed to the
+ # request. Anything that would push us over total_request_cap gets
+ # refused with concrete recovery actions.
+ b64_total = 0
+
+ for cp in context_paths:
+ path = cp.get("path", "") or ""
+ cp_type = cp.get("type", "file")
+ if not path or not os.path.exists(path):
+ sections.append(f"[Context: {path}, not found]")
+ continue
+ if cp_type == "directory" and os.path.isdir(path):
+ tree_lines = self._build_dir_tree(path, max_depth=4)
+ sections.append(
+ f"\n{chr(10).join(tree_lines)}\n"
+ )
+ continue
+ if cp_type != "file" or not os.path.isfile(path):
+ sections.append(f"[Context: {path}, type mismatch]")
+ continue
+ try:
+ size = os.path.getsize(path)
+ with open(path, "rb") as fh:
+ head = fh.read(4096)
+ kind, media_type = _sniff_file_kind(head, os.path.basename(path))
+
+ if kind == "text":
+ with open(path, "r", errors="replace") as f:
+ content = f.read(512_000)
+ sections.append(
+ f"\n{content}\n"
+ )
+ continue
+
+ # base64 expands ~4/3, ceil to be conservative.
+ b64_size = ((size + 2) // 3) * 4
+
+ if kind == "pdf":
+ if not supports_pdf:
+ if api == "openai":
+ # Falls here only for Codex variants (gpt-5.3-codex etc.),
+ # which don't accept PDFs even though their family does.
+ refusals.append(
+ f"[Attached PDF {os.path.basename(path)} ({size // 1024} KB) cannot be read on Codex models. "
+ "Switch to a non-Codex GPT-5 (e.g. gpt-5.5), Claude, Gemini 3.x, or "
+ "any model via OpenRouter to read PDFs natively.]"
+ )
+ else:
+ refusals.append(
+ f"[Attached PDF {os.path.basename(path)} ({size // 1024} KB) cannot be read on this provider. "
+ "Switch to a Claude model (Sonnet 4.6, Opus 4.7, Haiku 4.5), Gemini 3.x, GPT-5 (non-Codex), "
+ "or any model through OpenRouter to read PDFs natively.]"
+ )
+ continue
+ if size > per_file_cap:
+ refusals.append(
+ f"[Attached PDF {os.path.basename(path)} ({size // (1024*1024)} MB) exceeds the per-file cap "
+ f"of {per_file_cap // (1024*1024)} MB on this provider. Split the PDF or send a smaller excerpt.]"
+ )
+ continue
+ if b64_total + b64_size > total_request_cap:
+ room_mb = max(0, total_request_cap - b64_total) // (1024 * 1024)
+ refusals.append(
+ f"[Attached PDF {os.path.basename(path)} would push the request over "
+ f"{total_request_cap // (1024*1024)} MB encoded (provider cap). "
+ f"Only ~{room_mb} MB of room left this turn. Detach a file, or send PDFs in separate turns.]"
+ )
+ continue
+ with open(path, "rb") as fh:
+ data_b64 = _b64.b64encode(fh.read()).decode("ascii")
+ block = {
+ "type": "document",
+ "source": {
+ "type": "base64",
+ "media_type": "application/pdf",
+ "data": data_b64,
+ },
+ }
+ native.append(block)
+ b64_total += b64_size
+ continue
+
+ if kind == "image":
+ if not supports_image:
+ refusals.append(
+ f"[Attached image {os.path.basename(path)} cannot be displayed to this model. "
+ "Switch to a vision-capable model (Claude, GPT-4o/5, Gemini).]"
+ )
+ continue
+ if size > per_file_cap:
+ refusals.append(
+ f"[Attached image {os.path.basename(path)} ({size // (1024*1024)} MB) exceeds per-file cap.]"
+ )
+ continue
+ if b64_total + b64_size > total_request_cap:
+ room_mb = max(0, total_request_cap - b64_total) // (1024 * 1024)
+ refusals.append(
+ f"[Attached image {os.path.basename(path)} would push the request over "
+ f"{total_request_cap // (1024*1024)} MB encoded. ~{room_mb} MB of room left.]"
+ )
+ continue
+ with open(path, "rb") as fh:
+ data_b64 = _b64.b64encode(fh.read()).decode("ascii")
+ native.append({
+ "type": "image",
+ "source": {
+ "type": "base64",
+ "media_type": media_type or "image/png",
+ "data": data_b64,
+ },
+ })
+ b64_total += b64_size
+ continue
+
+ # binary, other
+ refusals.append(
+ f"[Attached binary file {os.path.basename(path)} not inlined. Convert to text first.]"
+ )
+ except Exception as e:
+ sections.append(f"[Context: {path}, error reading: {e}]")
+
+ # Anthropic prompt caching: tag the last document block as ephemeral
+ # so a follow-up turn referencing the same PDF stays cache-warm.
+ # Per Anthropic docs, only the trailing cache_control marker matters
+ # for cache prefix scope; earlier markers are ignored.
+ if api == "anthropic" and native:
+ for blk in reversed(native):
+ if blk.get("type") == "document":
+ blk["cache_control"] = {"type": "ephemeral"}
+ break
+
+ context_text = "\n\n".join(sections)
+ return context_text, native, refusals
+
+ # Legacy entry point retained for any external caller; routes to the
+ # new attachment resolver with anthropic-default routing (no native
+ # blocks emitted, so behavior is the safe text-only old path).
+ def _resolve_context_paths(self, context_paths: list | None) -> str:
+ text, _native, refusals = self._resolve_attachments(context_paths, api_type="anthropic", model="")
+ refusal_text = "\n\n".join(refusals)
+ return "\n\n".join(p for p in (text, refusal_text) if p)
+
async def _run_agent_loop(self, session_id: str, prompt: str, images: list | None = None, context_paths: list | None = None, forced_tools: list[str] | None = None, attached_skills: list | None = None, fork_session: bool = False, selected_browser_ids: list[str] | None = None):
"""Run the Claude Agent SDK query loop for a session."""
session = self.sessions.get(session_id)
if not session:
return
- prompt_content = self._build_prompt_content(prompt, images, context_paths, forced_tools, attached_skills)
+ from backend.apps.agents.providers.registry import get_api_type as _get_api_type
+ _api = _get_api_type(session.model)
+ prompt_content = self._build_prompt_content(
+ prompt, images, context_paths, forced_tools, attached_skills,
+ api_type=_api, model=session.model,
+ )
try:
from claude_agent_sdk import (
@@ -1715,6 +1957,7 @@ class AgentManager:
# than relying on the field's default_factory.
active_mcps=[],
)
+ _apply_context_window(sub_session)
self.sessions[sub_session_id] = sub_session
await ws_manager.broadcast_global("agent:status", {
"session_id": sub_session_id,
@@ -1814,10 +2057,13 @@ class AgentManager:
# Per-turn estimate of framework overhead (subtracted from displayed
# input). Conservative on purpose so honest over-shows beat lies.
- # 16K Claude Code preset, 12K base+deferred tools, 600/MCP, char/4 prompt.
+ # 16K Claude Code preset, 12K base+deferred tools, ~3K/MCP (real
+ # MCP tool definitions range 1-10K depending on server; 3K is a
+ # rough median that keeps the meter honest without over-trimming),
+ # char/4 of composed prompt.
_PRESET_OVERHEAD = 16_000
_TOOL_DEFS_OVERHEAD = 12_000
- _PER_MCP_OVERHEAD = 600
+ _PER_MCP_OVERHEAD = 3_000
_composed_tokens = len(composed_prompt or "") // 4
_mcp_tokens = len(session.active_mcps) * _PER_MCP_OVERHEAD
session.framework_overhead_tokens = (
@@ -2149,7 +2395,13 @@ class AgentManager:
options_kwargs = {
"model": resolved_model,
- "max_buffer_size": 5 * 1024 * 1024,
+ # 64 MB ceiling on the SDK <-> CLI JSON-RPC channel. The
+ # default 5 MB blocked any base64'd PDF over ~3.5 MB; we
+ # now route PDFs/images as native content blocks, which
+ # base64-expand by ~33%. 64 MB clears the largest single
+ # Anthropic PDF (32 MB raw) with headroom for prompt +
+ # tool results sharing the same frame.
+ "max_buffer_size": 64 * 1024 * 1024,
"permission_mode": "default",
"can_use_tool": can_use_tool,
"stderr": _stderr_cb,
@@ -2551,6 +2803,29 @@ class AgentManager:
"trimmed": trimmed,
"estimate_after": _est_tokens,
})
+ # Surface a visible system breadcrumb in the chat so
+ # the user (and the model on the next turn) know
+ # which MCPs got dropped. Without this, the model
+ # may keep trying to call a now-missing tool and
+ # the user has no idea why.
+ try:
+ _names = ", ".join(t.replace("mcp:", "") for t in trimmed)
+ _trim_msg = Message(
+ role="system",
+ content=(
+ f"Trimmed {len(trimmed)} app{'s' if len(trimmed) != 1 else ''} from this session to fit "
+ f"the model's context: {_names}. Re-activate via MCPSearch + MCPActivate "
+ "if you still need them."
+ ),
+ branch_id=session.active_branch_id,
+ )
+ session.messages.append(_trim_msg)
+ await ws_manager.send_to_session(session_id, "agent:message", {
+ "session_id": session_id,
+ "message": _trim_msg.model_dump(mode="json"),
+ })
+ except Exception:
+ logger.exception("failed to emit MCP-trimmed breadcrumb")
# Trimming changes mcp_servers / outputs context →
# rebuild options. The cheapest correct path is
# to flag for fork on next turn via needs_fork
@@ -3403,6 +3678,32 @@ class AgentManager:
(inp + cache_create + cache_read) * in_rate
+ out * out_rate
) / 1_000_000
+ elif api_type in ("openai", "gemini") or (
+ isinstance(resolved_model, str)
+ and (resolved_model.startswith("cp-openai/")
+ or resolved_model.startswith("cp-gemini/")
+ or resolved_model.startswith("cp-google/"))
+ ):
+ # Direct OpenAI/Gemini API key lane. SDK's
+ # total_cost_usd is computed at Anthropic
+ # rates (Opus pricing) — for GPT-5.4-Mini
+ # at $0.25/M input that's a 60x overcount
+ # ($30 instead of $0.04 per Mehmet-style
+ # 4-PDF turn). Use the published per-model
+ # rates instead.
+ from backend.apps.agents.providers.registry import get_direct_pricing
+ pricing = get_direct_pricing(resolved_model) or get_direct_pricing(session.model)
+ if pricing:
+ in_rate, out_rate = pricing
+ cost = (
+ (inp + cache_create + cache_read) * in_rate
+ + out * out_rate
+ ) / 1_000_000
+ else:
+ # Unknown model in this family: zero out
+ # rather than ship an Anthropic-rate
+ # estimate that's wildly wrong.
+ cost = 0.0
session.cost_usd = cost
await ws_manager.send_to_session(session_id, "agent:cost_update", {
@@ -3412,13 +3713,15 @@ class AgentManager:
if isinstance(usage, dict):
# Per-turn context-usage broadcast. Drives the UI
- # status pill, the auto-compact threshold (Phase 2),
- # and is the user's only honest signal that they're
- # approaching the context cap. 200K is the standard-
- # tier ceiling Anthropic returns the
- # long-context-required 429 against; it's also the
- # right denominator for OAuth Pro/Max users.
- ctx_used_pct = round(total_input / 200_000.0, 4) if total_input else 0.0
+ # status pill and the auto-compact threshold. The
+ # denominator is the session's real model cap,
+ # populated from registry.get_context_window at
+ # session creation, restore, and model-switch
+ # (see _apply_context_window). max(1, ...) is a
+ # belt-and-braces guard against zero/None drift
+ # from any future restore-from-disk corner case.
+ _ctx_window = max(1, getattr(session, "context_window", 0) or 200_000)
+ ctx_used_pct = round(total_input / _ctx_window, 4) if total_input else 0.0
cache_read_pct = round(cache_read / total_input, 4) if total_input else 0.0
try:
await ws_manager.send_to_session(session_id, "agent:context_update", {
@@ -3428,6 +3731,8 @@ class AgentManager:
"cache_read_tokens": cache_read,
"cache_read_pct": cache_read_pct,
"ctx_used_pct": ctx_used_pct,
+ "context_window": _ctx_window,
+ "framework_overhead_tokens": session.framework_overhead_tokens,
"active_mcps": list(session.active_mcps),
})
except Exception:
@@ -3533,26 +3838,70 @@ class AgentManager:
_stderr_tail = "\n".join(_stderr_buffer[-50:])
except Exception:
_stderr_tail = ""
+ # If we already streamed a substantive assistant response this
+ # turn, the user got their answer; the error fired on a
+ # subsequent step (title gen, follow-up tool turn, etc.).
+ # Don't blast a "context exceeded" card over a completed reply.
+ _streamed_substantive = bool(stream_text_msg_id) and _current_turn_emitted
+ if _streamed_substantive and _is_long_context_error(e, extra_text=_stderr_tail):
+ # Mark the session completed (not error), keep the assistant
+ # reply visible, and skip the overflow card. The next user
+ # turn will properly hit the pre-send guard if the chat is
+ # still over cap.
+ session.status = "completed"
+ if stream_text_msg_id:
+ try:
+ await ws_manager.send_to_session(session_id, "agent:stream_end", {
+ "session_id": session_id,
+ "message_id": stream_text_msg_id,
+ })
+ except Exception:
+ pass
+ return
if _is_long_context_error(e, extra_text=_stderr_tail):
friendly_msg = (
"This conversation has grown too large for your account's "
"standard context window. Long-context requests require an "
- "upgraded tier — switch to Chat mode or start a fresh chat "
+ "upgraded tier, switch to Chat mode or start a fresh chat "
"to continue."
)
error_msg = Message(role="system", content=friendly_msg, branch_id=session.active_branch_id)
session.messages.append(error_msg)
- await ws_manager.send_to_session(session_id, "agent:context_overflow", {
+ _ovf_payload = {
"session_id": session_id,
"reason": "long_context_required",
"message": friendly_msg,
+ "model": session.model,
+ "provider": session.provider,
+ "context_window": session.context_window,
+ "framework_overhead_tokens": session.framework_overhead_tokens,
"input_tokens": session.tokens.get("input", 0),
"active_mcps": list(session.active_mcps),
- })
+ "compact_threshold_pct": session.compact_threshold_pct,
+ "context_soft_cap_pct": session.context_soft_cap_pct,
+ }
+ await ws_manager.send_to_session(session_id, "agent:context_overflow", _ovf_payload)
await ws_manager.send_to_session(session_id, "agent:message", {
"session_id": session_id,
"message": error_msg.model_dump(mode="json"),
})
+ try:
+ from backend.apps.service.client import submit_diagnostic
+ submit_diagnostic({
+ "kind": "context_overflow",
+ "where": "agent_manager._run_streaming_turn",
+ "session_id": session_id,
+ "model": session.model,
+ "provider": session.provider,
+ "context_window": session.context_window,
+ "input_tokens": session.tokens.get("input", 0),
+ "framework_overhead_tokens": session.framework_overhead_tokens,
+ "active_mcps_count": len(session.active_mcps),
+ "messages_count": len(session.messages),
+ "error_preview": (str(e) or "")[:500],
+ })
+ except Exception:
+ logger.debug("submit_diagnostic for context_overflow failed", exc_info=True)
elif _is_auth_error(e, extra_text=_stderr_tail):
# Three sub-cases the user can hit, with distinct fixes:
# 1. "No credentials for provider: claude" — user picked a
@@ -3832,6 +4181,7 @@ class AgentManager:
data = _load_session_data(session_id)
if data:
session = AgentSession(**data)
+ _apply_context_window(session)
session.closed_at = None
self.sessions[session_id] = session
else:
@@ -3857,6 +4207,7 @@ class AgentManager:
logger.info(f"[MCP-DEBUG] Forking session: api_type changed {session.model}→{model}")
session.model = model
+ _apply_context_window(session)
session_changed = True
if mode and mode != session.mode:
session.mode = mode
@@ -4493,6 +4844,7 @@ class AgentManager:
raise ValueError(f"Session {session_id} not found in history")
session = AgentSession(**data)
+ _apply_context_window(session)
hours_since_closed = 0
if data.get("closed_at"):
@@ -4616,6 +4968,7 @@ class AgentManager:
if session.status in ("running", "waiting_approval"):
session.status = "stopped"
session.pending_approvals = []
+ _apply_context_window(session)
self.sessions[session.id] = session
_delete_session_file(sid)
logger.info(f"Restored session {session.id}")
@@ -4628,6 +4981,7 @@ class AgentManager:
if data is None:
raise ValueError(f"Session {session_id} not found")
source = AgentSession(**data)
+ _apply_context_window(source)
source_messages = list(source.messages)
if up_to_message_id:
@@ -4684,6 +5038,7 @@ class AgentManager:
sdk_session_id=source.sdk_session_id,
needs_fork=True,
)
+ _apply_context_window(new_session)
self.sessions[new_session.id] = new_session
@@ -4709,6 +5064,7 @@ class AgentManager:
if data is None:
raise ValueError(f"Session {source_session_id} not found")
source = AgentSession(**data)
+ _apply_context_window(source)
source_name = source.name
@@ -4724,7 +5080,14 @@ class AgentManager:
timestamp=msg.timestamp,
branch_id=msg.branch_id,
parent_id=old_to_new_msg.get(msg.parent_id) if msg.parent_id else None,
- context_paths=msg.context_paths,
+ # Sub-agents do NOT inherit parent's attached files. Each
+ # parent-message base64-expansion would re-fire in the
+ # sub-agent (cost explosion: a 25 MB PDF in parent +
+ # 5 InvokeAgent calls = 125 MB transmitted). The
+ # sub-agent receives the user's new message only; if it
+ # needs the file content, the parent message text from
+ # the prior turn already carries the model's summary.
+ context_paths=None,
attached_skills=msg.attached_skills,
forced_tools=msg.forced_tools,
images=msg.images,
@@ -4761,6 +5124,7 @@ class AgentManager:
dashboard_id=dashboard_id or source.dashboard_id,
parent_session_id=parent_session_id,
)
+ _apply_context_window(fork)
self.sessions[fork.id] = fork
diff --git a/backend/apps/agents/agents.py b/backend/apps/agents/agents.py
index 50885af5..4e7af578 100644
--- a/backend/apps/agents/agents.py
+++ b/backend/apps/agents/agents.py
@@ -257,6 +257,57 @@ async def warm_session_cache(session_id: str):
return {"ok": True}
+@agents.router.post("/sessions/{session_id}/compact")
+async def compact_session(session_id: str):
+ """Run the summarizer over older turns to free up context.
+
+ Wired to the 'Compact memory' button in the pre-send overflow banner
+ and the /compact slash command. Sets compacted_through_msg_id so the
+ next turn's history-builder uses the summary in place of the
+ original messages.
+ """
+ session = agent_manager.sessions.get(session_id)
+ if not session:
+ raise HTTPException(status_code=404, detail="session not found")
+ fired = agent_manager._maybe_compact(session, force=True)
+ if fired:
+ from backend.apps.agents.ws_manager import ws_manager
+ try:
+ await ws_manager.send_to_session(session_id, "agent:context_status", {
+ "session_id": session_id,
+ "reason": "compacted",
+ "compacted_through_msg_id": session.compacted_through_msg_id,
+ })
+ except Exception:
+ pass
+ return {"ok": True, "compacted": fired}
+
+
+@agents.router.post("/sessions/{session_id}/clear")
+async def clear_session(session_id: str):
+ """Drop all messages from the session, keep MCPs/model/tools.
+
+ Wired to the /clear slash command. Quickest path to recover from an
+ overflow short of starting a fresh chat."""
+ session = agent_manager.sessions.get(session_id)
+ if not session:
+ raise HTTPException(status_code=404, detail="session not found")
+ session.messages = []
+ session.compacted_through_msg_id = None
+ session.tokens = {"input": 0, "output": 0}
+ session.needs_fresh_session = True
+ from backend.apps.agents.ws_manager import ws_manager
+ try:
+ await ws_manager.send_to_session(session_id, "agent:status", {
+ "session_id": session_id,
+ "status": session.status,
+ "session": session.model_dump(mode="json"),
+ })
+ except Exception:
+ pass
+ return {"ok": True}
+
+
@agents.router.get("/subscriptions/status")
async def subscriptions_status():
"""Check if 9Router is running and list connected providers."""
diff --git a/backend/apps/agents/anthropic_proxy.py b/backend/apps/agents/anthropic_proxy.py
index 3fd93da9..7d3d16f6 100644
--- a/backend/apps/agents/anthropic_proxy.py
+++ b/backend/apps/agents/anthropic_proxy.py
@@ -91,8 +91,66 @@ def _is_openai_max_completion_tokens_model(model: str) -> bool:
return any(m.startswith(p) for p in _OPENAI_MAX_COMPLETION_TOKENS_MODELS)
+def _rewrite_document_to_openai_file(parsed: dict) -> None:
+ """In-place: Anthropic `document` (PDF) and `image` blocks → OpenAI
+ Chat Completions native shapes. Critically also handles `image` →
+ `image_url` because **9router 0.3.60 strips any block type that is
+ not 'text' or 'image_url'**, stringifying it into a text block (verified
+ in router/.next/server/chunks/318.js, the `b.messages.map` translator).
+ So we have to land on `image_url` for images AND `file` for PDFs.
+
+ For document: → `{type:"file", file:{filename, file_data:"data:application/pdf;base64,..."}}`.
+ For image: → `{type:"image_url", image_url:{url:"data:image/...;base64,..."}}`.
+
+ OpenAI Chat Completions natively accepts both shapes on GPT-5.x vision
+ models. 9router preserves `image_url` and (per the same chunk's check
+ for unknown types getting passed-through if NOT in the rewrite-list)
+ seems to preserve `file` too in this codepath. Verified empirically
+ May 2026 after fixing the image stringification bug.
+ """
+ msgs = parsed.get("messages") if isinstance(parsed, dict) else None
+ if not isinstance(msgs, list):
+ return
+ counter = 0
+ for m in msgs:
+ content = m.get("content") if isinstance(m, dict) else None
+ if not isinstance(content, list):
+ continue
+ for block in content:
+ if not isinstance(block, dict):
+ continue
+ btype = block.get("type")
+ src = block.get("source") or {}
+ if not isinstance(src, dict) or src.get("type") != "base64":
+ continue
+ data = src.get("data")
+ if not isinstance(data, str) or not data:
+ continue
+ media_type = src.get("media_type") or ""
+
+ # 9router 0.3.60 chunk 318 stringifies ANY non-`text`/`image_url`
+ # block. Image blocks → image_url with data: URL.
+ # PDFs on OpenAI direct are REFUSED upstream (agent_manager
+ # _resolve_attachments has openai NOT in supports_pdf) because
+ # OpenAI Chat Completions rejects non-image mime types inside
+ # image_url with "Invalid MIME type. Only image types are
+ # supported." (verified empirically May 2026). The shipping
+ # path for OpenAI PDFs is openrouter/openai/gpt-5 which uses
+ # OR's file-parser plugin.
+ if btype != "image":
+ continue
+ mt = media_type or "image/png"
+ block.clear()
+ block["type"] = "image_url"
+ block["image_url"] = {
+ "url": f"data:{mt};base64,{data}",
+ }
+
+
def _scrub_request_for_openai_gpt5(body: bytes) -> bytes:
- """Rename max_tokens to max_completion_tokens for GPT-5; bytes in/out, never raises."""
+ """Rename max_tokens→max_completion_tokens for GPT-5 AND rewrite any
+ Anthropic document blocks to OpenAI type:file shape so PDFs flow
+ natively on GPT-5.x vision models. Bytes in/out; never raises."""
if not body:
return body
try:
@@ -101,17 +159,122 @@ def _scrub_request_for_openai_gpt5(body: bytes) -> bytes:
return body
if not isinstance(parsed, dict):
return body
+ mutated = False
if "max_tokens" in parsed and "max_completion_tokens" not in parsed:
parsed["max_completion_tokens"] = parsed.pop("max_tokens")
- return json.dumps(parsed).encode("utf-8")
- if "max_tokens" in parsed and "max_completion_tokens" in parsed:
+ mutated = True
+ elif "max_tokens" in parsed and "max_completion_tokens" in parsed:
parsed.pop("max_tokens", None)
- return json.dumps(parsed).encode("utf-8")
- return body
+ mutated = True
+ try:
+ before = json.dumps(parsed.get("messages"), sort_keys=True) if "messages" in parsed else ""
+ _rewrite_document_to_openai_file(parsed)
+ after = json.dumps(parsed.get("messages"), sort_keys=True) if "messages" in parsed else ""
+ if before != after:
+ mutated = True
+ except Exception:
+ pass
+ return json.dumps(parsed).encode("utf-8") if mutated else body
+
+
+def _rewrite_document_to_image(parsed: dict) -> None:
+ """In-place: rewrite Anthropic `document` (PDF) AND `image` content
+ blocks → OpenAI `image_url` shape with a `data:` URL. Critical fix
+ for 9router 0.3.60 which **only translates `image_url` blocks** to
+ Gemini's `inlineData` (verified in router/.next/server/chunks/318.js:
+ `b.image_url?.url?.startsWith('data:')` → builds `{inlineData:{mime_type,data}}`).
+
+ Anthropic-shape `image`/`document` blocks fall through 9router's
+ content filter and either get stringified or dropped, which is why
+ PDFs were silently missing from Gemini requests until this rewrite.
+
+ For PDFs we set mime_type=application/pdf in the data URL; Gemini's
+ inlineData accepts it natively.
+
+ Strictly defensive: rewrite only when source.type='base64' and data
+ is present. Unknown shapes pass through untouched."""
+ msgs = parsed.get("messages") if isinstance(parsed, dict) else None
+ if not isinstance(msgs, list):
+ return
+ for m in msgs:
+ content = m.get("content") if isinstance(m, dict) else None
+ if not isinstance(content, list):
+ continue
+ for block in content:
+ if not isinstance(block, dict):
+ continue
+ btype = block.get("type")
+ if btype not in ("document", "image"):
+ continue
+ src = block.get("source") or {}
+ if not isinstance(src, dict) or src.get("type") != "base64":
+ continue
+ data = src.get("data")
+ if not isinstance(data, str) or not data:
+ continue
+ if btype == "document":
+ media_type = src.get("media_type") or "application/pdf"
+ else:
+ media_type = src.get("media_type") or "image/png"
+ block.clear()
+ block["type"] = "image_url"
+ block["image_url"] = {
+ "url": f"data:{media_type};base64,{data}",
+ }
+
+
+_OPENROUTER_MODEL_PREFIXES = ("openrouter/", "or:")
+
+
+def _is_openrouter_model(model: str) -> bool:
+ m = (model or "").strip().lower()
+ return any(m.startswith(p) for p in _OPENROUTER_MODEL_PREFIXES)
+
+
+def _inject_openrouter_file_parser(body: bytes) -> bytes:
+ """When the request has document blocks AND is bound for OpenRouter,
+ inject the file-parser plugin so OR's universal PDF support kicks in
+ on any model (free models get pdf-text engine; native PDF models can
+ still see the document directly). The plugins field sits at the top
+ level alongside `messages`; we don't touch the message content blocks,
+ OR's normaliser handles Anthropic→target translation.
+ Bytes-in/out, never raises."""
+ if not body:
+ return body
+ try:
+ parsed = json.loads(body)
+ except Exception:
+ return body
+ if not isinstance(parsed, dict):
+ return body
+ msgs = parsed.get("messages")
+ if not isinstance(msgs, list):
+ return body
+ has_doc = False
+ for m in msgs:
+ content = m.get("content") if isinstance(m, dict) else None
+ if not isinstance(content, list):
+ continue
+ for block in content:
+ if isinstance(block, dict) and block.get("type") == "document":
+ has_doc = True
+ break
+ if has_doc:
+ break
+ if not has_doc:
+ return body
+ existing = parsed.get("plugins")
+ plugins = existing if isinstance(existing, list) else []
+ if not any(isinstance(p, dict) and p.get("id") == "file-parser" for p in plugins):
+ plugins.append({"id": "file-parser", "pdf": {"engine": "pdf-text"}})
+ parsed["plugins"] = plugins
+ return json.dumps(parsed).encode("utf-8")
def _scrub_request_for_gemini(body: bytes) -> bytes:
- """Strip Gemini-incompatible schema keys from request tools. Bytes-in/out, never raises."""
+ """Strip Gemini-incompatible schema keys from request tools AND
+ rewrite Anthropic document blocks to image-shape so 9router's
+ inline_data translator picks them up. Bytes-in/out, never raises."""
if not body:
return body
try:
@@ -127,6 +290,11 @@ def _scrub_request_for_gemini(body: bytes) -> bytes:
_scrub_gemini_schema(t["input_schema"])
if isinstance(t.get("parameters"), (dict, list)):
_scrub_gemini_schema(t["parameters"])
+ try:
+ if isinstance(parsed, dict):
+ _rewrite_document_to_image(parsed)
+ except Exception:
+ pass
return json.dumps(parsed).encode("utf-8")
@@ -163,7 +331,14 @@ def _is_gemini_model(model: str) -> bool:
def _pick_upstream(model: str) -> tuple[str, dict[str, str]]:
- """Return (base_url_without_v1, auth_headers) for this model."""
+ """Return (base_url_without_v1, auth_headers) for this model.
+
+ Routing for Claude-family models:
+ 1. openswarm-pro mode → cloud proxy with bearer
+ 2. Direct Anthropic API key set → api.anthropic.com (preferred when
+ user has their own key, avoids the 8h OAuth expiry pain)
+ 3. Fallback → 9router (cc/ OAuth subscription, may 401 if expired)
+ Everything non-Claude goes to 9router for translation."""
from backend.apps.settings.settings import load_settings
s = load_settings()
@@ -173,6 +348,12 @@ def _pick_upstream(model: str) -> tuple[str, dict[str, str]]:
proxy = (getattr(s, "openswarm_proxy_url", "") or "https://api.openswarm.com").rstrip("/")
if bearer and proxy:
return (proxy, {"Authorization": f"Bearer {bearer}"})
+ ak = getattr(s, "anthropic_api_key", "") or ""
+ if ak.strip():
+ return ("https://api.anthropic.com", {
+ "x-api-key": ak.strip(),
+ "anthropic-version": "2023-06-01",
+ })
return ("http://127.0.0.1:20128", {"x-api-key": "9router"})
@@ -210,6 +391,8 @@ async def proxy(rest: str, request: Request):
body = _scrub_request_for_gemini(body)
if _is_openai_max_completion_tokens_model(model):
body = _scrub_request_for_openai_gpt5(body)
+ if _is_openrouter_model(model):
+ body = _inject_openrouter_file_parser(body)
base_url, auth_headers = _pick_upstream(model)
diff --git a/backend/apps/agents/models.py b/backend/apps/agents/models.py
index 640b338d..13e21676 100644
--- a/backend/apps/agents/models.py
+++ b/backend/apps/agents/models.py
@@ -126,11 +126,12 @@ class AgentSession(BaseModel):
active_mcps: list[str] = Field(default_factory=list)
# Heuristic preamble tokens (preset + tool defs + MCP descs + composed prompt); subtracted from displayed input.
framework_overhead_tokens: int = 0
- # Live ctx_used ratio triggering _maybe_compact at the next turn boundary; turn-based thresholds break under uneven workloads. 0.65 = 130K of 200K.
+ # Live ctx_used ratio triggering _maybe_compact at the next turn boundary; turn-based thresholds break under uneven workloads. Ratio of context_window, so 0.65 means 650K on a 1M-window model and 130K on a 200K-window model.
compact_threshold_pct: float = 0.65
compacted_through_msg_id: Optional[str] = None
- # Hard pre-send guard at 0.90 (= 180K); past compaction we LRU-trim active_mcps, then surface the overflow card.
+ # Hard pre-send guard at 0.90; past compaction we LRU-trim active_mcps, then surface the overflow card.
context_soft_cap_pct: float = 0.90
+ # Conservative default. Always overwritten at session creation, restore, and model-switch via _apply_context_window in agent_manager so the real model cap is used instead. Don't bump this without re-checking the trim/guard logic.
context_window: int = 200_000
# Provider-agnostic thinking level (off/low/medium/high/auto), translated per-API in agent_manager; only affects reasoning-flagged models.
thinking_level: Literal["off", "low", "medium", "high", "auto"] = "auto"
diff --git a/backend/apps/agents/providers/registry.py b/backend/apps/agents/providers/registry.py
index 06755efe..fbb09bbb 100644
--- a/backend/apps/agents/providers/registry.py
+++ b/backend/apps/agents/providers/registry.py
@@ -186,6 +186,43 @@ _or_models_cache: dict = {"models": None, "fetched_at": 0.0, "ok": False}
_9router_cache: dict = {"available": None, "checked_at": 0}
+# Per-model published pricing in $/1M tokens (input, output) for direct
+# API key lanes. Sourced from each provider's official pricing page as of
+# May 2026. The Claude Agent SDK ALWAYS computes total_cost_usd at
+# Anthropic rates; for any non-Anthropic upstream the SDK number is
+# 50-1000x wrong and we MUST recompute. Used by agent_manager's cost
+# recompute logic.
+_DIRECT_API_PRICING: dict[str, tuple[float, float]] = {
+ # OpenAI GPT-5.x family (source: platform.openai.com/docs/pricing).
+ "gpt-5.5": (1.25, 10.00),
+ "gpt-5.4": (1.25, 10.00),
+ "gpt-5.4-mini": (0.25, 2.00),
+ "gpt-5.3-codex": (1.25, 10.00),
+ "gpt-5.3-codex-high": (1.25, 10.00),
+ "gpt-5.3-codex-xhigh": (1.25, 10.00),
+ # Google Gemini direct API (ai.google.dev/pricing).
+ "gemini-3.1-pro-preview": (1.25, 10.00),
+ "gemini-3.1-flash-lite-preview": (0.10, 0.40),
+ "gemini-3-pro-preview": (1.25, 10.00),
+ "gemini-3-flash-preview": (0.30, 2.50),
+}
+
+
+def get_direct_pricing(model_id: str) -> tuple[float, float] | None:
+ """($/1M input, $/1M output) for an OpenAI or Gemini direct-API model_id.
+ Returns None for any model not in the pricing table; callers fall back
+ to the SDK's (Anthropic-rate) estimate which is wrong but at least
+ deterministic."""
+ if not isinstance(model_id, str):
+ return None
+ bare = model_id
+ for prefix in ("cp-openai/", "cp-gemini/", "cp-google/", "openai/", "google/", "gemini/"):
+ if bare.startswith(prefix):
+ bare = bare[len(prefix):]
+ break
+ return _DIRECT_API_PRICING.get(bare)
+
+
def get_openrouter_pricing(resolved_model: str) -> tuple[float, float] | None:
"""($/1M input, $/1M output) for an openrouter/ id, or None if not cached."""
if not isinstance(resolved_model, str) or not resolved_model.startswith("openrouter/"):
diff --git a/backend/apps/service/client.py b/backend/apps/service/client.py
index c92a217c..ba7ff5b3 100644
--- a/backend/apps/service/client.py
+++ b/backend/apps/service/client.py
@@ -300,8 +300,10 @@ _DEFAULT_SYNC_PATH = "/api/service/sync"
def submit(kind: str, payload: dict) -> None:
- """Legacy shim; routes through sync(). Kept for back-compat during
- migration. New code should call sync() directly."""
+ """Routes through sync(). The cloud demuxes by payload shape (state /
+ sync / diagnostic / event), so kind here is informational; the routing
+ happens server-side in openswarm-cloud/src/routes/service/ingest.ts.
+ New call sites should use sync() directly with a well-shaped payload."""
sync(payload)
diff --git a/backend/apps/settings/settings.py b/backend/apps/settings/settings.py
index 08bd3d6b..aa99cafe 100644
--- a/backend/apps/settings/settings.py
+++ b/backend/apps/settings/settings.py
@@ -64,11 +64,38 @@ async def settings_lifespan():
await sync_custom_providers(getattr(s, "custom_providers", None) or [])
_asyncio.create_task(_boot_router_then_sync())
+ _asyncio.create_task(_upload_dir_gc_loop())
except Exception as e:
logger.warning(f"9Router sync startup failed: {e}")
yield
+async def _upload_dir_gc_loop():
+ """Daily GC of UPLOAD_DIR. Without this, every PDF/image the user
+ drops sits in the OS temp dir forever, growing unbounded across
+ sessions. We keep files for 7 days to make resume-after-restart
+ work, then delete. macOS temp under /var/folders/... is auto-purged
+ by the OS but not aggressively; Windows temp is not. Belt and braces.
+ Errors are swallowed: a chmod hiccup or in-use lock should never
+ crash the backend."""
+ import asyncio as _a
+ while True:
+ try:
+ now = time.time()
+ cutoff = now - 7 * 86400
+ if os.path.isdir(UPLOAD_DIR):
+ for entry in os.listdir(UPLOAD_DIR):
+ p = os.path.join(UPLOAD_DIR, entry)
+ try:
+ if os.path.isfile(p) and os.path.getmtime(p) < cutoff:
+ os.remove(p)
+ except Exception:
+ continue
+ except Exception:
+ pass
+ await _a.sleep(24 * 3600)
+
+
settings = SubApp("settings", settings_lifespan)
@@ -311,29 +338,259 @@ UPLOAD_DIR = os.path.join(tempfile.gettempdir(), "self-swarm-uploads")
os.makedirs(UPLOAD_DIR, exist_ok=True)
+def _sniff_file_kind(contents: bytes, name: str) -> tuple[str, str | None]:
+ """Classify an uploaded file as text/pdf/image/binary so the agent
+ layer can route it (inline as text, send as native document/image
+ block, or refuse). Returns (kind, media_type)."""
+ head = contents[:4096]
+ if head.startswith(b"%PDF-"):
+ return ("pdf", "application/pdf")
+ if head.startswith(b"\x89PNG\r\n\x1a\n"):
+ return ("image", "image/png")
+ if head.startswith(b"\xff\xd8\xff"):
+ return ("image", "image/jpeg")
+ if head.startswith(b"GIF87a") or head.startswith(b"GIF89a"):
+ return ("image", "image/gif")
+ if head[:4] == b"RIFF" and head[8:12] == b"WEBP":
+ return ("image", "image/webp")
+ # Other common binary signatures that don't contain a null byte in the
+ # first few bytes (so the null-byte fallback below would miss them):
+ # zip/docx/xlsx/pptx/jar/apk/odt (PK\x03\x04), gzip (\x1f\x8b),
+ # 7z (7z\xbc\xaf), tar (ustar magic at offset 257), rar (Rar!\x1a\x07),
+ # ELF (\x7fELF), Mach-O (\xfe\xed\xfa\xce / \xce\xfa\xed\xfe), Win exe
+ # (MZ), Java class (\xca\xfe\xba\xbe), sqlite (SQLite format 3\x00).
+ if (head.startswith(b"PK\x03\x04") or head.startswith(b"PK\x05\x06") or
+ head.startswith(b"\x1f\x8b") or head.startswith(b"7z\xbc\xaf\x27\x1c") or
+ head.startswith(b"Rar!\x1a\x07") or head.startswith(b"\x7fELF") or
+ head.startswith(b"\xfe\xed\xfa\xce") or head.startswith(b"\xce\xfa\xed\xfe") or
+ head.startswith(b"\xfe\xed\xfa\xcf") or head.startswith(b"\xcf\xfa\xed\xfe") or
+ head.startswith(b"MZ") or head.startswith(b"\xca\xfe\xba\xbe") or
+ head.startswith(b"SQLite format 3\x00")):
+ return ("binary", None)
+ # Binary heuristic: any null bytes in the first 4KB is a strong "not text" signal.
+ # Falls back gracefully for unusual encodings (UTF-16 has nulls too, but we treat
+ # those as binary for safety since the agent's `open(..., "r")` would misread them).
+ if b"\x00" in head:
+ return ("binary", None)
+ try:
+ head.decode("utf-8")
+ return ("text", "text/plain")
+ except UnicodeDecodeError:
+ return ("binary", None)
+
+
+def _estimate_pdf_tokens(contents: bytes) -> int:
+ """Conservative PDF token estimate without a parser dep.
+
+ We use two signals and take the MAX so the chip + dry-run never
+ under-report:
+
+ 1) Page count from the PDF catalog (regex over /Type /Pages /Count
+ then a fallback for /Count just before /Kids). When found, we
+ estimate 750 tokens/page, a fair midpoint between dense academic
+ papers (~1200) and sparse decks (~300).
+ 2) Byte-size heuristic. PDFs compress text and embed images; the
+ actual token cost on Anthropic's vision tier scales with file
+ size. ~1 token per 80 bytes is conservative.
+
+ Taking max() means a small page count on a huge PDF (image-heavy)
+ still reads as expensive, and a huge page count on a small PDF still
+ reads as expensive. The chip never lies that an attachment is cheap."""
+ import re as _re
+ by_pages = 0
+ try:
+ # Prefer the root catalog's /Pages entry. PDFs can have nested
+ # /Count fields (outlines, sub-pages), so anchor on /Type /Pages.
+ m = _re.search(rb"/Type\s*/Pages\b[^>]{0,200}?/Count\s+(\d+)", contents, _re.DOTALL)
+ if not m:
+ # Fallback: catalog declares /Pages then references /Count via /Kids.
+ m = _re.search(rb"/Pages[^>]{0,200}?/Count\s+(\d+)", contents, _re.DOTALL)
+ if m:
+ pages = int(m.group(1))
+ if 0 < pages < 10_000:
+ by_pages = pages * 750
+ except Exception:
+ pass
+ by_bytes = max(1_000, min(len(contents) // 80, 2_000_000))
+ return max(by_pages, by_bytes)
+
+
@settings.router.post("/upload-files")
async def upload_files(files: list[UploadFile] = File(...)):
- """Accept dropped files, save them, and return their server-side paths."""
+ """Accept dropped files, sniff their kind, save them, and return
+ server-side paths + a `kind` + `tokens` estimate per file.
+
+ The chat UI uses `tokens` for the per-chip chip count and the pre-send
+ dry-run guard; it uses `kind` to decide whether the file routes as
+ inline text, a native document block (PDF on Anthropic/Gemini), an
+ image block (vision-capable models), or gets refused (other binary,
+ until we add Files API support).
+
+ Estimates per kind:
+ - text: char/4 of the actually-readable text (capped at 512KB)
+ - pdf: page-count * 750 (conservative; real text-heavy PDFs run
+ ~500-1200 tokens/page)
+ - image: 1500 (Anthropic's per-image baseline; varies by size)
+ - binary: 0 (refused at agent time, won't enter context)
+ """
results = []
for f in files:
safe_name = os.path.basename(f.filename or "untitled")
- dest = os.path.join(UPLOAD_DIR, safe_name)
-
- counter = 1
- base, ext = os.path.splitext(safe_name)
- while os.path.exists(dest):
- dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}")
- counter += 1
-
+ # Strip path separators that survived basename on Windows-typed
+ # uploads where filename arrived with backslashes preserved.
+ safe_name = safe_name.replace("\\", "_").replace("/", "_") or "untitled"
contents = await f.read()
- with open(dest, "wb") as fh:
- fh.write(contents)
- results.append({"path": dest, "name": safe_name, "size": len(contents)})
+ # Atomic create-with-collision-retry so two concurrent uploads with
+ # the same filename never overwrite each other. The previous
+ # exists() then open() pattern had a race window: both callers
+ # would observe `dest` free and both would write, with the second
+ # winning. O_EXCL fails the create if anyone else got there first.
+ base, ext = os.path.splitext(safe_name)
+ dest = os.path.join(UPLOAD_DIR, safe_name)
+ counter = 0
+ fd = None
+ while fd is None:
+ try:
+ fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644)
+ except FileExistsError:
+ counter += 1
+ if counter > 10_000:
+ raise HTTPException(status_code=500, detail="upload dedup exhausted")
+ dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}")
+ try:
+ with os.fdopen(fd, "wb") as fh:
+ fh.write(contents)
+ except Exception:
+ try:
+ os.remove(dest)
+ except Exception:
+ pass
+ raise
+
+ kind, media_type = _sniff_file_kind(contents, safe_name)
+
+ if kind == "text":
+ try:
+ with open(dest, "r", errors="replace") as fh:
+ txt = fh.read(512_000)
+ tokens_est = max(0, len(txt) // 4)
+ except Exception:
+ tokens_est = min(len(contents), 512_000) // 4
+ elif kind == "pdf":
+ tokens_est = _estimate_pdf_tokens(contents)
+ elif kind == "image":
+ tokens_est = 1_500
+ else:
+ tokens_est = 0
+
+ results.append({
+ "path": dest,
+ "name": safe_name,
+ "size": len(contents),
+ "tokens": tokens_est,
+ "kind": kind,
+ "media_type": media_type,
+ })
return JSONResponse({"files": results})
+class _SummarizeRequest(BaseModel):
+ path: str
+ target_tokens: int = 4_000
+ primary_model: Optional[str] = None
+
+
+@settings.router.post("/summarize-file")
+async def summarize_file(req: _SummarizeRequest):
+ """Compress an attached file down to a fact-dense summary the agent can
+ still reason over without paying the full token cost.
+
+ Called from the chat-input attach handler when one file alone would
+ exceed 50% of the selected model's context window. The summary is
+ written to a sibling file with `.summary.txt` suffix in UPLOAD_DIR so
+ the existing attachment plumbing (paths flow through context_paths)
+ works unchanged. Aux model picked via provider-agnostic
+ resolve_aux_model, so users on OpenAI/Gemini/OpenRouter get summarized
+ via their own provider's cheap tier (never hardcoded to Haiku).
+ """
+ src = req.path
+ if not os.path.isfile(src):
+ raise HTTPException(status_code=404, detail="file not found")
+ if not os.path.commonpath([os.path.realpath(src), os.path.realpath(UPLOAD_DIR)]) == os.path.realpath(UPLOAD_DIR):
+ raise HTTPException(status_code=400, detail="path outside upload dir")
+
+ try:
+ with open(src, "r", errors="replace") as fh:
+ raw = fh.read(2_000_000)
+ except Exception as e:
+ raise HTTPException(status_code=500, detail=f"read failed: {e}")
+
+ if (len(raw) // 4) <= max(1, req.target_tokens):
+ return JSONResponse({"path": src, "tokens": len(raw) // 4, "size": len(raw), "summarized": False})
+
+ try:
+ from backend.apps.agents.providers.registry import resolve_aux_model, get_api_type
+ from backend.apps.settings.credentials import get_anthropic_client_for_model
+ s = load_settings()
+ aux_model, _base = await resolve_aux_model(
+ s,
+ preferred_tier="haiku",
+ primary_api=get_api_type(req.primary_model) if req.primary_model else None,
+ )
+ client = get_anthropic_client_for_model(s, aux_model)
+ system = (
+ "You compress a document into a fact-dense summary while preserving every "
+ "specific entity, number, date, quote, code identifier, and decision. Use "
+ "short bullets grouped by section. Never invent. If a section is unclear, "
+ "say so. Aim for roughly the target token budget."
+ )
+ user = (
+ f"Target length: ~{req.target_tokens} tokens.\n\n"
+ f"\n{raw}\n\n\n"
+ "Summary:"
+ )
+ resp = await client.messages.create(
+ model=aux_model,
+ max_tokens=min(8_192, max(512, req.target_tokens + 1_024)),
+ system=system,
+ messages=[{"role": "user", "content": user}],
+ )
+ summary = ""
+ try:
+ for b in (getattr(resp, "content", None) or []):
+ t = getattr(b, "text", None)
+ if isinstance(t, str) and t:
+ summary += t
+ except Exception:
+ summary = ""
+ if not summary.strip():
+ raise RuntimeError("empty summary from aux model")
+ except Exception as e:
+ raise HTTPException(status_code=502, detail=f"summarize failed: {e}")
+
+ base, _ext = os.path.splitext(src)
+ dest = f"{base}.summary.txt"
+ counter = 1
+ while os.path.exists(dest):
+ dest = f"{base}.summary_{counter}.txt"
+ counter += 1
+ body = (
+ f"Summary of {os.path.basename(src)} "
+ f"(compressed from ~{len(raw) // 4} tokens to ~{len(summary) // 4} tokens)\n\n"
+ f"{summary}\n"
+ )
+ with open(dest, "w") as fh:
+ fh.write(body)
+ return JSONResponse({
+ "path": dest,
+ "tokens": len(body) // 4,
+ "size": len(body),
+ "summarized": True,
+ })
+
+
@settings.router.get("/browse-directories")
async def browse_directories(path: str = Query(default="")) -> BrowseResponse:
target = path.strip() if path.strip() else os.path.expanduser("~")
diff --git a/backend/tests/test_v2_invariants.py b/backend/tests/test_v2_invariants.py
index af815e98..12eefefc 100644
--- a/backend/tests/test_v2_invariants.py
+++ b/backend/tests/test_v2_invariants.py
@@ -994,6 +994,873 @@ def test_get_context_window_unknown_returns_default():
assert cw == 128_000
+def test_apply_context_window_overwrites_default_for_opus_4_7():
+ """Regression for issue #39: AgentSession used to stick at the 200k
+ dataclass default for every model. _apply_context_window must pull
+ the real 1M value from the registry for opus-4-7 / sonnet so the
+ soft-cap trim and the % meter both reflect the real model cap."""
+ from backend.apps.agents.models import AgentSession
+ from backend.apps.agents.agent_manager import _apply_context_window
+ s = AgentSession(id="x", name="t", model="opus-4-7", mode="agent")
+ assert s.context_window == 200_000
+ _apply_context_window(s)
+ assert s.context_window == 1_000_000
+ s2 = AgentSession(id="y", name="t", model="sonnet", mode="agent")
+ _apply_context_window(s2)
+ assert s2.context_window == 1_000_000
+ s3 = AgentSession(id="z", name="t", model="haiku", mode="agent")
+ _apply_context_window(s3)
+ assert s3.context_window == 200_000
+
+
+def test_apply_context_window_silent_on_unknown_model():
+ """Bad lookup must NEVER raise; sessions with unknown/custom models
+ that aren't in the registry fall back to the 128k registry default
+ without breaking session creation."""
+ from backend.apps.agents.models import AgentSession
+ from backend.apps.agents.agent_manager import _apply_context_window
+ s = AgentSession(id="x", name="t", model="nonexistent-model-xyz", mode="agent")
+ _apply_context_window(s)
+ assert s.context_window > 0
+
+
+def test_estimate_pdf_tokens_floors_empty_pdf_at_byte_heuristic():
+ """A truly empty / minimal PDF still returns a non-zero estimate so
+ the dry-run guard doesn't allow many tiny PDFs through silently."""
+ from backend.apps.settings.settings import _estimate_pdf_tokens
+ assert _estimate_pdf_tokens(b"") >= 1_000
+ assert _estimate_pdf_tokens(b"%PDF-1.4\n") >= 1_000
+
+
+def test_estimate_pdf_tokens_takes_max_of_pages_and_bytes():
+ """An image-heavy PDF with low page count should still report high
+ tokens via the byte-size signal; we never under-report."""
+ from backend.apps.settings.settings import _estimate_pdf_tokens
+ # 8MB PDF with 1 page (image-heavy) — byte heuristic should dominate.
+ fake = b"%PDF-1.4\n/Type /Pages /Count 1\n" + b"X" * (8 * 1024 * 1024)
+ tokens = _estimate_pdf_tokens(fake)
+ # byte heuristic: 8MB / 80 = 100k tokens > pages * 750 = 750
+ assert tokens >= 100_000
+
+
+def test_estimate_pdf_tokens_caps_malformed_count():
+ """A PDF with /Count 999999 (malformed or hostile) does NOT bypass
+ the 10k pages sanity cap; falls through to byte heuristic instead."""
+ from backend.apps.settings.settings import _estimate_pdf_tokens
+ fake = b"%PDF-1.4\n/Type /Pages /Count 999999\n"
+ t = _estimate_pdf_tokens(fake)
+ # Should NOT be 999999 * 750 = 750 million.
+ assert t < 50_000_000
+
+
+def test_upload_dedup_under_concurrent_uploads():
+ """Run N parallel uploads of the same logical filename through threads
+ and verify EVERY upload landed at a distinct path (no overwrites)."""
+ import os, threading
+ from backend.apps.settings.settings import UPLOAD_DIR
+ os.makedirs(UPLOAD_DIR, exist_ok=True)
+ name = f"test_concurrent_{os.getpid()}.txt"
+ results: list[str] = []
+ lock = threading.Lock()
+
+ def writer():
+ base, ext = os.path.splitext(name)
+ dest = os.path.join(UPLOAD_DIR, name)
+ counter = 0
+ fd = None
+ while fd is None:
+ try:
+ fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644)
+ except FileExistsError:
+ counter += 1
+ dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}")
+ with os.fdopen(fd, "wb") as fh:
+ fh.write(b"hi")
+ with lock:
+ results.append(dest)
+
+ threads = [threading.Thread(target=writer) for _ in range(10)]
+ for t in threads: t.start()
+ for t in threads: t.join()
+ try:
+ assert len(set(results)) == 10, f"expected 10 distinct paths, got {len(set(results))}"
+ finally:
+ for p in results:
+ try: os.remove(p)
+ except Exception: pass
+
+
+def test_resolve_attachments_handles_missing_path_gracefully():
+ """If a path in context_paths no longer exists (file deleted, TTL
+ cleanup fired, restored session referencing temp file across reboot),
+ we emit a 'not found' refusal instead of crashing."""
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ text, native, refusals = mgr._resolve_attachments(
+ [{"path": "/var/folders/nonexistent/definitely-gone.pdf", "type": "file"}],
+ api_type="anthropic", model="opus-4-7",
+ )
+ assert not native
+ # 'not found' lands in `text` (sections), not refusals, per implementation.
+ assert "not found" in text.lower()
+
+
+def test_resolve_attachments_handles_directory_path_not_file():
+ """A directory in context_paths gets dir-tree handling, not treated
+ as a file. Prevents trying to base64 a directory."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ tmpdir = tempfile.mkdtemp()
+ open(os.path.join(tmpdir, "a.txt"), "w").write("hello")
+ try:
+ text, native, refusals = mgr._resolve_attachments(
+ [{"path": tmpdir, "type": "directory"}],
+ api_type="anthropic", model="opus-4-7",
+ )
+ assert not native
+ assert "context_directory" in text
+ finally:
+ import shutil; shutil.rmtree(tmpdir)
+
+
+def test_resolve_attachments_mixed_kinds_total_size_guard():
+ """1 text + 1 PDF + 1 image attached together must respect both the
+ per-file caps AND the total-request-size cap as a single integrated
+ check, not three independent ones."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ paths = []
+ try:
+ # 10MB PDF + 10MB image + small text → 20MB raw = ~27MB base64,
+ # under Anthropic's 28MB cap so all should land natively.
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n"); fh.write(b"X" * (10 * 1024 * 1024))
+ paths.append(fh.name)
+ with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh:
+ fh.write(b"\x89PNG\r\n\x1a\n"); fh.write(b"X" * (10 * 1024 * 1024))
+ paths.append(fh.name)
+ with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as fh:
+ fh.write("# notes"); paths.append(fh.name)
+ text, native, refusals = mgr._resolve_attachments(
+ [{"path": p, "type": "file"} for p in paths],
+ api_type="anthropic", model="opus-4-7",
+ )
+ # All three should make it: PDF native, image native, text inline.
+ assert len(native) == 2
+ assert any(b["type"] == "document" for b in native)
+ assert any(b["type"] == "image" for b in native)
+ assert "notes" in text
+ assert not refusals
+ finally:
+ for p in paths:
+ try: os.unlink(p)
+ except Exception: pass
+
+
+def test_upload_dedup_handles_filename_collision_atomically():
+ """O_CREAT|O_EXCL must reserve the destination so two callers
+ racing on the same filename get distinct outputs, not one
+ overwriting the other."""
+ import os, tempfile, shutil
+ from backend.apps.settings.settings import UPLOAD_DIR
+ os.makedirs(UPLOAD_DIR, exist_ok=True)
+ name = f"test_dedup_{os.getpid()}.txt"
+ paths = []
+ try:
+ # Simulate two writers reserving the same base name back to back.
+ for _ in range(3):
+ base, ext = os.path.splitext(name)
+ dest = os.path.join(UPLOAD_DIR, name)
+ counter = 0
+ fd = None
+ while fd is None:
+ try:
+ fd = os.open(dest, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644)
+ except FileExistsError:
+ counter += 1
+ dest = os.path.join(UPLOAD_DIR, f"{base}_{counter}{ext}")
+ with os.fdopen(fd, "wb") as fh:
+ fh.write(b"hi")
+ paths.append(dest)
+ assert len(set(paths)) == 3
+ finally:
+ for p in paths:
+ try: os.remove(p)
+ except Exception: pass
+
+
+def test_sniff_recognises_macos_paths_with_spaces():
+ """File paths on macOS commonly contain spaces ('My Documents/file.pdf').
+ The sniffer reads contents, not the path, but agent_manager's
+ os.path.basename / open() must round-trip these correctly."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ tmpdir = tempfile.mkdtemp(prefix="space test ")
+ path = os.path.join(tmpdir, "my doc.pdf")
+ try:
+ with open(path, "wb") as f:
+ f.write(b"%PDF-1.4\n")
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ assert native and native[0]["type"] == "document"
+ assert not refusals
+ finally:
+ try: os.unlink(path)
+ except Exception: pass
+ try: os.rmdir(tmpdir)
+ except Exception: pass
+
+
+def test_resolve_attachments_uses_os_path_basename_for_windows_paths():
+ """When backend runs on Windows, paths arrive as C:\\Users\\X\\file.pdf.
+ os.path.basename handles backslash correctly on Windows (ntpath module),
+ but on POSIX (this test env) it treats backslash as a literal character.
+ Either way, the refusal copy embeds the result, so the test just verifies
+ no crash on Windows-shaped strings. Real Windows behavior is exercised
+ in CI on Windows hosts via .github/workflows/."""
+ import os, ntpath
+ # ntpath.basename simulates what Windows os.path.basename does on
+ # actual Windows hosts. Our backend uses os.path which == ntpath on
+ # Windows and posixpath on macOS/Linux, so paths go through correctly
+ # at runtime per host. This test asserts the parsing is correct WHEN
+ # routed through ntpath (the Windows code path).
+ win_path = r"C:\Users\rrios\AppData\Local\Temp\self-swarm-uploads\palm.pdf"
+ assert ntpath.basename(win_path) == "palm.pdf"
+ # And that os.path.join with mixed separators on Windows would still
+ # produce a valid path (ntpath is forgiving).
+ assert ntpath.basename(r"D:/Downloads\test.pdf") == "test.pdf"
+
+
+def test_sniff_file_kind_consistent_across_platforms():
+ """The sniffer reads bytes, never paths. So platform doesn't matter
+ for the classification logic — same bytes → same kind on Windows/Mac/Linux."""
+ from backend.apps.settings.settings import _sniff_file_kind
+ assert _sniff_file_kind(b"%PDF-1.4\n", "x.pdf") == ("pdf", "application/pdf")
+ assert _sniff_file_kind(b"\x89PNG\r\n\x1a\n", "x.png") == ("image", "image/png")
+ assert _sniff_file_kind(b"PK\x03\x04", "x.zip") == ("binary", None)
+ assert _sniff_file_kind(b"MZ\x90\x00", "x.exe") == ("binary", None)
+ assert _sniff_file_kind(b"hello world", "x.txt") == ("text", "text/plain")
+
+
+def test_estimate_pdf_tokens_consistent_across_platforms():
+ """Same byte-level math regardless of OS."""
+ from backend.apps.settings.settings import _estimate_pdf_tokens
+ # 5MB PDF should always estimate ≥ 5MB/80 = 65536 tokens.
+ fake = b"%PDF-1.4\n" + b"X" * (5 * 1024 * 1024)
+ assert _estimate_pdf_tokens(fake) >= 65000
+
+
+def test_sniff_handles_windows_style_backslash_path_string():
+ """Some Windows paths arrive at agent_manager with backslashes when
+ JSON-encoded or copied from Explorer. os.path.exists() handles
+ forward slashes on Windows but backslashes on POSIX would NOT find
+ the file. The basename() helper in the frontend already normalizes,
+ but verify the agent_manager refusal path is graceful."""
+ import os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ # A path that doesn't exist (POSIX cannot interpret backslashes as separator)
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": r"C:\fake\path\nope.pdf", "type": "file"}],
+ api_type="anthropic", model="opus-4-7",
+ )
+ assert not native
+ # Should produce a "not found" refusal, not crash.
+ assert any("not found" in s.lower() or "not found" in s for s in (_t, *refusals)) or "not found" in _t
+
+
+def test_upload_dir_writable_on_macos_temp():
+ """Audit: verify UPLOAD_DIR resolves to a writable path on this OS.
+ On macOS, tempfile.gettempdir() → /var/folders/... which is outside
+ the app sandbox restrictions; our entitlements don't grant explicit
+ temp access but it works due to standard process inheritance. On
+ Windows, tempfile → C:/Users/X/AppData/Local/Temp/ which is always
+ writable. Failure here would block every file attachment."""
+ import os
+ from backend.apps.settings.settings import UPLOAD_DIR
+ assert os.path.isdir(UPLOAD_DIR), f"UPLOAD_DIR not a directory: {UPLOAD_DIR}"
+ probe = os.path.join(UPLOAD_DIR, ".write_probe")
+ try:
+ with open(probe, "w") as f:
+ f.write("ok")
+ assert os.path.isfile(probe)
+ finally:
+ try: os.remove(probe)
+ except Exception: pass
+
+
+def test_resolve_attachments_classifies_renamed_binary_as_binary_not_pdf():
+ """A .pdf rename of a ZIP/PNG must NOT be inlined as a document
+ block; magic-byte sniff guards us."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"PK\x03\x04fake zip masquerading as pdf")
+ path = fh.name
+ try:
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ assert not native
+ assert refusals and "binary" in refusals[0].lower()
+ finally:
+ os.unlink(path)
+
+
+def test_gemini_proxy_rewrites_document_to_openai_image_url_for_9router():
+ """9router 0.3.60 only preserves `image_url` blocks (chunk 318 filter);
+ Anthropic-shape image/document blocks get stringified. We rewrite to
+ OpenAI image_url with data: URL so 9router emits Gemini inlineData."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini
+ body = json.dumps({
+ "model": "gemini-3.1-pro-preview",
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "summarize this"},
+ {"type": "document", "source": {
+ "type": "base64",
+ "media_type": "application/pdf",
+ "data": "JVBERi0xLjQK",
+ }},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_gemini(body))
+ blocks = out["messages"][0]["content"]
+ assert blocks[0]["type"] == "text"
+ assert blocks[1]["type"] == "image_url"
+ assert blocks[1]["image_url"]["url"] == "data:application/pdf;base64,JVBERi0xLjQK"
+
+
+def test_gemini_proxy_also_rewrites_anthropic_image_blocks_to_image_url():
+ """Same fix applies to plain images: Anthropic image → OpenAI image_url
+ with data: URL, so 9router's filter preserves it instead of stringifying."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini
+ body = json.dumps({
+ "model": "gemini-3-pro-preview",
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "image", "source": {
+ "type": "base64",
+ "media_type": "image/png",
+ "data": "iVBORw0KGgo=",
+ }},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_gemini(body))
+ block = out["messages"][0]["content"][0]
+ assert block["type"] == "image_url"
+ assert block["image_url"]["url"] == "data:image/png;base64,iVBORw0KGgo="
+
+
+def test_anthropic_document_block_schema_matches_docs():
+ """Schema-conformance: the document block our agent_manager emits for
+ Anthropic must structurally match the canonical shape from
+ https://docs.claude.com/en/docs/build-with-claude/pdf-support
+ (base64 inline). If Anthropic changes the schema we want a noisy test
+ failure here, not a runtime production failure."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%canonical schema test\n")
+ path = fh.name
+ try:
+ _t, native, _r = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ block = native[0]
+ # Per Anthropic docs, the exact required fields are:
+ assert set(block.keys()) >= {"type", "source"}
+ assert block["type"] == "document"
+ src = block["source"]
+ assert set(src.keys()) == {"type", "media_type", "data"}
+ assert src["type"] == "base64"
+ assert src["media_type"] == "application/pdf"
+ # cache_control is optional but our impl sets it on the last block
+ if "cache_control" in block:
+ assert block["cache_control"] == {"type": "ephemeral"}
+ # Base64 data must decode cleanly back to PDF magic header.
+ import base64 as _b64
+ decoded = _b64.b64decode(src["data"])
+ assert decoded.startswith(b"%PDF-")
+ finally:
+ os.unlink(path)
+
+
+def test_gemini_translated_block_matches_9router_image_url_filter():
+ """Per inspection of router/.next/server/chunks/318.js, 9router 0.3.60's
+ OpenAI→Gemini translator only handles `image_url` blocks with data: URLs
+ (it stringifies any other shape). Our translator must emit exactly that
+ shape for PDFs and images both."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini
+ body = json.dumps({
+ "model": "gemini-3.1-pro-preview",
+ "messages": [{"role": "user", "content": [
+ {"type": "document", "source": {
+ "type": "base64",
+ "media_type": "application/pdf",
+ "data": "JVBERi0xLjQK",
+ }},
+ ]}],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_gemini(body))
+ block = out["messages"][0]["content"][0]
+ assert block["type"] == "image_url"
+ assert "image_url" in block
+ assert block["image_url"]["url"].startswith("data:application/pdf;base64,")
+
+
+def test_openrouter_plugin_array_matches_docs():
+ """Per https://openrouter.ai/docs/features/multimodal/pdfs, the
+ plugins array shape is `[{id:"file-parser", pdf:{engine: "..."}}]`
+ at the top level. Engines: pdf-text (free, deprecated → cloudflare),
+ mistral-ocr ($2/1k pages), native (model-supported)."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser
+ body = json.dumps({
+ "model": "openrouter/qwen/qwen-2.5-72b-instruct",
+ "messages": [{"role": "user", "content": [
+ {"type": "document", "source": {
+ "type": "base64", "media_type": "application/pdf", "data": "x",
+ }},
+ ]}],
+ }).encode("utf-8")
+ out = json.loads(_inject_openrouter_file_parser(body))
+ plugins = out["plugins"]
+ assert isinstance(plugins, list)
+ fp = [p for p in plugins if p.get("id") == "file-parser"][0]
+ # Shape exactly matches https://openrouter.ai/docs/features/multimodal/pdfs
+ assert set(fp.keys()) == {"id", "pdf"}
+ assert isinstance(fp["pdf"], dict)
+ assert fp["pdf"]["engine"] in ("pdf-text", "mistral-ocr", "native")
+
+
+def test_openai_translated_image_block_matches_image_url_data_uri():
+ """OpenAI's image_url accepts data: URIs only for image/* mime types
+ (verified May 2026 — application/pdf returns HTTP 400). The
+ translator rewrites Anthropic image blocks; document blocks are
+ refused upstream in agent_manager."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5
+ body = json.dumps({
+ "model": "gpt-5.5",
+ "max_tokens": 100,
+ "messages": [{"role": "user", "content": [
+ {"type": "image", "source": {
+ "type": "base64",
+ "media_type": "image/png",
+ "data": "iVBORw0KGgo=",
+ }},
+ ]}],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_openai_gpt5(body))
+ block = out["messages"][0]["content"][0]
+ assert block["type"] == "image_url"
+ assert block["image_url"]["url"] == "data:image/png;base64,iVBORw0KGgo="
+
+
+def test_openai_proxy_rewrites_image_block_only_documents_pass_through():
+ """OpenAI image_url only accepts image/* mime; documents are refused
+ upstream. Translator handles images, leaves documents untouched."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5
+ body = json.dumps({
+ "model": "gpt-5.5",
+ "max_tokens": 500,
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "what's in this image?"},
+ {"type": "image", "source": {
+ "type": "base64",
+ "media_type": "image/png",
+ "data": "iVBORw0KGgo=",
+ }},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_openai_gpt5(body))
+ blocks = out["messages"][0]["content"]
+ assert blocks[0]["type"] == "text"
+ assert blocks[1]["type"] == "image_url"
+ assert blocks[1]["image_url"]["url"].startswith("data:image/png;base64,")
+ assert "max_completion_tokens" in out
+ assert "max_tokens" not in out
+
+
+def test_openai_proxy_skips_rewrite_when_no_document():
+ """Pure text turn on GPT-5 should only get the max_tokens rename."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5
+ body = json.dumps({
+ "model": "gpt-5.5",
+ "max_tokens": 100,
+ "messages": [{"role": "user", "content": "hi"}],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_openai_gpt5(body))
+ assert out["messages"][0]["content"] == "hi"
+ assert out.get("max_completion_tokens") == 100
+
+
+def test_openai_proxy_defensive_on_malformed_document_blocks():
+ """Malformed document blocks (missing source, missing data) pass
+ through untouched so the upstream returns a proper error rather
+ than us silently dropping the file."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_openai_gpt5
+ body = json.dumps({
+ "model": "gpt-5.5",
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "document"},
+ {"type": "document", "source": {"type": "url"}},
+ {"type": "document", "source": {"type": "base64"}},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_openai_gpt5(body))
+ for b in out["messages"][0]["content"]:
+ assert b["type"] == "document"
+
+
+def test_resolve_attachments_openai_codex_refused_for_pdfs():
+ """Codex variants refuse PDFs (both because Codex models don't read
+ PDFs AND because the OpenAI direct lane is currently disabled until
+ 9router translation lands)."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.3-codex",
+ )
+ assert not native
+ assert refusals
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_openai_codex_still_refuses_pdf():
+ """Codex variants don't support PDFs even though their OpenAI family
+ does; refusal should fire with switch hint."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.3-codex",
+ )
+ assert not native
+ assert refusals and "codex" in refusals[0].lower()
+ finally:
+ os.unlink(path)
+
+
+def test_openrouter_proxy_injects_file_parser_plugin_when_document_present():
+ """OR's universal-PDF feature requires top-level plugins:[{id:file-parser,...}].
+ When a document block is in the request bound for OR, inject it."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser
+ body = json.dumps({
+ "model": "openrouter/qwen/qwen-2.5-72b-instruct",
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "summarize"},
+ {"type": "document", "source": {
+ "type": "base64",
+ "media_type": "application/pdf",
+ "data": "JVBERi0xLjQK",
+ }},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_inject_openrouter_file_parser(body))
+ plugins = out.get("plugins")
+ assert isinstance(plugins, list) and len(plugins) >= 1
+ fp = next((p for p in plugins if p.get("id") == "file-parser"), None)
+ assert fp and fp["pdf"]["engine"] == "pdf-text"
+
+
+def test_openrouter_proxy_skips_plugin_when_no_document():
+ """No document block → don't inject the plugin (costs nothing, but
+ keeps the request body clean)."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser
+ body = json.dumps({
+ "model": "openrouter/qwen/qwen-2.5-72b-instruct",
+ "messages": [{"role": "user", "content": "just a question"}],
+ }).encode("utf-8")
+ out = json.loads(_inject_openrouter_file_parser(body))
+ assert "plugins" not in out
+
+
+def test_openrouter_proxy_dedupes_existing_file_parser_plugin():
+ """If a caller already provided file-parser, don't duplicate it."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _inject_openrouter_file_parser
+ body = json.dumps({
+ "model": "openrouter/qwen/qwen-2.5-72b-instruct",
+ "plugins": [{"id": "file-parser", "pdf": {"engine": "mistral-ocr"}}],
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "x"}},
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_inject_openrouter_file_parser(body))
+ fps = [p for p in out["plugins"] if p.get("id") == "file-parser"]
+ assert len(fps) == 1
+ assert fps[0]["pdf"]["engine"] == "mistral-ocr" # caller's engine wins
+
+
+def test_gemini_proxy_defensive_on_malformed_blocks():
+ """Bad shapes (missing data, wrong source.type, non-string data)
+ must NOT be rewritten; they pass through so the upstream sees the
+ error rather than a silently-corrupted block."""
+ import json
+ from backend.apps.agents.anthropic_proxy import _scrub_request_for_gemini
+ body = json.dumps({
+ "model": "gemini-3.1-pro-preview",
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "document"}, # no source
+ {"type": "document", "source": {}}, # empty source
+ {"type": "document", "source": {"type": "url"}}, # not base64
+ {"type": "document", "source": {"type": "base64"}}, # no data
+ ],
+ }],
+ }).encode("utf-8")
+ out = json.loads(_scrub_request_for_gemini(body))
+ for b in out["messages"][0]["content"]:
+ assert b["type"] == "document"
+
+
+def test_resolve_attachments_anthropic_emits_native_document():
+ """Anthropic upstream gets a `document` content block for PDFs, not
+ a text placeholder."""
+ import base64, tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ text, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ assert native and native[0]["type"] == "document"
+ assert native[0]["source"]["media_type"] == "application/pdf"
+ assert not refusals
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_openai_refuses_pdf_with_openrouter_hint():
+ """Empirical probe May 2026: OpenAI image_url rejects non-image mime
+ types with HTTP 400 'Invalid MIME type'. The type:file shape gets
+ stringified by 9router 0.3.60. Until we write a 9router-bypass
+ direct-API translator, refuse OpenAI PDFs with switch hint to
+ openrouter/openai/gpt-5 which has working PDF support via OR's
+ file-parser plugin."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ _text, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="openai", model="gpt-5.5",
+ )
+ assert not native
+ assert refusals
+ joined = " ".join(refusals).lower()
+ assert "openrouter" in joined or "claude" in joined
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_gemini_emits_native_document_after_translator_fix():
+ """After fixing the 9router 0.3.60 block-stripping bug via
+ anthropic_proxy._rewrite_document_to_image (now rewrites both
+ image AND document → OpenAI image_url with data: URL, which 9router
+ translates to Gemini inlineData), PDFs flow on Gemini natively."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ _text, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="gemini", model="gemini-3.1-pro-api",
+ )
+ assert native and native[0]["type"] == "document"
+ assert not refusals
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_text_file_inlined_not_native():
+ """Text files keep flowing through the existing context_file inline
+ path (no native block)."""
+ import tempfile, os
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".md", mode="w", delete=False) as fh:
+ fh.write("# hello\nworld")
+ path = fh.name
+ try:
+ text, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="opus-4-7", model="opus-4-7",
+ )
+ assert not native
+ assert not refusals
+ assert "hello" in text
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_pdf_refused_when_too_large():
+ """Anthropic's per-file cap blocks PDFs over 24MB."""
+ import os, tempfile
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n")
+ fh.write(b"X" * (25 * 1024 * 1024))
+ path = fh.name
+ try:
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ assert not native
+ assert refusals
+ assert "per-file cap" in refusals[0].lower() or "exceeds" in refusals[0].lower()
+ finally:
+ os.unlink(path)
+
+
+def test_resolve_attachments_refuses_when_total_exceeds_request_cap():
+ """4 medium PDFs that each pass the per-file cap should still be
+ blocked when their combined base64 size would exceed Anthropic's
+ 32MB request cap. This is the exact Mehmet scenario (30.3MB raw
+ of 4 PDFs base64 to ~40MB)."""
+ import os, tempfile
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ # 4 PDFs at ~8MB each = 32MB raw = ~43MB base64, exceeds 28MB cap.
+ paths = []
+ try:
+ for i in range(4):
+ with tempfile.NamedTemporaryFile(suffix=f"_{i}.pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n")
+ fh.write(b"X" * (8 * 1024 * 1024))
+ paths.append(fh.name)
+ _t, native, refusals = mgr._resolve_attachments(
+ [{"path": p, "type": "file"} for p in paths],
+ api_type="anthropic", model="opus-4-7",
+ )
+ # First few PDFs fit; later ones refused with "request over" message.
+ assert refusals, "expected refusals on multi-PDF over-cap"
+ assert any("encoded" in r.lower() and "provider cap" in r.lower() for r in refusals), \
+ f"expected total-size refusal copy; got: {refusals}"
+ finally:
+ for p in paths:
+ try: os.unlink(p)
+ except Exception: pass
+
+
+def test_resolve_attachments_anthropic_marks_last_document_ephemeral_for_cache():
+ """Anthropic prompt caching: the last document block gets
+ cache_control:ephemeral so multi-turn PDF chats stay cache-warm."""
+ import os, tempfile
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ paths = []
+ try:
+ for i in range(2):
+ with tempfile.NamedTemporaryFile(suffix=f"_{i}.pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ paths.append(fh.name)
+ _t, native, _r = mgr._resolve_attachments(
+ [{"path": p, "type": "file"} for p in paths],
+ api_type="anthropic", model="opus-4-7",
+ )
+ # Only the LAST document gets cache_control per Anthropic docs.
+ assert native[-1].get("cache_control") == {"type": "ephemeral"}
+ assert "cache_control" not in native[0]
+ finally:
+ for p in paths:
+ try: os.unlink(p)
+ except Exception: pass
+
+
+def test_resolve_attachments_anthropic_does_mark_ephemeral_but_only_anthropic():
+ """cache_control is Anthropic-only; don't pollute other-provider
+ blocks. Anthropic should get ephemeral on the last document block;
+ OpenRouter (which also supports PDFs) should NOT."""
+ import os, tempfile
+ from backend.apps.agents.agent_manager import AgentManager
+ mgr = AgentManager()
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as fh:
+ fh.write(b"%PDF-1.4\n%test\n")
+ path = fh.name
+ try:
+ _t, ant_native, _r = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="anthropic", model="opus-4-7",
+ )
+ assert ant_native and ant_native[0].get("cache_control") == {"type": "ephemeral"}
+ _t, or_native, _r = mgr._resolve_attachments(
+ [{"path": path, "type": "file"}], api_type="openrouter", model="openrouter/openai/gpt-5",
+ )
+ assert or_native and "cache_control" not in or_native[0]
+ finally:
+ os.unlink(path)
+
+
+def test_apply_context_window_respects_custom_provider_value():
+ """Custom OpenAI-compatible models supply their own context_window
+ via settings.custom_providers. _apply_context_window must look them
+ up the same way get_context_window does."""
+ from backend.apps.agents.models import AgentSession
+ from backend.apps.agents.agent_manager import _apply_context_window
+ from backend.apps.settings.models import AppSettings, CustomProvider
+ s = AgentSession(id="x", name="t", provider="custom", model="custom/ollama/qwen2.5:7b", mode="agent")
+ settings = AppSettings(custom_providers=[
+ CustomProvider(
+ name="Ollama",
+ base_url="http://localhost:11434/v1",
+ api_key="",
+ models=[{"value": "qwen2.5:7b", "label": "Qwen 2.5 7B", "context_window": 32_000}],
+ ),
+ ])
+ _apply_context_window(s, settings)
+ assert s.context_window == 32_000
+
+
# ---------------------------------------------------------------------------
# Custom OpenAI-compatible providers (Ollama Cloud, Together, etc.)
# ---------------------------------------------------------------------------
diff --git a/electron/package-lock.json b/electron/package-lock.json
index 7ec1e11c..8bfe1d54 100644
--- a/electron/package-lock.json
+++ b/electron/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "openswarm",
- "version": "1.1.40-exp.2",
+ "version": "1.1.41",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "openswarm",
- "version": "1.1.40-exp.2",
+ "version": "1.1.41",
"hasInstallScript": true,
"dependencies": {
"electron-updater": "^6.3.0",
diff --git a/frontend/src/app/components/DirectoryBrowser.tsx b/frontend/src/app/components/DirectoryBrowser.tsx
index 00294dec..3a45bf48 100644
--- a/frontend/src/app/components/DirectoryBrowser.tsx
+++ b/frontend/src/app/components/DirectoryBrowser.tsx
@@ -27,6 +27,9 @@ const SETTINGS_API = `${API_BASE}/settings`;
export interface ContextPath {
path: string;
type: 'file' | 'directory';
+ tokens?: number;
+ kind?: 'text' | 'pdf' | 'image' | 'binary';
+ media_type?: string;
}
interface DirectoryBrowserProps {
diff --git a/frontend/src/app/pages/AgentChat/AgentChat.tsx b/frontend/src/app/pages/AgentChat/AgentChat.tsx
index 2fc2516f..f19c90a9 100644
--- a/frontend/src/app/pages/AgentChat/AgentChat.tsx
+++ b/frontend/src/app/pages/AgentChat/AgentChat.tsx
@@ -56,8 +56,9 @@ import { setGlowingBrowserCards, fadeGlowingBrowserCards, clearGlowingBrowserCar
import { useClaudeTokens } from '@/shared/styles/ThemeContext';
const CONTEXT_WINDOWS: Record = {
- sonnet: 200_000,
- opus: 200_000,
+ 'opus-4-7': 1_000_000,
+ opus: 1_000_000,
+ sonnet: 1_000_000,
haiku: 200_000,
};
@@ -619,15 +620,22 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose
}, [id, dispatch, onBranch, session?.dashboard_id]);
const contextEstimate = useMemo(() => {
- // Look up the actual context window from the models store (backend
- // registry is the source of truth). Fall back to the legacy hardcoded
- // map for any model that isn't in the store yet.
+ // Prefer the live API-reported input token count once we have one
+ // (session.tokens.input includes the full request: messages + system +
+ // tool defs + cached prefix). That number is authoritative because
+ // Anthropic counts it against the context window. Before the first
+ // turn completes, fall back to a char/4 estimate of visible message
+ // content as a rough pre-send hint.
let limit = 0;
for (const ms of Object.values(modelsByProvider)) {
const hit = ms.find((m) => m.value === model);
if (hit?.context_window) { limit = hit.context_window; break; }
}
- if (!limit) limit = CONTEXT_WINDOWS[model] || 200_000;
+ if (!limit) limit = (session?.context_window) || CONTEXT_WINDOWS[model] || 200_000;
+ const liveInput = session?.tokens?.input ?? 0;
+ if (liveInput > 0) {
+ return { used: liveInput, limit };
+ }
let totalChars = 0;
if (session?.system_prompt) totalChars += session.system_prompt.length;
for (const msg of activeBranchMessages) {
@@ -640,7 +648,7 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose
// text and re-run this sum on every painted character, defeating
// the whole point of isolating AgentChat from delta updates. The
// header gauge will catch up when stream_end commits the message.
- }, [activeBranchMessages, session?.system_prompt, streamingMessageId, model, modelsByProvider]);
+ }, [activeBranchMessages, session?.system_prompt, session?.tokens?.input, session?.context_window, streamingMessageId, model, modelsByProvider]);
const sessionRunning = session?.status === 'running' || session?.status === 'waiting_approval';
@@ -926,16 +934,32 @@ const AgentChat: React.FC = ({ sessionId: sessionIdProp, onClose
);
})()}
{(() => {
- const pct = session.ctx_used_pct ?? 0;
- if (!pct) return null;
+ const liveWindow = session.context_window || contextEstimate.limit || 200_000;
+ const liveInput = session.tokens?.input ?? 0;
+ const pct = liveInput > 0
+ ? Math.min(1, liveInput / Math.max(1, liveWindow))
+ : (contextEstimate.used / Math.max(1, liveWindow));
+ if (pct <= 0) return null;
const pctTxt = `${Math.round(pct * 100)}%`;
- const color = pct >= 0.9 ? '#ef4444' : pct >= 0.7 ? '#f59e0b' : c.text.tertiary;
+ const color = pct >= 0.85 ? '#ef4444' : pct >= 0.60 ? '#f59e0b' : c.text.tertiary;
const mcpCount = session.active_mcps?.length ?? 0;
+ const fwOverhead = session.framework_overhead_tokens ?? 0;
+ const systemTokens = Math.round((session.system_prompt?.length ?? 0) / 4);
+ const historyTokens = Math.max(0, contextEstimate.used - systemTokens);
+ const fmt = (n: number) => n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n);
+ const breakdown = [
+ `Context ${pctTxt} of ${fmt(liveWindow)}`,
+ liveInput > 0 ? `Reported by API: ${fmt(liveInput)} input tokens` : null,
+ `History: ~${fmt(historyTokens)}`,
+ `System prompt: ~${fmt(systemTokens)}`,
+ fwOverhead > 0 ? `Tools + MCPs + preset: ~${fmt(fwOverhead)}` : null,
+ `${mcpCount} MCP${mcpCount === 1 ? '' : 's'} active`,
+ ].filter(Boolean).join(' · ');
return (
{pctTxt} ctx · {mcpCount} mcp
diff --git a/frontend/src/app/pages/AgentChat/ChatInput.tsx b/frontend/src/app/pages/AgentChat/ChatInput.tsx
index f49c03c0..cd81bce1 100644
--- a/frontend/src/app/pages/AgentChat/ChatInput.tsx
+++ b/frontend/src/app/pages/AgentChat/ChatInput.tsx
@@ -160,6 +160,20 @@ function formatTokenCount(n: number): string {
return String(n);
}
+// Path basename that works on both POSIX (/Users/x/file.pdf) and Windows
+// (C:\Users\x\file.pdf). Splits on either separator; falls back to the
+// raw path so empty segments don't yield ''.
+function basename(p: string): string {
+ if (!p) return '';
+ const parts = p.split(/[\\/]/).filter(Boolean);
+ return parts[parts.length - 1] || p;
+}
+function pathTail(p: string, n: number): string {
+ if (!p) return '';
+ const parts = p.split(/[\\/]/).filter(Boolean);
+ return parts.slice(-n).join('/');
+}
+
const ContextRing: React.FC<{ used: number; limit: number; accentColor: string; trackColor: string }> = ({ used, limit, accentColor, trackColor }) => {
if (used === 0) return null;
const pct = Math.min((used / limit) * 100, 100);
@@ -337,6 +351,9 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
const modelsLoaded = useAppSelector((state) => state.models.loaded);
const connectionMode = useAppSelector((state) => state.settings.data.connection_mode);
const toolItems = useAppSelector((state) => state.tools.items);
+ const sessionFrameworkOverhead = useAppSelector((state) =>
+ sessionId ? (state.agents.sessions[sessionId]?.framework_overhead_tokens ?? 0) : 0,
+ );
const allModelOptions = useMemo(() => {
@@ -644,6 +661,19 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
const [contextPaths, setContextPaths] = useState([]);
const [forcedTools, setForcedTools] = useState([]);
const [copiedPathIdx, setCopiedPathIdx] = useState(null);
+ const [oversizeQueue, setOversizeQueue] = useState>([]);
+ const [summarizingPath, setSummarizingPath] = useState(null);
+ const [summarizeError, setSummarizeError] = useState(null);
+ const [sendBlock, setSendBlock] = useState(null);
useImperativeHandle(ref, () => ({
getConfig: () => {
@@ -756,6 +786,43 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
});
}, []);
+ const currentModelCtx = useMemo(() => {
+ const m = allModelOptions.flat.find((x: any) => x.value === model) as any;
+ return (m?.context_window as number) || 200_000;
+ }, [allModelOptions.flat, model]);
+
+ const currentModelApi = useMemo(() => {
+ const m = allModelOptions.flat.find((x: any) => x.value === model) as any;
+ return ((m?.api as string) || 'anthropic').toLowerCase();
+ }, [allModelOptions.flat, model]);
+
+ // Mirrors backend agent_manager._resolve_attachments support matrix:
+ // PDFs route natively on Anthropic + Gemini (via anthropic-proxy
+ // document→image rewrite); refused on OpenAI/OpenRouter/custom until
+ // we land file-parser plugin / type:file translation.
+ // Mirrors backend agent_manager._resolve_attachments support matrix.
+ // PDFs: Anthropic, Gemini, OpenRouter (all empirically verified May
+ // 2026 via probe-pdf-roundtrip.py). OpenAI direct refused because
+ // OpenAI's image_url rejects non-image mime; user should switch to
+ // openrouter/openai/gpt-5 for OpenAI-via-OR with PDF support.
+ // Images: every provider via 9router image_url translation.
+ const pdfSupported = ['anthropic', 'gemini', 'gemini-cli', 'openrouter'].includes(currentModelApi);
+ const imageSupported = ['anthropic', 'gemini', 'gemini-cli', 'openai', 'openrouter'].includes(currentModelApi);
+
+ const pendingPayloadEstimate = useMemo(() => {
+ const history = Math.max(0, contextEstimate?.used ?? 0);
+ const filesSum = contextPaths.reduce((acc, cp) => acc + (cp.tokens || 0), 0);
+ return history + (sessionFrameworkOverhead || 0) + filesSum;
+ }, [contextEstimate, contextPaths, sessionFrameworkOverhead]);
+
+ const pendingKinds = useMemo(() => {
+ const set = new Set();
+ for (const cp of contextPaths) {
+ if (cp.kind) set.add(cp.kind);
+ }
+ return set;
+ }, [contextPaths]);
+
const uploadAndAttachFiles = useCallback(async (files: File[]) => {
if (files.length === 0) return;
setIsUploading(true);
@@ -768,25 +835,61 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
});
if (!resp.ok) throw new Error('Upload failed');
const data = await resp.json();
- const newPaths: ContextPath[] = (data.files || []).map((f: { path: string }) => ({
- path: f.path,
- type: 'file' as const,
- }));
+ const halfCap = Math.floor(currentModelCtx * 0.5);
+ const oversize: Array<{ path: string; name: string; tokens: number }> = [];
+ const newPaths: ContextPath[] = (data.files || []).map((f: { path: string; name?: string; tokens?: number; kind?: 'text' | 'pdf' | 'image' | 'binary'; media_type?: string }) => {
+ const t = typeof f.tokens === 'number' ? f.tokens : 0;
+ if (t > halfCap) oversize.push({ path: f.path, name: f.name || basename(f.path) || 'file', tokens: t });
+ return { path: f.path, type: 'file' as const, tokens: t, kind: f.kind, media_type: f.media_type };
+ });
setContextPaths((prev) => [...prev, ...newPaths]);
+ if (oversize.length > 0) setOversizeQueue((q) => [...q, ...oversize]);
} catch (err) {
console.error('File upload failed:', err);
} finally {
setIsUploading(false);
}
- }, []);
+ }, [currentModelCtx]);
const handleSend = useCallback(async () => {
const editor = editorRef.current;
if (!editor || disabled) return;
+ if (summarizingPath) return;
+ if (oversizeQueue.length > 0) return;
const serialized = serializeEditorContent(editor, attachedSkillsRef.current);
let trimmed = serialized.trim();
if (!trimmed) return;
+ // Pre-send dry-run guard. Sums every known component of next-turn input
+ // (history estimate from props, system prompt, framework/MCP overhead
+ // last reported by the API, attached file token estimates, and the
+ // prompt itself). If the sum exceeds 95% of the model's window, block
+ // the send and surface a banner with concrete recovery actions instead
+ // of round-tripping to a doomed API call. Conservative on purpose:
+ // tokenizers differ across providers (char/4 is rough), so we leave
+ // 5% headroom plus the API's own response budget.
+ {
+ const win = currentModelCtx;
+ const history = Math.max(0, contextEstimate?.used ?? 0);
+ const filesSum = contextPaths.reduce((acc, cp) => acc + (cp.tokens || 0), 0);
+ const promptTokens = Math.ceil(trimmed.length / 4);
+ const framework = sessionFrameworkOverhead || 0;
+ const systemTokens = 0;
+ const estimate = history + framework + filesSum + promptTokens + systemTokens;
+ if (win > 0 && estimate > Math.floor(win * 0.95)) {
+ let largest: { path: string; tokens: number } | undefined;
+ for (const cp of contextPaths) {
+ if ((cp.tokens || 0) > (largest?.tokens || 0)) largest = { path: cp.path, tokens: cp.tokens || 0 };
+ }
+ setSendBlock({
+ estimate, window: win,
+ history, system: systemTokens, framework, files: filesSum, prompt: promptTokens,
+ largestFile: largest,
+ });
+ return;
+ }
+ }
+
onboardingBus.emit('chat:message_sent');
if (window.location.hash.includes('/apps/')) {
onboardingBus.emit('app:generation_started');
@@ -1116,6 +1219,57 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
if (otherFiles.length > 0) uploadAndAttachFiles(otherFiles);
}, [addImageFiles, uploadAndAttachFiles]);
+ useEffect(() => {
+ const halfCap = Math.floor(currentModelCtx * 0.5);
+ const stillOversize: Array<{ path: string; name: string; tokens: number }> = [];
+ for (const cp of contextPaths) {
+ const t = cp.tokens || 0;
+ if (t > halfCap) {
+ const name = basename(cp.path) || cp.path;
+ stillOversize.push({ path: cp.path, name, tokens: t });
+ }
+ }
+ setOversizeQueue((q) => {
+ const next = stillOversize.filter((o) => !q.find((qq) => qq.path === o.path));
+ return [...q.filter((qq) => stillOversize.find((o) => o.path === qq.path)), ...next];
+ });
+ }, [currentModelCtx, contextPaths]);
+
+ const detachOversize = useCallback((path: string) => {
+ setContextPaths((prev) => prev.filter((cp) => cp.path !== path));
+ setOversizeQueue((q) => q.filter((o) => o.path !== path));
+ }, []);
+
+ const summarizeOversize = useCallback(async (path: string) => {
+ if (summarizingPath) return; // another summarize is in flight; ignore
+ setSummarizingPath(path);
+ try {
+ const tok = (() => { try { return getAuthToken(); } catch { return ''; } })();
+ const headers: Record = { 'Content-Type': 'application/json' };
+ if (tok) headers['Authorization'] = `Bearer ${tok}`;
+ const target = Math.min(8_000, Math.max(1_000, Math.floor(currentModelCtx * 0.05)));
+ const resp = await fetch(`${API_BASE}/settings/summarize-file`, {
+ method: 'POST', headers,
+ body: JSON.stringify({ path, target_tokens: target, primary_model: model }),
+ });
+ if (!resp.ok) {
+ let detail = `summarize failed (${resp.status})`;
+ try { const j = await resp.json(); if (j?.detail) detail = String(j.detail); } catch {}
+ throw new Error(detail);
+ }
+ const data = await resp.json();
+ const newPath: string = data.path;
+ const newTokens: number = data.tokens || 0;
+ setContextPaths((prev) => prev.map((cp) => cp.path === path ? { ...cp, path: newPath, tokens: newTokens, kind: 'text', media_type: 'text/plain' } : cp));
+ setOversizeQueue((q) => q.filter((o) => o.path !== path));
+ } catch (err) {
+ const msg = err instanceof Error ? err.message : 'summarize failed';
+ setSummarizeError(`${msg}. Detach the file or connect an aux provider in Settings.`);
+ } finally {
+ setSummarizingPath(null);
+ }
+ }, [currentModelCtx, model, summarizingPath]);
+
const removeImage = useCallback((idx: number) => {
setImages((prev) => {
const removed = prev[idx];
@@ -1219,6 +1373,82 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
visible={picker.visible}
/>
+ {sendBlock && (() => {
+ const fmt = (n: number) => n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n);
+ const over = sendBlock.estimate - sendBlock.window;
+ return (
+
+
+ This send would overflow the model's context window
+
+
+ ~{fmt(sendBlock.estimate)} of {fmt(sendBlock.window)} tokens ({over > 0 ? `${fmt(over)} over` : 'at cap'}). History {fmt(sendBlock.history)} · Files {fmt(sendBlock.files)} · Tools/MCPs {fmt(sendBlock.framework)} · This message {fmt(sendBlock.prompt)}.
+
+
+ {sessionId && (
+ {
+ try {
+ const tok = (() => { try { return getAuthToken(); } catch { return ''; } })();
+ const headers: Record = { 'Content-Type': 'application/json' };
+ if (tok) headers['Authorization'] = `Bearer ${tok}`;
+ await fetch(`${API_BASE}/agents/sessions/${sessionId}/compact`, { method: 'POST', headers });
+ setSendBlock(null);
+ } catch (err) { console.error(err); }
+ }}
+ sx={{
+ background: c.accent.primary, color: '#fff', border: 'none', borderRadius: '6px',
+ px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer', '&:hover': { opacity: 0.9 },
+ }}
+ >
+ Compact memory
+
+ )}
+ {sendBlock.largestFile && (
+ {
+ const p = sendBlock.largestFile!.path;
+ setContextPaths((prev) => prev.filter((cp) => cp.path !== p));
+ setSendBlock(null);
+ }}
+ sx={{
+ background: 'transparent', color: c.text.primary, border: `1px solid ${c.border.subtle}`,
+ borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer',
+ '&:hover': { background: c.bg.secondary },
+ }}
+ >
+ Detach largest file (~{fmt(sendBlock.largestFile.tokens)})
+
+ )}
+ { setModelAnchor(e.currentTarget as HTMLElement); setSendBlock(null); }}
+ sx={{
+ background: 'transparent', color: c.text.primary, border: `1px solid ${c.border.subtle}`,
+ borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer',
+ '&:hover': { background: c.bg.secondary },
+ }}
+ >
+ Switch model
+
+ setSendBlock(null)}
+ sx={{
+ background: 'transparent', color: c.text.muted, border: 'none',
+ borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer',
+ '&:hover': { background: c.bg.secondary },
+ }}
+ >
+ Dismiss
+
+
+
+ );
+ })()}
+
{images.length > 0 && (
(({ onSend, disabled, mode,
const isAppWorkspace = /\/outputs_workspace\/ws-[^/]+\/?$/.test(cp.path);
const label = isAppWorkspace
? 'App files'
- : cp.path.split('/').filter(Boolean).slice(-2).join('/');
+ : pathTail(cp.path, 2);
return (
(({ onSend, disabled, mode,
},
}}
>
+ {(() => {
+ const unsupported = (cp.kind === 'pdf' && !pdfSupported) ||
+ (cp.kind === 'image' && !imageSupported) ||
+ cp.kind === 'binary';
+ const chipColor = unsupported ? c.status.warning : c.accent.primary;
+ return (
:
}
- label={label}
+ label={(() => {
+ const kindTag = cp.kind && cp.kind !== 'text' ? ` · ${cp.kind}` : '';
+ const tokTag = typeof cp.tokens === 'number' && cp.tokens > 0 ? ` · ${formatTokenCount(cp.tokens)}` : '';
+ const warn = unsupported ? ' · not on this model' : '';
+ return `${label}${kindTag}${tokTag}${warn}`;
+ })()}
size="small"
onClick={() => {
navigator.clipboard.writeText(cp.path);
@@ -1315,21 +1556,23 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
}}
onDelete={() => setContextPaths((prev) => prev.filter((_, i) => i !== idx))}
sx={{
- bgcolor: `${c.accent.primary}12`,
- color: c.accent.primary,
+ bgcolor: `${chipColor}12`,
+ color: chipColor,
fontSize: '0.72rem',
fontFamily: c.font.mono,
height: 26,
- maxWidth: 220,
+ maxWidth: 280,
cursor: 'pointer',
'& .MuiChip-label': { overflow: 'hidden', textOverflow: 'ellipsis' },
'& .MuiChip-deleteIcon': {
- color: c.accent.primary,
+ color: chipColor,
fontSize: 16,
'&:hover': { color: c.status.error },
},
}}
/>
+ );
+ })()}
);
})}
@@ -2009,6 +2252,45 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
primary={highlightMatch(displayLabel)}
slotProps={{ primary: { sx: { fontSize: '0.8rem', color: model === opt.value ? c.text.primary : c.text.muted } } }}
/>
+ {(() => {
+ const win = (opt.context_window as number) || 0;
+ const api = (opt.api as string || 'anthropic').toLowerCase();
+ const optSupportsPdf = ['anthropic', 'gemini', 'gemini-cli', 'openrouter'].includes(api);
+ const optSupportsImage = ['anthropic', 'gemini', 'gemini-cli', 'openai', 'openrouter'].includes(api);
+ const cannotPdf = pendingKinds.has('pdf') && !optSupportsPdf;
+ const cannotImg = pendingKinds.has('image') && !optSupportsImage;
+ if (!win) return null;
+ const fits = pendingPayloadEstimate > 0 && win >= Math.floor(pendingPayloadEstimate * 1.1);
+ const tight = pendingPayloadEstimate > 0 && !fits && win >= pendingPayloadEstimate;
+ const tooSmall = pendingPayloadEstimate > 0 && win < pendingPayloadEstimate;
+ return (
+
+ {(cannotPdf || cannotImg) && (
+
+ No {cannotPdf ? 'PDF' : 'image'}
+
+ )}
+ {!cannotPdf && !cannotImg && fits && (
+
+ Fits
+
+ )}
+ {!cannotPdf && !cannotImg && tight && (
+
+ Tight
+
+ )}
+ {!cannotPdf && !cannotImg && tooSmall && (
+
+ Too small
+
+ )}
+
+ {formatTokenCount(win)}
+
+
+ );
+ })()}
);
@@ -2359,6 +2641,65 @@ const ChatInput = forwardRef(({ onSend, disabled, mode,
+ 0}
+ anchorOrigin={{ vertical: 'bottom', horizontal: 'center' }}
+ sx={{ mb: 10 }}
+ >
+
+ oversizeQueue[0] && summarizeOversize(oversizeQueue[0].path)}
+ sx={{
+ background: 'rgba(255,255,255,0.18)', color: 'inherit', border: 'none',
+ borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer',
+ '&:hover': { background: 'rgba(255,255,255,0.28)' },
+ '&:disabled': { opacity: 0.6, cursor: 'wait' },
+ }}
+ >
+ {summarizingPath === oversizeQueue[0]?.path ? 'Summarizing…' : 'Summarize instead'}
+
+ oversizeQueue[0] && detachOversize(oversizeQueue[0].path)}
+ sx={{
+ background: 'transparent', color: 'inherit', border: '1px solid rgba(255,255,255,0.4)',
+ borderRadius: '6px', px: 1, py: 0.5, fontSize: '0.72rem', cursor: 'pointer',
+ '&:hover': { background: 'rgba(255,255,255,0.12)' },
+ }}
+ >
+ Detach
+
+
+ }
+ >
+ {oversizeQueue[0] ? (
+
+ {oversizeQueue[0].name} is ~{formatTokenCount(oversizeQueue[0].tokens)} tokens, over 50% of this model's window ({formatTokenCount(currentModelCtx)}). Summarize sends the file content to your configured aux provider.
+
+ ) : null}
+
+
+
+ setSummarizeError(null)}
+ sx={{ mb: 18 }}
+ >
+ setSummarizeError(null)} sx={{ fontSize: '0.78rem', maxWidth: 520 }}>
+ {summarizeError}
+
+
+
);
});
diff --git a/frontend/src/app/pages/AgentChat/MessageBubble.tsx b/frontend/src/app/pages/AgentChat/MessageBubble.tsx
index 52bec3bd..94d27575 100644
--- a/frontend/src/app/pages/AgentChat/MessageBubble.tsx
+++ b/frontend/src/app/pages/AgentChat/MessageBubble.tsx
@@ -20,7 +20,7 @@ import ReactMarkdown from 'react-markdown';
import remarkGfm from 'remark-gfm';
import { AgentMessage } from '@/shared/state/agentsSlice';
import { openSettingsModal } from '@/shared/state/settingsSlice';
-import { useAppDispatch } from '@/shared/hooks';
+import { useAppDispatch, useAppSelector } from '@/shared/hooks';
import { useClaudeTokens } from '@/shared/styles/ThemeContext';
import { SKILL_COLOR } from '@/app/components/richEditorUtils';
import PlanPicker from '@/app/components/PlanPicker';
@@ -70,8 +70,23 @@ interface OpenSwarmErrorInfo {
ctaAction?: 'upgrade' | 'retry' | 'settings' | 'waitlist';
}
+interface OverflowContext {
+ model?: string;
+ contextWindow?: number;
+ inputTokens?: number;
+ frameworkOverhead?: number;
+ activeMcpCount?: number;
+ messagesCount?: number;
+}
+
+function formatTokens(n: number): string {
+ if (n >= 1_000_000) return `${(n / 1_000_000).toFixed(1)}M`;
+ if (n >= 1_000) return `${(n / 1_000).toFixed(1)}K`;
+ return String(n);
+}
+
/** Parses raw error text into a friendly card; returns null when the error isn't one we recognize. */
-function parseOpenSwarmError(text: string): OpenSwarmErrorInfo | null {
+function parseOpenSwarmError(text: string, ctx?: OverflowContext): OpenSwarmErrorInfo | null {
if (!text) return null;
if (/rate_limit_error|reached your OpenSwarm.*plan limit|Usage cap exceeded/i.test(text)) {
const reset = text.match(/Resets in ([\dhms\s]+)/)?.[1];
@@ -93,15 +108,48 @@ function parseOpenSwarmError(text: string): OpenSwarmErrorInfo | null {
};
}
if (/Prompt is too long|prompt_too_long|input length and `max_tokens`|context length/i.test(text)) {
+ const modelLower = (ctx?.model || '').toLowerCase();
+ const isHaiku = modelLower.includes('haiku');
+ const win = ctx?.contextWindow || 0;
+ const input = ctx?.inputTokens || 0;
+ const fw = ctx?.frameworkOverhead || 0;
+ const mcps = ctx?.activeMcpCount || 0;
+ if (isHaiku && mcps >= 5) {
+ return {
+ kind: 'too_many_tools',
+ title: 'Too many connected apps for Haiku',
+ detail:
+ `Haiku has the smallest memory of the Claude models${win ? ` (${formatTokens(win)} tokens)` : ''}. ` +
+ `Each of the ${mcps} active apps adds instructions Claude has to read before it can answer. ` +
+ 'Turn off a few apps (Microsoft 365 is the heaviest), or switch to Sonnet or Opus, both have 5x more room.',
+ ctaLabel: 'Open Settings',
+ ctaAction: 'settings',
+ };
+ }
+ let lead: string;
+ if (win && input) {
+ // input is the API-reported total which includes our preset, tool
+ // defs, MCP descriptions etc. Subtract those for the user-facing
+ // "your content" number so we don't blame the user for our overhead.
+ const userContent = Math.max(0, input - fw);
+ lead = `The request totalled ~${formatTokens(input)} of ${formatTokens(win)} tokens this model can hold (your messages + files: ~${formatTokens(userContent)}).`;
+ } else if (win) {
+ lead = `This model holds ${formatTokens(win)} tokens and the request exceeded that.`;
+ } else {
+ lead = 'The request exceeded this model\'s context window.';
+ }
+ const extras: string[] = [];
+ if (fw) extras.push(`built-in tools + system prompt ~${formatTokens(fw)}`);
+ if (mcps > 0) extras.push(`${mcps} active app${mcps === 1 ? '' : 's'}`);
+ const breakdown = extras.length > 0 ? ` Overhead from OpenSwarm: ${extras.join(', ')}.` : '';
return {
kind: 'too_many_tools',
- title: 'Too many connected apps for this model',
- detail:
- "Haiku is fast but has the smallest memory of the three Claude models. " +
- "Each connected app adds instructions Claude has to read before it can answer, " +
- "and you've added more than Haiku can hold in one go. Either turn off a few apps " +
- "(Microsoft 365 is the heaviest by far), or switch to Sonnet or Opus; both have " +
- "5x more room.",
+ title: 'This chat exceeded the model\'s context window',
+ detail: (
+ lead + breakdown +
+ ' Try detaching large files, running /compact to summarize older turns, ' +
+ 'starting a fresh chat, or switching to a model with a larger window.'
+ ),
ctaLabel: 'Open Settings',
ctaAction: 'settings',
};
@@ -248,7 +296,7 @@ function buildContextGroups(
color: '#10b981',
label,
chips: allPaths.map((cp) => {
- const name = cp.path.split('/').filter(Boolean).pop() || cp.path;
+ const name = cp.path.split(/[\\/]/).filter(Boolean).pop() || cp.path;
return {
label: name,
tooltip: cp.path,
@@ -848,7 +896,21 @@ const MessageBubble: React.FC = React.memo(({ message, editing = false, o
>{rawText}
), [rawText]);
- const openswarmError = !isUser ? parseOpenSwarmError(rawText) : null;
+ const overflowCtx = useAppSelector((state) => {
+ const sid = state.agents.activeSessionId;
+ if (!sid) return undefined;
+ const s = state.agents.sessions[sid];
+ if (!s) return undefined;
+ return {
+ model: s.model,
+ contextWindow: s.context_window,
+ inputTokens: s.tokens?.input,
+ frameworkOverhead: s.framework_overhead_tokens,
+ activeMcpCount: s.active_mcps?.length ?? 0,
+ messagesCount: s.messages?.length ?? 0,
+ } as OverflowContext;
+ });
+ const openswarmError = !isUser ? parseOpenSwarmError(rawText, overflowCtx) : null;
// (message.id, kind) keys so cap card analytics fire once, not on edits.
React.useEffect(() => {
diff --git a/frontend/src/shared/state/agentsSlice.ts b/frontend/src/shared/state/agentsSlice.ts
index 187a1625..9c5aa6c8 100644
--- a/frontend/src/shared/state/agentsSlice.ts
+++ b/frontend/src/shared/state/agentsSlice.ts
@@ -87,6 +87,8 @@ export interface AgentSession {
ctx_used_pct?: number;
cache_read_pct?: number;
cache_read_tokens?: number;
+ context_window?: number;
+ framework_overhead_tokens?: number;
context_overflow?: { reason: string; message: string; at: string } | null;
mcp_suggestions?: Array<{ id: string; title: string; description: string; reason?: string }>;
mcp_suggestions_is_vague?: boolean;
@@ -804,6 +806,8 @@ const agentsSlice = createSlice({
cacheReadTokens: number;
cacheReadPct: number;
ctxUsedPct: number;
+ contextWindow?: number;
+ frameworkOverheadTokens?: number;
activeMcps: string[];
}>
) {
@@ -817,6 +821,12 @@ const agentsSlice = createSlice({
session.cache_read_tokens = action.payload.cacheReadTokens;
session.cache_read_pct = action.payload.cacheReadPct;
session.ctx_used_pct = action.payload.ctxUsedPct;
+ if (typeof action.payload.contextWindow === 'number' && action.payload.contextWindow > 0) {
+ session.context_window = action.payload.contextWindow;
+ }
+ if (typeof action.payload.frameworkOverheadTokens === 'number') {
+ session.framework_overhead_tokens = action.payload.frameworkOverheadTokens;
+ }
session.active_mcps = action.payload.activeMcps;
}
},
diff --git a/frontend/src/shared/ws/WebSocketManager.ts b/frontend/src/shared/ws/WebSocketManager.ts
index d0c29796..315e8e03 100644
--- a/frontend/src/shared/ws/WebSocketManager.ts
+++ b/frontend/src/shared/ws/WebSocketManager.ts
@@ -578,6 +578,8 @@ class WebSocketManager {
cacheReadTokens: data.cache_read_tokens ?? 0,
cacheReadPct: data.cache_read_pct ?? 0,
ctxUsedPct: data.ctx_used_pct ?? 0,
+ contextWindow: typeof data.context_window === 'number' ? data.context_window : undefined,
+ frameworkOverheadTokens: typeof data.framework_overhead_tokens === 'number' ? data.framework_overhead_tokens : undefined,
activeMcps: Array.isArray(data.active_mcps) ? data.active_mcps : [],
}));
}
diff --git a/scripts/probe-pdf-roundtrip.py b/scripts/probe-pdf-roundtrip.py
new file mode 100755
index 00000000..9119bce4
--- /dev/null
+++ b/scripts/probe-pdf-roundtrip.py
@@ -0,0 +1,130 @@
+#!/usr/bin/env python3
+"""Empirical end-to-end PDF roundtrip probe.
+
+Verifies that an Anthropic-shape `document` content block flows through
+anthropic_proxy → 9router → upstream provider → response. Use one of
+the model presets below; the script picks the right backend lane.
+
+Usage:
+ 1. Configure the matching API key in OpenSwarm Settings (or env).
+ 2. Start the dev stack: bash run.sh
+ 3. python3 scripts/probe-pdf-roundtrip.py
+
+Providers:
+ anthropic direct Anthropic API key (claude-opus-4-7)
+ gemini direct Google AI Studio key (gemini-3-pro-preview)
+ openai direct OpenAI key (gpt-5.5)
+ openrouter OpenRouter free model with file-parser plugin (openrouter/openai/gpt-5)
+
+A 2xx response with non-empty content blocks confirms the full
+translator chain works for that provider. A 4xx with a provider-specific
+error message tells you exactly which step broke.
+"""
+import base64
+import json
+import os
+import sys
+import urllib.request
+import urllib.error
+
+# Each model_id below targets a SPECIFIC routing lane so a probe failure
+# pinpoints exactly which auth path is broken. "anthropic" uses the SDK
+# model_id form that lands on the cc/ OAuth subscription lane; if you
+# only have an Anthropic API key (no Pro subscription), use the explicit
+# direct-API model_id by editing PRESETS or pass --model.
+PRESETS = {
+ "anthropic": "claude-opus-4-7", # cc/ lane via 9router; needs Pro OAuth
+ "anthropic-api": "claude-opus-4-7", # same wire shape; routes by settings
+ "gemini": "gemini-3-pro-preview",
+ "openai": "gpt-5.5",
+ "openrouter": "openrouter/openai/gpt-5",
+}
+PROVIDER_CAPS_MB = {
+ "anthropic": 28,
+ "gemini": 14,
+ "openai": 45,
+ "openrouter": 45,
+}
+
+
+def probe(provider: str, pdf_path: str) -> int:
+ if provider not in PRESETS:
+ print(f"FAIL: unknown provider '{provider}'. Use one of: {', '.join(PRESETS)}", file=sys.stderr)
+ return 2
+ if not os.path.isfile(pdf_path):
+ print(f"FAIL: file not found: {pdf_path}", file=sys.stderr)
+ return 2
+ size = os.path.getsize(pdf_path)
+ cap = PROVIDER_CAPS_MB[provider] * 1024 * 1024
+ if size > cap:
+ print(f"FAIL: PDF is {size // (1024*1024)} MB, over {provider}'s {PROVIDER_CAPS_MB[provider]} MB inline cap.", file=sys.stderr)
+ return 2
+
+ with open(pdf_path, "rb") as fh:
+ data_b64 = base64.b64encode(fh.read()).decode("ascii")
+
+ token_path = os.path.join(
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
+ "backend", "data", "auth.token",
+ )
+ if not os.path.exists(token_path):
+ print(f"FAIL: auth.token not found at {token_path}. Start the dev backend first (bash run.sh).", file=sys.stderr)
+ return 2
+ with open(token_path) as fh:
+ token = fh.read().strip()
+
+ body = {
+ "model": PRESETS[provider],
+ "max_tokens": 300,
+ "messages": [{
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "In one sentence, what is this PDF about?"},
+ {"type": "document", "source": {
+ "type": "base64",
+ "media_type": "application/pdf",
+ "data": data_b64,
+ }},
+ ],
+ }],
+ }
+
+ req = urllib.request.Request(
+ "http://127.0.0.1:8324/api/anthropic-proxy/v1/messages",
+ data=json.dumps(body).encode("utf-8"),
+ headers={
+ "Content-Type": "application/json",
+ "x-api-key": token,
+ "anthropic-version": "2023-06-01",
+ },
+ method="POST",
+ )
+ try:
+ with urllib.request.urlopen(req, timeout=120) as resp:
+ status = resp.status
+ payload = resp.read(4096).decode("utf-8", errors="replace")
+ except urllib.error.HTTPError as e:
+ status = e.code
+ payload = e.read(4096).decode("utf-8", errors="replace")
+ except Exception as e:
+ print(f"FAIL: network error: {e}", file=sys.stderr)
+ return 3
+
+ print(f"=== {provider} ({PRESETS[provider]}) ===")
+ print(f"HTTP {status}")
+ print(payload[:800])
+ if 200 <= status < 300:
+ if any(k in payload.lower() for k in ("content", "text", "completion", "choice")):
+ print(f"\n✓ {provider} accepted the PDF document block end-to-end.")
+ return 0
+ print(f"\n⚠ HTTP 200 but no recognizable content; inspect response above.")
+ return 1
+ print(f"\n✗ {provider} PDF roundtrip failed at HTTP {status}. See response above.")
+ return 1
+
+
+if __name__ == "__main__":
+ if len(sys.argv) != 3:
+ print(f"usage: probe-pdf-roundtrip.py <{'|'.join(PRESETS)}> path/to/test.pdf", file=sys.stderr)
+ sys.exit(2)
+ sys.exit(probe(sys.argv[1], sys.argv[2]))