diff --git a/backend/apps/agents/agents.py b/backend/apps/agents/agents.py index 03717b6c..5b5bea72 100644 --- a/backend/apps/agents/agents.py +++ b/backend/apps/agents/agents.py @@ -757,7 +757,7 @@ async def list_models(): r["label"] += " (API key)" r["billing_kind"] = "api_key" r["is_free"] = False - # Models that only exist on the API-key route (Fable 5, whose sub route 404s on our pinned 9Router) have no adaptive twin to relabel, so add them or they vanish. + # Models that only exist on the API-key route have no adaptive twin to relabel, so add them or they vanish. adaptive_ids = {m.get("model_id") for m in adaptive} api_only = [m for m in api_variants if m.get("model_id") not in adaptive_ids] rows = p_serialize(api_only) + rows @@ -765,7 +765,7 @@ async def list_models(): # Only a sub: the adaptive rows route through 9router's cc/ lane, so they're covered by the subscription, not pay-per-use. for r in rows: r["billing_kind"] = "subscription" - # Sub-only models with no adaptive twin (Fable 5) won't ride the relabeled rows, so add their cc/ entry. + # Sub-only models with no adaptive twin won't ride the relabeled rows, so add their cc/ entry. adaptive_ids = {m.get("model_id") for m in adaptive} cc_only = [m for m in cc_variants if m.get("model_id") not in adaptive_ids] rows = p_serialize(cc_only) + rows diff --git a/backend/apps/agents/manager/run/run_options_helpers.py b/backend/apps/agents/manager/run/run_options_helpers.py index 54ec7078..3142f27a 100644 --- a/backend/apps/agents/manager/run/run_options_helpers.py +++ b/backend/apps/agents/manager/run/run_options_helpers.py @@ -216,9 +216,7 @@ def inject_thinking_options(options_kwargs: Dict, session: AgentSession, prompt: level = "off" if api_type == "anthropic": if level == "off": - # Fable 5 400s on an explicit thinking:disabled; off is its default (omit the param). - if not (isinstance(resolved_model, str) and "fable" in resolved_model): - options_kwargs["thinking"] = {"type": "disabled"} + options_kwargs["thinking"] = {"type": "disabled"} elif level in ("low", "medium", "high"): options_kwargs["effort"] = level elif api_type in ("openai", "codex"): diff --git a/backend/apps/agents/providers/pricing.py b/backend/apps/agents/providers/pricing.py index b6f274fb..a750f2ef 100644 --- a/backend/apps/agents/providers/pricing.py +++ b/backend/apps/agents/providers/pricing.py @@ -6,8 +6,6 @@ from __future__ import annotations # --------------------------------------------------------------------------- Curated model tiers; Intelligence, Speed, Cost on a 1-5 scale --------------------------------------------------------------------------- Hand-tuned from public benchmarks + per-token pricing (knowledge cutoff Jan 2026). The tier numbers serve the picker hover card so users can pick a model that fits the task without reading a leaderboard. Intelligence: 5 = frontier reasoner, 1 = nano / specialised tiny Speed: 5 = sub-second TTFT + 250 tok/s, 1 = slow + thinking Cost: 5 = $25+/M output, 1 = under $0.50/M output (or free) Lookup order (compute_tiers below): 1. Bare model_id direct 2. ":free" stripped (so anthropic/claude-opus-4.7:free shares scoring with anthropic/claude-opus-4.7) 3. Vendor-prefixed and bare-after-slash variants for cross-format coverage (so "claude-opus-4-7" matches "anthropic/claude-opus-4.7") 4. Last-path-component normalised (dashes ↔ dots) Models not in this map fall through to a heuristic that uses cost bucket + reasoning flag + name-keyword adjustments. (intelligence, speed, cost) on a 1-5 scale. Tiers: 5 frontier, 4 top open / strong sub, 3 solid mid, 2 small specialised, 1 nano. MODEL_TIERS: dict[str, tuple[int, int, int]] = { # Anthropic - "claude-fable-5": (5, 2, 5), - "anthropic/claude-fable-5": (5, 2, 5), "claude-opus-4-8": (5, 2, 5), "claude-opus-4.8": (5, 2, 5), "anthropic/claude-opus-4.8": (5, 2, 5), diff --git a/backend/apps/agents/providers/registry.py b/backend/apps/agents/providers/registry.py index ae89e774..3709b01f 100644 --- a/backend/apps/agents/providers/registry.py +++ b/backend/apps/agents/providers/registry.py @@ -71,10 +71,6 @@ BUILTIN_MODELS: dict[str, list[dict[str, Any]]] = { "model_id": "claude-haiku-4-5", "router_model_id": "cc/claude-haiku-4-5-20251001", "api": "anthropic", "reasoning": True, "route": "cc"}, # Fable 5 re-added 2026-07-02 after the ban lifted (Eric confirmed access is back); pull both rows again if it errors live. - {"value": "fable-5-cc", "label": "Claude Fable 5", "context_window": 1_000_000, - "model_id": "claude-fable-5", "router_model_id": "cc/claude-fable-5", "api": "anthropic", "reasoning": True, "route": "cc"}, - {"value": "fable-5-api", "label": "Claude Fable 5 (API key)", "context_window": 1_000_000, - "model_id": "claude-fable-5", "router_model_id": "claude-fable-5", "api": "anthropic", "reasoning": True, "route": "api"}, {"value": "opus-5-api", "label": "Claude Opus 5 (API key)", "context_window": 1_000_000, "model_id": "claude-opus-5", "router_model_id": "claude-opus-5", "api": "anthropic", "reasoning": True, "route": "api"}, {"value": "opus-4-8-api", "label": "Claude Opus 4.8 (API key)", "context_window": 1_000_000, @@ -263,9 +259,6 @@ def p_antigravity_connected() -> bool: def resolve_model_id_for_sdk(short_name: str, settings: AppSettings) -> str: """Short model name → id string for ClaudeAgentOptions.""" - # Free trial funds only Haiku via the cloud proxy; force it so a session left on a gpt-*/sub model can't escape to a lane the trial can't fund (which snags as a 401/404). - if getattr(settings, "connection_mode", "own_key") == "free-trial": - short_name = "haiku" entry = find_builtin_model(short_name) if entry is None: return short_name @@ -410,7 +403,6 @@ COST_PER_1M_TOKENS: dict[tuple[str, str], tuple[float, float]] = { ("Anthropic", "opus-4-7"): (5.0, 25.0), ("Anthropic", "opus-4-8"): (5.0, 25.0), ("Anthropic", "opus-5"): (5.0, 25.0), - ("Anthropic", "fable-5-api"): (10.0, 50.0), ("Anthropic", "haiku"): (1.0, 5.0), # OpenAI API-key rates (GPT-5.6 tiers, GA 2026-07-09) ("OpenAI", "gpt-5.6-api"): (5.0, 30.0), diff --git a/backend/apps/agents/providers/thinking_params_for.py b/backend/apps/agents/providers/thinking_params_for.py index 87736766..c7ecf5a6 100644 --- a/backend/apps/agents/providers/thinking_params_for.py +++ b/backend/apps/agents/providers/thinking_params_for.py @@ -20,9 +20,6 @@ def thinking_params_for(api: str, level: str, model_id: str = "") -> dict | None if level == "off": if api == "anthropic": - # Fable 5 400s on an explicit thinking:disabled; omit the param to turn thinking off (off is its default). Other Claude models accept it. - if "fable" in model_id: - return None return {"thinking": {"type": "disabled"}} if api == "codex": return {"reasoning": {"effort": "none"}} diff --git a/backend/apps/service/service.py b/backend/apps/service/service.py index 6ff1c36f..bec6b99f 100644 --- a/backend/apps/service/service.py +++ b/backend/apps/service/service.py @@ -285,7 +285,7 @@ def p_friendly_model(raw: str) -> str: base = (raw or "unknown").removesuffix("-cc") names = { "opus-5": "Claude Opus 5", "opus": "Claude Opus", "sonnet-5": "Claude Sonnet 5", - "sonnet": "Claude Sonnet", "haiku": "Claude Haiku", "fable-5": "Claude Fable 5", + "sonnet": "Claude Sonnet", "haiku": "Claude Haiku", } if base in names: return names[base] diff --git a/backend/apps/settings/store.py b/backend/apps/settings/store.py index 12fc6afb..29a5976e 100644 --- a/backend/apps/settings/store.py +++ b/backend/apps/settings/store.py @@ -78,6 +78,12 @@ def migrate_legacy_fields(raw: dict) -> dict: raw["connection_mode"] = "openswarm-pro" if "openswarm_auth_token" in raw and "openswarm_bearer_token" not in raw: raw["openswarm_bearer_token"] = raw.pop("openswarm_auth_token") + # The free tier is retired. Nothing arms it any more, but installs that already carry the mode + # would otherwise keep it forever, and with it the silent mid-run pin to Haiku. Move them to + # own_key so the model they picked is the model they get. + if raw.get("connection_mode") == "free-trial": + raw["connection_mode"] = "own_key" + raw.pop("free_trial_token", None) return raw diff --git a/backend/tests/test_free_tier_retired.py b/backend/tests/test_free_tier_retired.py new file mode 100644 index 00000000..b9f6ea09 --- /dev/null +++ b/backend/tests/test_free_tier_retired.py @@ -0,0 +1,45 @@ +"""The free tier is retired. It was never something a user picked: the app minted it at boot for +anyone with no key, and while armed it silently pinned every session to Haiku mid-run, so people ran +a different model than the one they chose with no indication. These pin the removal in all three +places it has to hold: nobody new gets in, nobody already in stays, and the pin itself is gone.""" + +from backend.apps.settings.store import migrate_legacy_fields +from backend.apps.settings.models import AppSettings +from backend.apps.agents.providers.registry import resolve_model_id_for_sdk + + +def test_an_install_already_on_the_free_tier_is_moved_off_it(): + raw = migrate_legacy_fields({"connection_mode": "free-trial", "free_trial_token": "tok-abc"}) + assert raw["connection_mode"] == "own_key" + assert "free_trial_token" not in raw, "a retired tier must not leave its credential behind" + + +def test_the_migration_leaves_every_other_mode_alone(): + for mode in ("own_key", "openswarm-pro", "custom"): + assert migrate_legacy_fields({"connection_mode": mode})["connection_mode"] == mode + + +def test_the_chosen_model_is_the_model_that_runs(): + # The old behavior rewrote short_name to "haiku" whenever the mode was free-trial, which is the + # reported "swapped me to a different model halfway through" with no indication. + s = AppSettings(connection_mode="own_key") + assert "opus" in resolve_model_id_for_sdk("opus-5", s).lower() + assert "sonnet" in resolve_model_id_for_sdk("sonnet-5", s).lower() + + +def test_no_haiku_pin_survives_even_if_the_retired_mode_is_forced_in(): + # Belt over the migration: hand the resolver the retired mode directly and it must still honor + # the caller's model rather than reaching for Haiku. + s = AppSettings() + object.__setattr__(s, "__dict__", {**s.__dict__, "connection_mode": "free-trial"}) + assert "haiku" not in resolve_model_id_for_sdk("opus-5", s).lower() + + +def test_nothing_in_the_app_mints_a_free_trial_on_boot(): + """The mint was fired unconditionally from the renderer's boot effect, which is why the tier was + still shipping long after the decision to drop it. Guard the renderer, not just the backend.""" + from pathlib import Path + main_tsx = Path(__file__).resolve().parents[2] / "frontend" / "src" / "app" / "Main.tsx" + assert "free-trial/mint" not in main_tsx.read_text(encoding="utf-8"), ( + "Main.tsx arms the retired free tier at boot again" + ) diff --git a/backend/tests/test_v2_invariants.py b/backend/tests/test_v2_invariants.py index 5d8c463d..dc74782c 100644 --- a/backend/tests/test_v2_invariants.py +++ b/backend/tests/test_v2_invariants.py @@ -668,16 +668,19 @@ def test_banned_models_not_offered(): can't serve it, AI Studio key 429s pro-preview. (gpt-5.5's cx entry left this list 2026-07-26: the pinned 0.3.60's old 404 healed upstream and a live probe returned a real completion, so both its lanes work. Fable 5 - left 2026-07-02 after its ban lifted.)""" + left 2026-07-02 after its ban lifted, and came BACK 2026-08-12: it was + selectable and simply did not work, and Opus covers the same ground.)""" from backend.apps.agents.providers.registry import BUILTIN_MODELS all_values = {m["value"] for models in BUILTIN_MODELS.values() for m in models} - for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api"): + for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api", + "fable-5-cc", "fable-5-api"): assert dead not in all_values, f"{dead} is back in the picker" assert "gpt-5.5-api" in all_values assert "gpt-5.5" in all_values # cx lane restored 2026-07-26 (live-probed) # No '3.1 pro' label survives in any provider group either. all_labels = " | ".join(m["label"].lower() for models in BUILTIN_MODELS.values() for m in models) assert "3.1 pro" not in all_labels + assert "fable" not in all_labels assert "gemini 3 flash" not in all_labels diff --git a/frontend/src/app/Main.tsx b/frontend/src/app/Main.tsx index a807efbc..fd59eaf4 100644 --- a/frontend/src/app/Main.tsx +++ b/frontend/src/app/Main.tsx @@ -246,11 +246,10 @@ const SettingsLoader: React.FC<{ children: React.ReactNode }> = ({ children }) = }) .catch(() => {}) .finally(() => { - // Arm the zero-config free trial when nothing is connected so a brand-new user can run an agent immediately. The backend no-ops if a real key or subscription exists, so this is safe to fire on every launch. - fetch(`${API_BASE}/subscription/free-trial/mint`, { method: 'POST' }) - .catch(() => {}) - // The backend arms server-side regardless of whether the browser can read the mint response (a transient boot-time CORS/timing miss makes `data` unreadable), so refetch unconditionally, the GET is the only reliable signal the UI gets that it armed. - .finally(() => { dispatch(fetchSettings()); dispatch(fetchSubscriptionStatus()); dispatch(markFreeTrialArmSettled()); }); + // The free tier is retired. Nothing arms it any more, so there is no mint on boot; the flag + // still settles so the "connect a model" banner is not held back waiting for a call that + // will never happen. + dispatch(markFreeTrialArmSettled()); }); return () => { if (bootTimer) clearInterval(bootTimer);