[eric] models: retire the free tier (nothing mints it, existing installs migrate off) and drop the silent mid-run pin to Haiku it carried; pull Fable, which was selectable and never worked

This commit is contained in:
ciregenz
2026-08-11 21:00:38 -07:00
parent 6214e92a0c
commit 62e707293e
10 changed files with 64 additions and 26 deletions
+2 -2
View File
@@ -757,7 +757,7 @@ async def list_models():
r["label"] += " (API key)"
r["billing_kind"] = "api_key"
r["is_free"] = False
# Models that only exist on the API-key route (Fable 5, whose sub route 404s on our pinned 9Router) have no adaptive twin to relabel, so add them or they vanish.
# Models that only exist on the API-key route have no adaptive twin to relabel, so add them or they vanish.
adaptive_ids = {m.get("model_id") for m in adaptive}
api_only = [m for m in api_variants if m.get("model_id") not in adaptive_ids]
rows = p_serialize(api_only) + rows
@@ -765,7 +765,7 @@ async def list_models():
# Only a sub: the adaptive rows route through 9router's cc/ lane, so they're covered by the subscription, not pay-per-use.
for r in rows:
r["billing_kind"] = "subscription"
# Sub-only models with no adaptive twin (Fable 5) won't ride the relabeled rows, so add their cc/ entry.
# Sub-only models with no adaptive twin won't ride the relabeled rows, so add their cc/ entry.
adaptive_ids = {m.get("model_id") for m in adaptive}
cc_only = [m for m in cc_variants if m.get("model_id") not in adaptive_ids]
rows = p_serialize(cc_only) + rows
@@ -216,9 +216,7 @@ def inject_thinking_options(options_kwargs: Dict, session: AgentSession, prompt:
level = "off"
if api_type == "anthropic":
if level == "off":
# Fable 5 400s on an explicit thinking:disabled; off is its default (omit the param).
if not (isinstance(resolved_model, str) and "fable" in resolved_model):
options_kwargs["thinking"] = {"type": "disabled"}
options_kwargs["thinking"] = {"type": "disabled"}
elif level in ("low", "medium", "high"):
options_kwargs["effort"] = level
elif api_type in ("openai", "codex"):
-2
View File
@@ -6,8 +6,6 @@ from __future__ import annotations
# --------------------------------------------------------------------------- Curated model tiers; Intelligence, Speed, Cost on a 1-5 scale --------------------------------------------------------------------------- Hand-tuned from public benchmarks + per-token pricing (knowledge cutoff Jan 2026). The tier numbers serve the picker hover card so users can pick a model that fits the task without reading a leaderboard. Intelligence: 5 = frontier reasoner, 1 = nano / specialised tiny Speed: 5 = sub-second TTFT + 250 tok/s, 1 = slow + thinking Cost: 5 = $25+/M output, 1 = under $0.50/M output (or free) Lookup order (compute_tiers below): 1. Bare model_id direct 2. ":free" stripped (so anthropic/claude-opus-4.7:free shares scoring with anthropic/claude-opus-4.7) 3. Vendor-prefixed and bare-after-slash variants for cross-format coverage (so "claude-opus-4-7" matches "anthropic/claude-opus-4.7") 4. Last-path-component normalised (dashes ↔ dots) Models not in this map fall through to a heuristic that uses cost bucket + reasoning flag + name-keyword adjustments. (intelligence, speed, cost) on a 1-5 scale. Tiers: 5 frontier, 4 top open / strong sub, 3 solid mid, 2 small specialised, 1 nano.
MODEL_TIERS: dict[str, tuple[int, int, int]] = {
# Anthropic
"claude-fable-5": (5, 2, 5),
"anthropic/claude-fable-5": (5, 2, 5),
"claude-opus-4-8": (5, 2, 5),
"claude-opus-4.8": (5, 2, 5),
"anthropic/claude-opus-4.8": (5, 2, 5),
@@ -71,10 +71,6 @@ BUILTIN_MODELS: dict[str, list[dict[str, Any]]] = {
"model_id": "claude-haiku-4-5", "router_model_id": "cc/claude-haiku-4-5-20251001", "api": "anthropic", "reasoning": True, "route": "cc"},
# Fable 5 re-added 2026-07-02 after the ban lifted (Eric confirmed access is back); pull both rows again if it errors live.
{"value": "fable-5-cc", "label": "Claude Fable 5", "context_window": 1_000_000,
"model_id": "claude-fable-5", "router_model_id": "cc/claude-fable-5", "api": "anthropic", "reasoning": True, "route": "cc"},
{"value": "fable-5-api", "label": "Claude Fable 5 (API key)", "context_window": 1_000_000,
"model_id": "claude-fable-5", "router_model_id": "claude-fable-5", "api": "anthropic", "reasoning": True, "route": "api"},
{"value": "opus-5-api", "label": "Claude Opus 5 (API key)", "context_window": 1_000_000,
"model_id": "claude-opus-5", "router_model_id": "claude-opus-5", "api": "anthropic", "reasoning": True, "route": "api"},
{"value": "opus-4-8-api", "label": "Claude Opus 4.8 (API key)", "context_window": 1_000_000,
@@ -263,9 +259,6 @@ def p_antigravity_connected() -> bool:
def resolve_model_id_for_sdk(short_name: str, settings: AppSettings) -> str:
"""Short model name → id string for ClaudeAgentOptions."""
# Free trial funds only Haiku via the cloud proxy; force it so a session left on a gpt-*/sub model can't escape to a lane the trial can't fund (which snags as a 401/404).
if getattr(settings, "connection_mode", "own_key") == "free-trial":
short_name = "haiku"
entry = find_builtin_model(short_name)
if entry is None:
return short_name
@@ -410,7 +403,6 @@ COST_PER_1M_TOKENS: dict[tuple[str, str], tuple[float, float]] = {
("Anthropic", "opus-4-7"): (5.0, 25.0),
("Anthropic", "opus-4-8"): (5.0, 25.0),
("Anthropic", "opus-5"): (5.0, 25.0),
("Anthropic", "fable-5-api"): (10.0, 50.0),
("Anthropic", "haiku"): (1.0, 5.0),
# OpenAI API-key rates (GPT-5.6 tiers, GA 2026-07-09)
("OpenAI", "gpt-5.6-api"): (5.0, 30.0),
@@ -20,9 +20,6 @@ def thinking_params_for(api: str, level: str, model_id: str = "") -> dict | None
if level == "off":
if api == "anthropic":
# Fable 5 400s on an explicit thinking:disabled; omit the param to turn thinking off (off is its default). Other Claude models accept it.
if "fable" in model_id:
return None
return {"thinking": {"type": "disabled"}}
if api == "codex":
return {"reasoning": {"effort": "none"}}
+1 -1
View File
@@ -285,7 +285,7 @@ def p_friendly_model(raw: str) -> str:
base = (raw or "unknown").removesuffix("-cc")
names = {
"opus-5": "Claude Opus 5", "opus": "Claude Opus", "sonnet-5": "Claude Sonnet 5",
"sonnet": "Claude Sonnet", "haiku": "Claude Haiku", "fable-5": "Claude Fable 5",
"sonnet": "Claude Sonnet", "haiku": "Claude Haiku",
}
if base in names:
return names[base]
+6
View File
@@ -78,6 +78,12 @@ def migrate_legacy_fields(raw: dict) -> dict:
raw["connection_mode"] = "openswarm-pro"
if "openswarm_auth_token" in raw and "openswarm_bearer_token" not in raw:
raw["openswarm_bearer_token"] = raw.pop("openswarm_auth_token")
# The free tier is retired. Nothing arms it any more, but installs that already carry the mode
# would otherwise keep it forever, and with it the silent mid-run pin to Haiku. Move them to
# own_key so the model they picked is the model they get.
if raw.get("connection_mode") == "free-trial":
raw["connection_mode"] = "own_key"
raw.pop("free_trial_token", None)
return raw
+45
View File
@@ -0,0 +1,45 @@
"""The free tier is retired. It was never something a user picked: the app minted it at boot for
anyone with no key, and while armed it silently pinned every session to Haiku mid-run, so people ran
a different model than the one they chose with no indication. These pin the removal in all three
places it has to hold: nobody new gets in, nobody already in stays, and the pin itself is gone."""
from backend.apps.settings.store import migrate_legacy_fields
from backend.apps.settings.models import AppSettings
from backend.apps.agents.providers.registry import resolve_model_id_for_sdk
def test_an_install_already_on_the_free_tier_is_moved_off_it():
raw = migrate_legacy_fields({"connection_mode": "free-trial", "free_trial_token": "tok-abc"})
assert raw["connection_mode"] == "own_key"
assert "free_trial_token" not in raw, "a retired tier must not leave its credential behind"
def test_the_migration_leaves_every_other_mode_alone():
for mode in ("own_key", "openswarm-pro", "custom"):
assert migrate_legacy_fields({"connection_mode": mode})["connection_mode"] == mode
def test_the_chosen_model_is_the_model_that_runs():
# The old behavior rewrote short_name to "haiku" whenever the mode was free-trial, which is the
# reported "swapped me to a different model halfway through" with no indication.
s = AppSettings(connection_mode="own_key")
assert "opus" in resolve_model_id_for_sdk("opus-5", s).lower()
assert "sonnet" in resolve_model_id_for_sdk("sonnet-5", s).lower()
def test_no_haiku_pin_survives_even_if_the_retired_mode_is_forced_in():
# Belt over the migration: hand the resolver the retired mode directly and it must still honor
# the caller's model rather than reaching for Haiku.
s = AppSettings()
object.__setattr__(s, "__dict__", {**s.__dict__, "connection_mode": "free-trial"})
assert "haiku" not in resolve_model_id_for_sdk("opus-5", s).lower()
def test_nothing_in_the_app_mints_a_free_trial_on_boot():
"""The mint was fired unconditionally from the renderer's boot effect, which is why the tier was
still shipping long after the decision to drop it. Guard the renderer, not just the backend."""
from pathlib import Path
main_tsx = Path(__file__).resolve().parents[2] / "frontend" / "src" / "app" / "Main.tsx"
assert "free-trial/mint" not in main_tsx.read_text(encoding="utf-8"), (
"Main.tsx arms the retired free tier at boot again"
)
+5 -2
View File
@@ -668,16 +668,19 @@ def test_banned_models_not_offered():
can't serve it, AI Studio key 429s pro-preview. (gpt-5.5's cx entry left
this list 2026-07-26: the pinned 0.3.60's old 404 healed upstream and a
live probe returned a real completion, so both its lanes work. Fable 5
left 2026-07-02 after its ban lifted.)"""
left 2026-07-02 after its ban lifted, and came BACK 2026-08-12: it was
selectable and simply did not work, and Opus covers the same ground.)"""
from backend.apps.agents.providers.registry import BUILTIN_MODELS
all_values = {m["value"] for models in BUILTIN_MODELS.values() for m in models}
for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api"):
for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api",
"fable-5-cc", "fable-5-api"):
assert dead not in all_values, f"{dead} is back in the picker"
assert "gpt-5.5-api" in all_values
assert "gpt-5.5" in all_values # cx lane restored 2026-07-26 (live-probed)
# No '3.1 pro' label survives in any provider group either.
all_labels = " | ".join(m["label"].lower() for models in BUILTIN_MODELS.values() for m in models)
assert "3.1 pro" not in all_labels
assert "fable" not in all_labels
assert "gemini 3 flash" not in all_labels
+4 -5
View File
@@ -246,11 +246,10 @@ const SettingsLoader: React.FC<{ children: React.ReactNode }> = ({ children }) =
})
.catch(() => {})
.finally(() => {
// Arm the zero-config free trial when nothing is connected so a brand-new user can run an agent immediately. The backend no-ops if a real key or subscription exists, so this is safe to fire on every launch.
fetch(`${API_BASE}/subscription/free-trial/mint`, { method: 'POST' })
.catch(() => {})
// The backend arms server-side regardless of whether the browser can read the mint response (a transient boot-time CORS/timing miss makes `data` unreadable), so refetch unconditionally, the GET is the only reliable signal the UI gets that it armed.
.finally(() => { dispatch(fetchSettings()); dispatch(fetchSubscriptionStatus()); dispatch(markFreeTrialArmSettled()); });
// The free tier is retired. Nothing arms it any more, so there is no mint on boot; the flag
// still settles so the "connect a model" banner is not held back waiting for a call that
// will never happen.
dispatch(markFreeTrialArmSettled());
});
return () => {
if (bootTimer) clearInterval(bootTimer);