mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-10-01 05:54:56 +02:00
[eric] models: retire the free tier (nothing mints it, existing installs migrate off) and drop the silent mid-run pin to Haiku it carried; pull Fable, which was selectable and never worked
This commit is contained in:
@@ -757,7 +757,7 @@ async def list_models():
|
||||
r["label"] += " (API key)"
|
||||
r["billing_kind"] = "api_key"
|
||||
r["is_free"] = False
|
||||
# Models that only exist on the API-key route (Fable 5, whose sub route 404s on our pinned 9Router) have no adaptive twin to relabel, so add them or they vanish.
|
||||
# Models that only exist on the API-key route have no adaptive twin to relabel, so add them or they vanish.
|
||||
adaptive_ids = {m.get("model_id") for m in adaptive}
|
||||
api_only = [m for m in api_variants if m.get("model_id") not in adaptive_ids]
|
||||
rows = p_serialize(api_only) + rows
|
||||
@@ -765,7 +765,7 @@ async def list_models():
|
||||
# Only a sub: the adaptive rows route through 9router's cc/ lane, so they're covered by the subscription, not pay-per-use.
|
||||
for r in rows:
|
||||
r["billing_kind"] = "subscription"
|
||||
# Sub-only models with no adaptive twin (Fable 5) won't ride the relabeled rows, so add their cc/ entry.
|
||||
# Sub-only models with no adaptive twin won't ride the relabeled rows, so add their cc/ entry.
|
||||
adaptive_ids = {m.get("model_id") for m in adaptive}
|
||||
cc_only = [m for m in cc_variants if m.get("model_id") not in adaptive_ids]
|
||||
rows = p_serialize(cc_only) + rows
|
||||
|
||||
@@ -216,9 +216,7 @@ def inject_thinking_options(options_kwargs: Dict, session: AgentSession, prompt:
|
||||
level = "off"
|
||||
if api_type == "anthropic":
|
||||
if level == "off":
|
||||
# Fable 5 400s on an explicit thinking:disabled; off is its default (omit the param).
|
||||
if not (isinstance(resolved_model, str) and "fable" in resolved_model):
|
||||
options_kwargs["thinking"] = {"type": "disabled"}
|
||||
options_kwargs["thinking"] = {"type": "disabled"}
|
||||
elif level in ("low", "medium", "high"):
|
||||
options_kwargs["effort"] = level
|
||||
elif api_type in ("openai", "codex"):
|
||||
|
||||
@@ -6,8 +6,6 @@ from __future__ import annotations
|
||||
# --------------------------------------------------------------------------- Curated model tiers; Intelligence, Speed, Cost on a 1-5 scale --------------------------------------------------------------------------- Hand-tuned from public benchmarks + per-token pricing (knowledge cutoff Jan 2026). The tier numbers serve the picker hover card so users can pick a model that fits the task without reading a leaderboard. Intelligence: 5 = frontier reasoner, 1 = nano / specialised tiny Speed: 5 = sub-second TTFT + 250 tok/s, 1 = slow + thinking Cost: 5 = $25+/M output, 1 = under $0.50/M output (or free) Lookup order (compute_tiers below): 1. Bare model_id direct 2. ":free" stripped (so anthropic/claude-opus-4.7:free shares scoring with anthropic/claude-opus-4.7) 3. Vendor-prefixed and bare-after-slash variants for cross-format coverage (so "claude-opus-4-7" matches "anthropic/claude-opus-4.7") 4. Last-path-component normalised (dashes ↔ dots) Models not in this map fall through to a heuristic that uses cost bucket + reasoning flag + name-keyword adjustments. (intelligence, speed, cost) on a 1-5 scale. Tiers: 5 frontier, 4 top open / strong sub, 3 solid mid, 2 small specialised, 1 nano.
|
||||
MODEL_TIERS: dict[str, tuple[int, int, int]] = {
|
||||
# Anthropic
|
||||
"claude-fable-5": (5, 2, 5),
|
||||
"anthropic/claude-fable-5": (5, 2, 5),
|
||||
"claude-opus-4-8": (5, 2, 5),
|
||||
"claude-opus-4.8": (5, 2, 5),
|
||||
"anthropic/claude-opus-4.8": (5, 2, 5),
|
||||
|
||||
@@ -71,10 +71,6 @@ BUILTIN_MODELS: dict[str, list[dict[str, Any]]] = {
|
||||
"model_id": "claude-haiku-4-5", "router_model_id": "cc/claude-haiku-4-5-20251001", "api": "anthropic", "reasoning": True, "route": "cc"},
|
||||
|
||||
# Fable 5 re-added 2026-07-02 after the ban lifted (Eric confirmed access is back); pull both rows again if it errors live.
|
||||
{"value": "fable-5-cc", "label": "Claude Fable 5", "context_window": 1_000_000,
|
||||
"model_id": "claude-fable-5", "router_model_id": "cc/claude-fable-5", "api": "anthropic", "reasoning": True, "route": "cc"},
|
||||
{"value": "fable-5-api", "label": "Claude Fable 5 (API key)", "context_window": 1_000_000,
|
||||
"model_id": "claude-fable-5", "router_model_id": "claude-fable-5", "api": "anthropic", "reasoning": True, "route": "api"},
|
||||
{"value": "opus-5-api", "label": "Claude Opus 5 (API key)", "context_window": 1_000_000,
|
||||
"model_id": "claude-opus-5", "router_model_id": "claude-opus-5", "api": "anthropic", "reasoning": True, "route": "api"},
|
||||
{"value": "opus-4-8-api", "label": "Claude Opus 4.8 (API key)", "context_window": 1_000_000,
|
||||
@@ -263,9 +259,6 @@ def p_antigravity_connected() -> bool:
|
||||
|
||||
def resolve_model_id_for_sdk(short_name: str, settings: AppSettings) -> str:
|
||||
"""Short model name → id string for ClaudeAgentOptions."""
|
||||
# Free trial funds only Haiku via the cloud proxy; force it so a session left on a gpt-*/sub model can't escape to a lane the trial can't fund (which snags as a 401/404).
|
||||
if getattr(settings, "connection_mode", "own_key") == "free-trial":
|
||||
short_name = "haiku"
|
||||
entry = find_builtin_model(short_name)
|
||||
if entry is None:
|
||||
return short_name
|
||||
@@ -410,7 +403,6 @@ COST_PER_1M_TOKENS: dict[tuple[str, str], tuple[float, float]] = {
|
||||
("Anthropic", "opus-4-7"): (5.0, 25.0),
|
||||
("Anthropic", "opus-4-8"): (5.0, 25.0),
|
||||
("Anthropic", "opus-5"): (5.0, 25.0),
|
||||
("Anthropic", "fable-5-api"): (10.0, 50.0),
|
||||
("Anthropic", "haiku"): (1.0, 5.0),
|
||||
# OpenAI API-key rates (GPT-5.6 tiers, GA 2026-07-09)
|
||||
("OpenAI", "gpt-5.6-api"): (5.0, 30.0),
|
||||
|
||||
@@ -20,9 +20,6 @@ def thinking_params_for(api: str, level: str, model_id: str = "") -> dict | None
|
||||
|
||||
if level == "off":
|
||||
if api == "anthropic":
|
||||
# Fable 5 400s on an explicit thinking:disabled; omit the param to turn thinking off (off is its default). Other Claude models accept it.
|
||||
if "fable" in model_id:
|
||||
return None
|
||||
return {"thinking": {"type": "disabled"}}
|
||||
if api == "codex":
|
||||
return {"reasoning": {"effort": "none"}}
|
||||
|
||||
@@ -285,7 +285,7 @@ def p_friendly_model(raw: str) -> str:
|
||||
base = (raw or "unknown").removesuffix("-cc")
|
||||
names = {
|
||||
"opus-5": "Claude Opus 5", "opus": "Claude Opus", "sonnet-5": "Claude Sonnet 5",
|
||||
"sonnet": "Claude Sonnet", "haiku": "Claude Haiku", "fable-5": "Claude Fable 5",
|
||||
"sonnet": "Claude Sonnet", "haiku": "Claude Haiku",
|
||||
}
|
||||
if base in names:
|
||||
return names[base]
|
||||
|
||||
@@ -78,6 +78,12 @@ def migrate_legacy_fields(raw: dict) -> dict:
|
||||
raw["connection_mode"] = "openswarm-pro"
|
||||
if "openswarm_auth_token" in raw and "openswarm_bearer_token" not in raw:
|
||||
raw["openswarm_bearer_token"] = raw.pop("openswarm_auth_token")
|
||||
# The free tier is retired. Nothing arms it any more, but installs that already carry the mode
|
||||
# would otherwise keep it forever, and with it the silent mid-run pin to Haiku. Move them to
|
||||
# own_key so the model they picked is the model they get.
|
||||
if raw.get("connection_mode") == "free-trial":
|
||||
raw["connection_mode"] = "own_key"
|
||||
raw.pop("free_trial_token", None)
|
||||
return raw
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
"""The free tier is retired. It was never something a user picked: the app minted it at boot for
|
||||
anyone with no key, and while armed it silently pinned every session to Haiku mid-run, so people ran
|
||||
a different model than the one they chose with no indication. These pin the removal in all three
|
||||
places it has to hold: nobody new gets in, nobody already in stays, and the pin itself is gone."""
|
||||
|
||||
from backend.apps.settings.store import migrate_legacy_fields
|
||||
from backend.apps.settings.models import AppSettings
|
||||
from backend.apps.agents.providers.registry import resolve_model_id_for_sdk
|
||||
|
||||
|
||||
def test_an_install_already_on_the_free_tier_is_moved_off_it():
|
||||
raw = migrate_legacy_fields({"connection_mode": "free-trial", "free_trial_token": "tok-abc"})
|
||||
assert raw["connection_mode"] == "own_key"
|
||||
assert "free_trial_token" not in raw, "a retired tier must not leave its credential behind"
|
||||
|
||||
|
||||
def test_the_migration_leaves_every_other_mode_alone():
|
||||
for mode in ("own_key", "openswarm-pro", "custom"):
|
||||
assert migrate_legacy_fields({"connection_mode": mode})["connection_mode"] == mode
|
||||
|
||||
|
||||
def test_the_chosen_model_is_the_model_that_runs():
|
||||
# The old behavior rewrote short_name to "haiku" whenever the mode was free-trial, which is the
|
||||
# reported "swapped me to a different model halfway through" with no indication.
|
||||
s = AppSettings(connection_mode="own_key")
|
||||
assert "opus" in resolve_model_id_for_sdk("opus-5", s).lower()
|
||||
assert "sonnet" in resolve_model_id_for_sdk("sonnet-5", s).lower()
|
||||
|
||||
|
||||
def test_no_haiku_pin_survives_even_if_the_retired_mode_is_forced_in():
|
||||
# Belt over the migration: hand the resolver the retired mode directly and it must still honor
|
||||
# the caller's model rather than reaching for Haiku.
|
||||
s = AppSettings()
|
||||
object.__setattr__(s, "__dict__", {**s.__dict__, "connection_mode": "free-trial"})
|
||||
assert "haiku" not in resolve_model_id_for_sdk("opus-5", s).lower()
|
||||
|
||||
|
||||
def test_nothing_in_the_app_mints_a_free_trial_on_boot():
|
||||
"""The mint was fired unconditionally from the renderer's boot effect, which is why the tier was
|
||||
still shipping long after the decision to drop it. Guard the renderer, not just the backend."""
|
||||
from pathlib import Path
|
||||
main_tsx = Path(__file__).resolve().parents[2] / "frontend" / "src" / "app" / "Main.tsx"
|
||||
assert "free-trial/mint" not in main_tsx.read_text(encoding="utf-8"), (
|
||||
"Main.tsx arms the retired free tier at boot again"
|
||||
)
|
||||
@@ -668,16 +668,19 @@ def test_banned_models_not_offered():
|
||||
can't serve it, AI Studio key 429s pro-preview. (gpt-5.5's cx entry left
|
||||
this list 2026-07-26: the pinned 0.3.60's old 404 healed upstream and a
|
||||
live probe returned a real completion, so both its lanes work. Fable 5
|
||||
left 2026-07-02 after its ban lifted.)"""
|
||||
left 2026-07-02 after its ban lifted, and came BACK 2026-08-12: it was
|
||||
selectable and simply did not work, and Opus covers the same ground.)"""
|
||||
from backend.apps.agents.providers.registry import BUILTIN_MODELS
|
||||
all_values = {m["value"] for models in BUILTIN_MODELS.values() for m in models}
|
||||
for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api"):
|
||||
for dead in ("gemini-3.1-pro", "gemini-3.1-pro-api", "gemini-3-flash", "gemini-3-flash-api",
|
||||
"fable-5-cc", "fable-5-api"):
|
||||
assert dead not in all_values, f"{dead} is back in the picker"
|
||||
assert "gpt-5.5-api" in all_values
|
||||
assert "gpt-5.5" in all_values # cx lane restored 2026-07-26 (live-probed)
|
||||
# No '3.1 pro' label survives in any provider group either.
|
||||
all_labels = " | ".join(m["label"].lower() for models in BUILTIN_MODELS.values() for m in models)
|
||||
assert "3.1 pro" not in all_labels
|
||||
assert "fable" not in all_labels
|
||||
assert "gemini 3 flash" not in all_labels
|
||||
|
||||
|
||||
|
||||
@@ -246,11 +246,10 @@ const SettingsLoader: React.FC<{ children: React.ReactNode }> = ({ children }) =
|
||||
})
|
||||
.catch(() => {})
|
||||
.finally(() => {
|
||||
// Arm the zero-config free trial when nothing is connected so a brand-new user can run an agent immediately. The backend no-ops if a real key or subscription exists, so this is safe to fire on every launch.
|
||||
fetch(`${API_BASE}/subscription/free-trial/mint`, { method: 'POST' })
|
||||
.catch(() => {})
|
||||
// The backend arms server-side regardless of whether the browser can read the mint response (a transient boot-time CORS/timing miss makes `data` unreadable), so refetch unconditionally, the GET is the only reliable signal the UI gets that it armed.
|
||||
.finally(() => { dispatch(fetchSettings()); dispatch(fetchSubscriptionStatus()); dispatch(markFreeTrialArmSettled()); });
|
||||
// The free tier is retired. Nothing arms it any more, so there is no mint on boot; the flag
|
||||
// still settles so the "connect a model" banner is not held back waiting for a call that
|
||||
// will never happen.
|
||||
dispatch(markFreeTrialArmSettled());
|
||||
});
|
||||
return () => {
|
||||
if (bootTimer) clearInterval(bootTimer);
|
||||
|
||||
Reference in New Issue
Block a user