From 0f4ac4e009bee0e5d68412c62ee0696b341bc405 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Wed, 15 Jul 2026 22:34:10 -0700 Subject: [PATCH] [eric] onboarding: read real ChatGPT history at first run via the codex connect token (no login/browser) --- backend/apps/onboarding/chatgpt_usage.py | 99 ++++++++++++++++++++++++ backend/apps/onboarding/onboarding.py | 7 ++ backend/tests/test_onboarding.py | 19 +++++ 3 files changed, 125 insertions(+) create mode 100644 backend/apps/onboarding/chatgpt_usage.py diff --git a/backend/apps/onboarding/chatgpt_usage.py b/backend/apps/onboarding/chatgpt_usage.py new file mode 100644 index 00000000..fc654158 --- /dev/null +++ b/backend/apps/onboarding/chatgpt_usage.py @@ -0,0 +1,99 @@ +"""Read the user's own ChatGPT conversation titles + Memory straight from ChatGPT's +backend using the codex connect token, no website login or browser session needed. + +The token 9Router already holds from "Sign in with ChatGPT" is a valid bearer for +chatgpt.com/backend-api (that is how Codex runs); paired with the chatgpt-account-id +claim from the id token it reaches /conversations and /memories from the user's own +machine. Read-only, capped, and fails open to "" on anything (expired token, +Cloudflare, shape drift) so prep just falls back to the local scan. +""" + +from typing import List, Optional, Tuple + +import httpx +from typeguard import typechecked + +from backend.apps.nine_router.process import read_persisted_connections +from backend.apps.onboarding.identity import decode_jwt_payload + +BASE = "https://chatgpt.com/backend-api" +PAGE = 100 +CAP_PAGES = 40 +CAP_TITLES = 1000 +UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" + + +@typechecked +def p_codex_creds() -> Optional[Tuple[str, Optional[str]]]: + for c in read_persisted_connections(): + if c.get("provider") == "codex" and c.get("isActive") and c.get("accessToken"): + claims = decode_jwt_payload(c.get("idToken") or "") + auth = claims.get("https://api.openai.com/auth", {}) if isinstance(claims, dict) else {} + acct = auth.get("chatgpt_account_id") if isinstance(auth, dict) else None + return (str(c["accessToken"]), str(acct) if acct else None) + return None + + +@typechecked +def summarize_chatgpt_usage(total: int, memories: List[str], titles: List[str]) -> str: + parts: List[str] = [] + if total > 0: + parts.append(f"They have {total} past AI conversations.") + if memories: + parts.append("Facts their AI remembers about them: " + "; ".join(memories)) + if titles: + parts.append("Topics they keep coming back to (recent first): " + "; ".join(titles[:150])) + return "\n".join(parts)[:4000] + + +@typechecked +async def harvest_chatgpt_usage() -> str: + creds = p_codex_creds() + if creds is None: + return "" + token, acct = creds + headers = { + "Authorization": f"Bearer {token}", + "Accept": "application/json", + "User-Agent": UA, + "Origin": "https://chatgpt.com", + "Referer": "https://chatgpt.com/", + } + if acct: + headers["chatgpt-account-id"] = acct + titles: List[str] = [] + seen: set = set() + memories: List[str] = [] + try: + async with httpx.AsyncClient(timeout=15.0, headers=headers) as client: + offset = 0 + for _ in range(CAP_PAGES): + if len(titles) >= CAP_TITLES: + break + r = await client.get(f"{BASE}/conversations", params={"offset": offset, "limit": PAGE, "order": "updated"}) + if r.status_code != 200: + return "" # expired token / Cloudflare / shape drift: fail open, prep uses the scan + items = (r.json() or {}).get("items") or [] + if not items: + break + fresh = 0 + for it in items: + cid = it.get("id") + if cid and cid not in seen: + seen.add(cid) + title = it.get("title") + if title: + titles.append(str(title)) + fresh += 1 + if fresh == 0 or len(items) < PAGE: + break + offset += PAGE + try: + mr = await client.get(f"{BASE}/memories", params={"include_memory_entries": "true"}) + if mr.status_code == 200: + memories = [str(m.get("content")) for m in (mr.json() or {}).get("memories", []) if m.get("content")][:40] + except Exception: + pass + except Exception: + return "" + return summarize_chatgpt_usage(len(seen), memories, titles) diff --git a/backend/apps/onboarding/onboarding.py b/backend/apps/onboarding/onboarding.py index c24f3eab..f7d52283 100644 --- a/backend/apps/onboarding/onboarding.py +++ b/backend/apps/onboarding/onboarding.py @@ -43,6 +43,13 @@ def post_scan() -> dict: @onboarding.router.post("/prep") @typechecked async def post_prep(body: PrepRequest) -> dict: + from backend.apps.onboarding.chatgpt_usage import harvest_chatgpt_usage from backend.apps.settings.store import load_settings + # The frontend read needs a logged-in provider CARD, which a fresh install lacks; the codex connect token reads the ChatGPT backend directly, so fill the gap here. + if not body.usage_summary.strip(): + harvested = await harvest_chatgpt_usage() + if harvested: + body.usage_summary = harvested + return (await build_prep(load_settings(), body)).model_dump() diff --git a/backend/tests/test_onboarding.py b/backend/tests/test_onboarding.py index 7fb78a3b..ffab5ae6 100644 --- a/backend/tests/test_onboarding.py +++ b/backend/tests/test_onboarding.py @@ -113,6 +113,25 @@ def test_parse_prep_carries_reasons(): assert parsed.app_reason == "you plan lifts with ChatGPT daily" +def test_summarize_chatgpt_usage_leads_with_memory_and_caps(): + from backend.apps.onboarding.chatgpt_usage import summarize_chatgpt_usage + + s = summarize_chatgpt_usage(812, ["Has an Akita", "Squats 495"], ["Swift concurrency", "Deadlift form"]) + assert "812 past AI conversations" in s + assert "Has an Akita; Squats 495" in s + assert "Swift concurrency; Deadlift form" in s + big = summarize_chatgpt_usage(1000, [], [f"topic number {i} about something" for i in range(1000)]) + assert len(big) <= 4000 + + +@pytest.mark.asyncio +async def test_harvest_chatgpt_usage_fails_open_without_codex(monkeypatch): + from backend.apps.onboarding import chatgpt_usage + + monkeypatch.setattr(chatgpt_usage, "read_persisted_connections", lambda: []) + assert await chatgpt_usage.harvest_chatgpt_usage() == "" + + @pytest.mark.asyncio async def test_build_prep_fails_open_without_provider(monkeypatch): async def boom(*args, **kwargs):