[eric] onboarding: read real ChatGPT history at first run via the codex connect token (no login/browser)

This commit is contained in:
ciregenz
2026-07-15 22:34:10 -07:00
parent 1bbc8d7dfc
commit 0f4ac4e009
3 changed files with 125 additions and 0 deletions
+99
View File
@@ -0,0 +1,99 @@
"""Read the user's own ChatGPT conversation titles + Memory straight from ChatGPT's
backend using the codex connect token, no website login or browser session needed.
The token 9Router already holds from "Sign in with ChatGPT" is a valid bearer for
chatgpt.com/backend-api (that is how Codex runs); paired with the chatgpt-account-id
claim from the id token it reaches /conversations and /memories from the user's own
machine. Read-only, capped, and fails open to "" on anything (expired token,
Cloudflare, shape drift) so prep just falls back to the local scan.
"""
from typing import List, Optional, Tuple
import httpx
from typeguard import typechecked
from backend.apps.nine_router.process import read_persisted_connections
from backend.apps.onboarding.identity import decode_jwt_payload
BASE = "https://chatgpt.com/backend-api"
PAGE = 100
CAP_PAGES = 40
CAP_TITLES = 1000
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
@typechecked
def p_codex_creds() -> Optional[Tuple[str, Optional[str]]]:
for c in read_persisted_connections():
if c.get("provider") == "codex" and c.get("isActive") and c.get("accessToken"):
claims = decode_jwt_payload(c.get("idToken") or "")
auth = claims.get("https://api.openai.com/auth", {}) if isinstance(claims, dict) else {}
acct = auth.get("chatgpt_account_id") if isinstance(auth, dict) else None
return (str(c["accessToken"]), str(acct) if acct else None)
return None
@typechecked
def summarize_chatgpt_usage(total: int, memories: List[str], titles: List[str]) -> str:
parts: List[str] = []
if total > 0:
parts.append(f"They have {total} past AI conversations.")
if memories:
parts.append("Facts their AI remembers about them: " + "; ".join(memories))
if titles:
parts.append("Topics they keep coming back to (recent first): " + "; ".join(titles[:150]))
return "\n".join(parts)[:4000]
@typechecked
async def harvest_chatgpt_usage() -> str:
creds = p_codex_creds()
if creds is None:
return ""
token, acct = creds
headers = {
"Authorization": f"Bearer {token}",
"Accept": "application/json",
"User-Agent": UA,
"Origin": "https://chatgpt.com",
"Referer": "https://chatgpt.com/",
}
if acct:
headers["chatgpt-account-id"] = acct
titles: List[str] = []
seen: set = set()
memories: List[str] = []
try:
async with httpx.AsyncClient(timeout=15.0, headers=headers) as client:
offset = 0
for _ in range(CAP_PAGES):
if len(titles) >= CAP_TITLES:
break
r = await client.get(f"{BASE}/conversations", params={"offset": offset, "limit": PAGE, "order": "updated"})
if r.status_code != 200:
return "" # expired token / Cloudflare / shape drift: fail open, prep uses the scan
items = (r.json() or {}).get("items") or []
if not items:
break
fresh = 0
for it in items:
cid = it.get("id")
if cid and cid not in seen:
seen.add(cid)
title = it.get("title")
if title:
titles.append(str(title))
fresh += 1
if fresh == 0 or len(items) < PAGE:
break
offset += PAGE
try:
mr = await client.get(f"{BASE}/memories", params={"include_memory_entries": "true"})
if mr.status_code == 200:
memories = [str(m.get("content")) for m in (mr.json() or {}).get("memories", []) if m.get("content")][:40]
except Exception:
pass
except Exception:
return ""
return summarize_chatgpt_usage(len(seen), memories, titles)
+7
View File
@@ -43,6 +43,13 @@ def post_scan() -> dict:
@onboarding.router.post("/prep")
@typechecked
async def post_prep(body: PrepRequest) -> dict:
from backend.apps.onboarding.chatgpt_usage import harvest_chatgpt_usage
from backend.apps.settings.store import load_settings
# The frontend read needs a logged-in provider CARD, which a fresh install lacks; the codex connect token reads the ChatGPT backend directly, so fill the gap here.
if not body.usage_summary.strip():
harvested = await harvest_chatgpt_usage()
if harvested:
body.usage_summary = harvested
return (await build_prep(load_settings(), body)).model_dump()