diff --git a/docker/context-profiles/ai-eval-lib.js b/docker/context-profiles/ai-eval-lib.js index 4b38c9b78..c90793cf2 100644 --- a/docker/context-profiles/ai-eval-lib.js +++ b/docker/context-profiles/ai-eval-lib.js @@ -255,8 +255,10 @@ function createAuthLease(authHome) { if (!after.equals(original)) { JSON.parse(after.toString('utf8')); const temp = `${source}.${process.pid}.tmp`; - fs.writeFileSync(temp, after, { flag: 'wx', mode: 0o600 }); - fs.renameSync(temp, source); + try { + fs.writeFileSync(temp, after, { flag: 'wx', mode: 0o600 }); + fs.renameSync(temp, source); + } finally { fs.rmSync(temp, { force: true }); } } } catch { /* An unreadable refresh keeps the previous login; the next call reports any auth failure. */ } fs.rmSync(leased, { force: true }); @@ -282,7 +284,7 @@ function readClaudeKeychainToken() { return token; } -function createClaudeProvider({ allowRealProvider = false, executable, model, +function createClaudeProvider({ allowRealProvider = false, allowCredentialedTools = false, executable, model, apiKey = process.env.ANTHROPIC_API_KEY, oauthToken = process.env.CLAUDE_CODE_OAUTH_TOKEN, tokenSource = readClaudeKeychainToken, persistSessions = false, execute = spawnSync } = {}) { if (allowRealProvider !== true) throw new Error('Real provider requires explicit opt-in'); @@ -301,6 +303,9 @@ function createClaudeProvider({ allowRealProvider = false, executable, model, const provider = request => { if (fingerprintExecutable(binary.path).digest !== pin.executableDigest) fail('source-drift'); const selection = request.phase === 'selection'; + if (!selection && !allowCredentialedTools) { + throw new Error('Claude task tools can read provider credentials; explicit credentialed-tool opt-in is required'); + } // Selection is tool-free and read-only; task execution may edit and run commands in the workspace. // Claude has no cwd-write sandbox flag, so containment relies on the isolated home and temp workspace. const args = ['--print', '--output-format', 'json', @@ -682,8 +687,9 @@ function outcomeTrial(item, arm, repeat, repoRoot, execute, cwd, environment, ta if (harvest) harvest(arm, `${item.id}--step${index + 1}`, repeat, environment); if (result.status !== 'completed') { // A failed ticket ends the chain; remaining tickets are unscored. - for (let rest = index; rest < item.steps.length; rest++) { - steps.push({ score: 0, ...(metrics ? metricsSince(metrics, start) : {}) }); + steps.push({ score: 0, ...(metrics ? metricsSince(metrics, start) : {}) }); + for (let rest = index + 1; rest < item.steps.length; rest++) { + steps.push({ score: 0, ...(metrics ? metricsSince(metrics, metrics.length) : {}) }); } break; } @@ -748,7 +754,8 @@ function createHarvester(artifactDir, envs) { } function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), registration, - repeats = 1, provider, family, allowRealProvider = false, executable, model, effort, authHome, environments, + repeats = 1, provider, family, allowRealProvider = false, allowCredentialedTools = false, + executable, model, effort, authHome, environments, arms = undefined, artifactDir = null, maxCalls = 300, deadlineMs = 3600000, callTimeoutMs = 300000 } = {}) { if (!provider && !allowRealProvider) throw new Error('Evaluation requires an injected provider or explicit opt-in'); if (!bounded(maxCalls, 1, 2000) || !bounded(deadlineMs, 1, 8 * 3600000) @@ -756,6 +763,9 @@ function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), re if (!provider && !registration) throw new Error('Real evaluation requires prior registration'); const resolvedFamily = provider ? (family || 'codex') : resolveFamily(family, executable); if (resolvedFamily === 'claude' && effort !== undefined) throw new Error('Reasoning effort applies only to the Codex provider'); + if (!provider && resolvedFamily === 'claude' && !allowCredentialedTools) { + throw new Error('Claude task tools can read provider credentials; explicit credentialed-tool opt-in is required'); + } const pin = preregister({ repoRoot, corpus, repeats, model, executable, effort, arms }); if (!provider && resolvedFamily === 'codex' && pin.arms.includes('ecc-legacy')) { throw new Error('Codex real evaluation requires --arms without ecc-legacy; the pinned legacy skills arm is Claude-only'); @@ -763,7 +773,8 @@ function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), re if (registration && !isDeepStrictEqual(registration, pin)) throw new Error('Registration pin mismatch'); const injected = Boolean(provider); const liveProvider = provider || (resolvedFamily === 'claude' - ? createClaudeProvider({ allowRealProvider, executable, model, persistSessions: Boolean(artifactDir) }) + ? createClaudeProvider({ allowRealProvider, allowCredentialedTools, executable, model, + persistSessions: Boolean(artifactDir) }) : createCodexProvider({ allowRealProvider, executable, model, effort, authHome })); const state = { calls: 0, metrics: [], maxCalls, callTimeoutMs, family: resolvedFamily, deadline: Date.now() + deadlineMs, provider: liveProvider, diff --git a/docker/context-profiles/ai-eval.js b/docker/context-profiles/ai-eval.js index c3195c981..92903d6c8 100644 --- a/docker/context-profiles/ai-eval.js +++ b/docker/context-profiles/ai-eval.js @@ -5,7 +5,7 @@ const { preregister, runEvaluation, loadCorpus } = require('./ai-eval-lib'); function main(argv = process.argv.slice(2), injected = {}) { const flags = new Map(); - const switches = new Set(['--plan', '--allow-real-provider', '--help']); + const switches = new Set(['--plan', '--allow-real-provider', '--allow-credentialed-tools', '--help']); const values = new Set(['--registration', '--model', '--executable', '--provider', '--auth-home', '--effort', '--repeats', '--max-calls', '--deadline-ms', '--artifact-dir', '--corpus', '--call-timeout-ms', '--arms']); for (let i = 0; i < argv.length; i++) { const flag = argv[i]; @@ -14,10 +14,13 @@ function main(argv = process.argv.slice(2), injected = {}) { flags.set(flag, switches.has(flag) ? true : argv[++i]); } if (flags.has('--help')) { - return { usage: 'ai-eval.js --plan [--corpus FILE] [--arms a,b] [--repeats N] [--model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL]] | --allow-real-provider --registration FILE --model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL (Codex only)] [--auth-home ABSOLUTE_DIR (Codex only)] [--corpus FILE] [--arms a,b] [--repeats N] [--max-calls N] [--deadline-ms N] [--call-timeout-ms N]. Claude auth: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, or the macOS Keychain login.' }; + return { usage: 'ai-eval.js --plan [--corpus FILE] [--arms a,b] [--repeats N] [--model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL]] | --allow-real-provider --registration FILE --model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--allow-credentialed-tools (Claude only)] [--effort LEVEL (Codex only)] [--auth-home ABSOLUTE_DIR (Codex only)] [--corpus FILE] [--arms a,b] [--repeats N] [--max-calls N] [--deadline-ms N] [--call-timeout-ms N]. Claude auth: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, or the macOS Keychain login.' }; } if (flags.get('--provider') !== undefined && !['claude', 'codex'].includes(flags.get('--provider'))) throw new Error('Provider must be claude or codex'); if (flags.get('--provider') === 'claude' && flags.has('--effort')) throw new Error('Reasoning effort applies only to the Codex provider'); + if (flags.has('--allow-credentialed-tools') && (!flags.has('--allow-real-provider') || flags.get('--provider') !== 'claude')) { + throw new Error('Credentialed-tool opt-in requires a real Claude evaluation'); + } const repeats = flags.has('--repeats') ? Number(flags.get('--repeats')) : 1; const corpus = flags.has('--corpus') ? loadCorpus(flags.get('--corpus')) : undefined; const arms = flags.has('--arms') ? flags.get('--arms').split(',').map(a => a.trim()).filter(Boolean) : undefined; @@ -30,6 +33,7 @@ function main(argv = process.argv.slice(2), injected = {}) { if (!flags.has('--registration')) throw new Error('Evaluation requires a preregistration file'); const registration = JSON.parse(fs.readFileSync(flags.get('--registration'), 'utf8')); return runEvaluation({ ...injected, registration, repeats, allowRealProvider: flags.has('--allow-real-provider'), + allowCredentialedTools: flags.has('--allow-credentialed-tools'), executable: flags.get('--executable'), model: flags.get('--model'), family: flags.get('--provider'), effort: flags.get('--effort'), authHome: flags.get('--auth-home'), artifactDir: flags.get('--artifact-dir'), ...(corpus ? { corpus } : {}), ...(arms ? { arms } : {}), ...(flags.has('--max-calls') ? { maxCalls: Number(flags.get('--max-calls')) } : {}), diff --git a/docker/context-profiles/complex-eval/DESIGN.md b/docker/context-profiles/complex-eval/DESIGN.md index ff697bf9f..03ada0372 100644 --- a/docker/context-profiles/complex-eval/DESIGN.md +++ b/docker/context-profiles/complex-eval/DESIGN.md @@ -136,7 +136,7 @@ node docker/context-profiles/ai-eval.js --plan \ > registration.json # 5. Run (requires your own Claude subscription login or API key). -node docker/context-profiles/ai-eval.js --allow-real-provider \ +node docker/context-profiles/ai-eval.js --allow-real-provider --allow-credentialed-tools \ --registration registration.json \ --corpus docker/context-profiles/complex-corpus.json \ --provider claude --model --executable /absolute/path/to/claude \ @@ -144,6 +144,11 @@ node docker/context-profiles/ai-eval.js --allow-real-provider \ --artifact-dir /absolute/path/for/transcripts > report.json ``` +Claude task tools inherit the provider credential through the CLI process and can read it. Use +`--allow-credentialed-tools` only with a trusted local corpus and credential. Without that +explicit flag, real Claude task evaluation stops before a provider call; selection-only calls +remain tool-free. This development evaluator does not provide a credential isolation boundary. + The registration digest binds the exact corpus, evaluator source, model, and executable; the run refuses to start if any of them drift, and aborts if the tree changes mid-run. `--artifact-dir` retains per-trial session transcripts diff --git a/docs/design/context-profile-delivery.md b/docs/design/context-profile-delivery.md index 093891b94..82d6151fe 100644 --- a/docs/design/context-profile-delivery.md +++ b/docs/design/context-profile-delivery.md @@ -40,7 +40,7 @@ ecc profile start --state-root /absolute/dedicated/profile-store --native-root / `resolve --state-root` uses the saved base, mode and exclusions. It rejects overrides and stale source generations. `mode` preserves the configured profile and explicit selections while recording the new mode transactionally. -A task input contains caller-assigned `sessionId`, `taskId`, positive integer `revision`, and `phase`. Optional fields are `query`, `explicitIds`, `proposedIds`, and `noWorkflow`. Increment revision for material task changes; keep it stable for rewording. Task prose is consumed locally and omitted from returned receipts. +A task input contains caller-assigned `sessionId`, `taskId`, positive integer `revision`, and `phase`. Optional fields are `query`, `explicitIds`, `proposedIds`, and `noWorkflow`. Increment revision for material task changes. A changed query, including rewording, also invalidates selection reuse. Task prose is consumed locally and omitted from returned receipts. ```json { @@ -54,7 +54,7 @@ A task input contains caller-assigned `sessionId`, `taskId`, positive integer `r Auto uses explicit user IDs first, then a completed pinned decision, an unambiguous ranked match, one cited skill name, or admitted agent-proposed IDs. Ambiguous free text shortlists up to five candidates for a bounded proposal. Manual uses explicit IDs; suggest emits a proposal without bodies. `--load` returns selected UTF-8 instructions and declared required resources, capped at 32,000 bytes across at most eight skills. `--task-input -` accepts one UTF-8 JSON object on standard input, capped at 65,536 bytes. These byte caps are output and transport bounds, not native tokenizer results. -Save the returned `selection.receipt` as a separate JSON document to use `--previous receipt.json`. `--expected-digest` can bind a load to a prior selection digest. Source, routing-policy version, profile, mode, exclusions, session, task revision and phase invalidate stale reuse. A pending proposal cannot be reused as a completed decision. Receipts are integrity checks for local operation, not an authorization signature. +Save the returned `selection.receipt` as a separate JSON document to use `--previous receipt.json`. `--expected-digest` can bind a load to a prior selection digest. Source, trigger content, routing-policy version, profile, mode, exclusions, session, task revision, phase, and a digest of the query invalidate stale reuse. A pending proposal cannot be reused as a completed decision. Receipts are integrity checks for local operation, not an authorization signature. An agent can call the resolver at task boundaries and read the returned context. This integration is prompt-advisory. Returning a body never grants tools, invokes shell interpolation, starts a native skill, changes hooks or installs dependencies. Native manual-only flags and authority-bearing metadata are checked before selection. Base profiles remain stable during task routing. diff --git a/scripts/dev/generate-skill-triggers.js b/scripts/dev/generate-skill-triggers.js index 1ff1c6de7..44ed6116c 100644 --- a/scripts/dev/generate-skill-triggers.js +++ b/scripts/dev/generate-skill-triggers.js @@ -43,6 +43,10 @@ function parseFlags(argv) { flags[arg.slice(2).replace(/-([a-z])/g, (_, c) => c.toUpperCase())] = argv[index += 1]; } else throw new Error(`Unknown flag: ${arg}`); } + if (!/^[1-9][0-9]*$/.test(String(flags.batch)) || !Number.isSafeInteger(Number(flags.batch))) { + throw new Error('--batch must be a positive integer'); + } + flags.batch = Number(flags.batch); return flags; } diff --git a/scripts/lib/context-selection.js b/scripts/lib/context-selection.js index a243bef66..85a496800 100644 --- a/scripts/lib/context-selection.js +++ b/scripts/lib/context-selection.js @@ -154,7 +154,9 @@ function resolveTaskContext({ repoRoot = DEFAULT_REPO_ROOT, task, profileId = 'l if (excluded.has(id)) throw new Error(`Context ID is excluded: ${id}`); }); const taskBinding = { sessionId: task.sessionId, taskId: task.taskId, revision: task.revision, phase: task.phase }; - const bindingDigest = digestObject({ ...taskBinding, planDigest: plan.planDigest, routingPolicyVersion: ROUTING_POLICY_VERSION }); + const bindingDigest = digestObject({ ...taskBinding, planDigest: plan.planDigest, + routingPolicyVersion: ROUTING_POLICY_VERSION, triggersDigest: digestObject(triggers), + queryDigest: digestObject(task.query || '') }); const reused = Boolean(previous && previous.bindingDigest === bindingDigest && !task.noWorkflow && ['selected', 'none'].includes(previous.decision) && !explicitIds.length && !proposedIds.length); const admissible = id => { diff --git a/scripts/lib/utils.js b/scripts/lib/utils.js index 9766fcd0c..2e8550e5e 100644 --- a/scripts/lib/utils.js +++ b/scripts/lib/utils.js @@ -195,9 +195,11 @@ function sameRepoIdentity(a, b) { if (!a || !b) return false; if (normalizeRepoPath(a) === normalizeRepoPath(b)) return true; try { - const sa = fs.statSync(a); - const sb = fs.statSync(b); - return sa.ino !== 0 && sa.dev === sb.dev && sa.ino === sb.ino; + // Windows file IDs can exceed Number.MAX_SAFE_INTEGER; rounded IDs may + // otherwise make distinct files look identical. + const sa = fs.statSync(a, { bigint: true }); + const sb = fs.statSync(b, { bigint: true }); + return sa.ino !== 0n && sa.dev === sb.dev && sa.ino === sb.ino; } catch { return false; } diff --git a/tests/lib/context-profile-eval.test.js b/tests/lib/context-profile-eval.test.js index a641bb116..c34f8cb25 100644 --- a/tests/lib/context-profile-eval.test.js +++ b/tests/lib/context-profile-eval.test.js @@ -187,6 +187,13 @@ test('subscription lease copies private auth into the call home, returns refresh }); assert.equal(fs.existsSync(path.join(codexHome, 'auth.json')), false); assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"new"}}'); + const staleTemp = path.join(authHome, `auth.json.${process.pid}.tmp`); + fs.writeFileSync(staleTemp, 'stale', { mode: 0o600 }); + lease.run(codexHome, () => fs.writeFileSync(path.join(codexHome, 'auth.json'), '{"tokens":{"refresh_token":"latest"}}')); + assert.equal(fs.existsSync(staleTemp), false); + assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"new"}}'); + lease.run(codexHome, () => fs.writeFileSync(path.join(codexHome, 'auth.json'), '{"tokens":{"refresh_token":"latest"}}')); + assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"latest"}}'); assert.throws(() => lease.run(codexHome, () => { throw new Error('provider crashed'); }), /crashed/); assert.equal(fs.existsSync(path.join(codexHome, 'auth.json')), false); fs.chmodSync(authHome, 0o755); @@ -334,7 +341,11 @@ test('Claude provider runs tool-free selection and permissioned tasks with a san const request = { phase: 'selection', input: 'request', cwd, timeoutMs: 5, maxBuffer: 1000, env: { PATH: '/bin', HOME: cwd, CLAUDE_CONFIG_DIR: path.join(cwd, 'cfg'), TMPDIR: '/tmp', CODEX_HOME: '/tmp/x', SECRET: 's' } }; provider(request); - provider({ ...request, phase: 'task' }); + assert.throws(() => provider({ ...request, phase: 'task' }), /credentialed-tool opt-in/); + const credentialed = createClaudeProvider({ allowRealProvider: true, allowCredentialedTools: true, + executable: process.execPath, model: 'pinned-model', oauthToken: 'test-token', tokenSource: null, + execute(command, args, options) { calls.push({ args, options }); return { status: 0, stdout: claudeJson() }; } }); + credentialed({ ...request, phase: 'task' }); assert.ok(calls[0].args.includes('--tools')); assert.ok(!calls[0].args.join(' ').includes('bypassPermissions')); assert.ok(calls[1].args.includes('--permission-mode') && calls[1].args.includes('bypassPermissions')); diff --git a/tests/lib/context-selection.test.js b/tests/lib/context-selection.test.js index 407c0c8ae..4e3be6169 100644 --- a/tests/lib/context-selection.test.js +++ b/tests/lib/context-selection.test.js @@ -104,14 +104,27 @@ test('authority-bearing metadata cannot become automatic invocation', () => with test('receipt pins source and task identity without retaining query text', () => withFixture(repoRoot => { const first = resolve(repoRoot, { proposedIds: ['skill:feature'], query: 'private task prose' }); assert.ok(!JSON.stringify(first.receipt).includes('private task prose')); - const second = resolve(repoRoot, { query: 'reworded' }, { previous: first.receipt }); + const second = resolve(repoRoot, { query: 'private task prose' }, { previous: first.receipt }); assert.deepEqual(second.selectedIds, first.selectedIds); assert.equal(second.reused, true); + const reworded = resolve(repoRoot, { query: 'reworded' }, { previous: first.receipt }); + assert.equal(reworded.reused, false); + assert.notEqual(reworded.receipt.bindingDigest, first.receipt.bindingDigest); assert.throws(() => resolve(repoRoot, {}, { previous: { ...first.receipt, selectedIds: ['skill:shared'] } }), /receipt/); const changed = resolve(repoRoot, { sessionId: 'session-2' }, { previous: first.receipt }); assert.equal(changed.reused, false); })); +test('trigger changes invalidate a pinned Auto receipt', () => withFixture(repoRoot => { + const first = resolve(repoRoot, { proposedIds: ['skill:feature'], query: 'feature work' }); + write(repoRoot, 'manifests/context-packs/skill-triggers@1.json', JSON.stringify({ + schemaVersion: 1, triggers: { 'skill:feature': ['feature work'] }, + })); + const second = resolve(repoRoot, { query: 'feature work' }, { previous: first.receipt }); + assert.equal(second.reused, false); + assert.notEqual(second.receipt.bindingDigest, first.receipt.bindingDigest); +})); + for (const [label, taskChanges, options] of [ ['task', { taskId: 'task-2' }, {}], ['revision', { revision: 2 }, {}], diff --git a/tests/lib/utils.test.js b/tests/lib/utils.test.js index f9921a503..a04c97c85 100644 --- a/tests/lib/utils.test.js +++ b/tests/lib/utils.test.js @@ -307,6 +307,17 @@ function runTests() { } })) passed++; else failed++; + if (test('sameRepoIdentity compares full-width filesystem IDs', () => { + const originalStat = fs.statSync; + try { + fs.statSync = (file, options) => { + assert.equal(options?.bigint, true); + return { dev: 1n, ino: file.endsWith('first') ? 9007199254740993n : 9007199254740994n }; + }; + assert.ok(!utils.sameRepoIdentity('/missing/first', '/missing/second')); + } finally { fs.statSync = originalStat; } + })) passed++; else failed++; + // sanitizeSessionId tests console.log('\nsanitizeSessionId:'); diff --git a/tests/scripts/generate-skill-triggers.test.js b/tests/scripts/generate-skill-triggers.test.js new file mode 100644 index 000000000..4bccc738e --- /dev/null +++ b/tests/scripts/generate-skill-triggers.test.js @@ -0,0 +1,18 @@ +'use strict'; + +const assert = require('node:assert/strict'); +const path = require('node:path'); +const { spawnSync } = require('node:child_process'); +const test = require('node:test'); + +const command = path.resolve(__dirname, '../../scripts/dev/generate-skill-triggers.js'); + +test('trigger generation rejects batch sizes that cannot advance', () => { + for (const value of ['0', '-1', 'NaN', '1.5', '9007199254740992']) { + const result = spawnSync(process.execPath, [command, '--batch', value, '--dry-run'], { + encoding: 'utf8', timeout: 5000, + }); + assert.equal(result.status, 1, `${value}: ${result.stderr}`); + assert.match(result.stderr, /--batch must be a positive integer/); + } +});