mirror of
https://github.com/affaan-m/ECC.git
synced 2026-09-28 04:25:11 +02:00
fix(profiles): bind routing receipts and harden review paths
This commit is contained in:
@@ -255,8 +255,10 @@ function createAuthLease(authHome) {
|
||||
if (!after.equals(original)) {
|
||||
JSON.parse(after.toString('utf8'));
|
||||
const temp = `${source}.${process.pid}.tmp`;
|
||||
fs.writeFileSync(temp, after, { flag: 'wx', mode: 0o600 });
|
||||
fs.renameSync(temp, source);
|
||||
try {
|
||||
fs.writeFileSync(temp, after, { flag: 'wx', mode: 0o600 });
|
||||
fs.renameSync(temp, source);
|
||||
} finally { fs.rmSync(temp, { force: true }); }
|
||||
}
|
||||
} catch { /* An unreadable refresh keeps the previous login; the next call reports any auth failure. */ }
|
||||
fs.rmSync(leased, { force: true });
|
||||
@@ -282,7 +284,7 @@ function readClaudeKeychainToken() {
|
||||
return token;
|
||||
}
|
||||
|
||||
function createClaudeProvider({ allowRealProvider = false, executable, model,
|
||||
function createClaudeProvider({ allowRealProvider = false, allowCredentialedTools = false, executable, model,
|
||||
apiKey = process.env.ANTHROPIC_API_KEY, oauthToken = process.env.CLAUDE_CODE_OAUTH_TOKEN,
|
||||
tokenSource = readClaudeKeychainToken, persistSessions = false, execute = spawnSync } = {}) {
|
||||
if (allowRealProvider !== true) throw new Error('Real provider requires explicit opt-in');
|
||||
@@ -301,6 +303,9 @@ function createClaudeProvider({ allowRealProvider = false, executable, model,
|
||||
const provider = request => {
|
||||
if (fingerprintExecutable(binary.path).digest !== pin.executableDigest) fail('source-drift');
|
||||
const selection = request.phase === 'selection';
|
||||
if (!selection && !allowCredentialedTools) {
|
||||
throw new Error('Claude task tools can read provider credentials; explicit credentialed-tool opt-in is required');
|
||||
}
|
||||
// Selection is tool-free and read-only; task execution may edit and run commands in the workspace.
|
||||
// Claude has no cwd-write sandbox flag, so containment relies on the isolated home and temp workspace.
|
||||
const args = ['--print', '--output-format', 'json',
|
||||
@@ -682,8 +687,9 @@ function outcomeTrial(item, arm, repeat, repoRoot, execute, cwd, environment, ta
|
||||
if (harvest) harvest(arm, `${item.id}--step${index + 1}`, repeat, environment);
|
||||
if (result.status !== 'completed') {
|
||||
// A failed ticket ends the chain; remaining tickets are unscored.
|
||||
for (let rest = index; rest < item.steps.length; rest++) {
|
||||
steps.push({ score: 0, ...(metrics ? metricsSince(metrics, start) : {}) });
|
||||
steps.push({ score: 0, ...(metrics ? metricsSince(metrics, start) : {}) });
|
||||
for (let rest = index + 1; rest < item.steps.length; rest++) {
|
||||
steps.push({ score: 0, ...(metrics ? metricsSince(metrics, metrics.length) : {}) });
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -748,7 +754,8 @@ function createHarvester(artifactDir, envs) {
|
||||
}
|
||||
|
||||
function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), registration,
|
||||
repeats = 1, provider, family, allowRealProvider = false, executable, model, effort, authHome, environments,
|
||||
repeats = 1, provider, family, allowRealProvider = false, allowCredentialedTools = false,
|
||||
executable, model, effort, authHome, environments,
|
||||
arms = undefined, artifactDir = null, maxCalls = 300, deadlineMs = 3600000, callTimeoutMs = 300000 } = {}) {
|
||||
if (!provider && !allowRealProvider) throw new Error('Evaluation requires an injected provider or explicit opt-in');
|
||||
if (!bounded(maxCalls, 1, 2000) || !bounded(deadlineMs, 1, 8 * 3600000)
|
||||
@@ -756,6 +763,9 @@ function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), re
|
||||
if (!provider && !registration) throw new Error('Real evaluation requires prior registration');
|
||||
const resolvedFamily = provider ? (family || 'codex') : resolveFamily(family, executable);
|
||||
if (resolvedFamily === 'claude' && effort !== undefined) throw new Error('Reasoning effort applies only to the Codex provider');
|
||||
if (!provider && resolvedFamily === 'claude' && !allowCredentialedTools) {
|
||||
throw new Error('Claude task tools can read provider credentials; explicit credentialed-tool opt-in is required');
|
||||
}
|
||||
const pin = preregister({ repoRoot, corpus, repeats, model, executable, effort, arms });
|
||||
if (!provider && resolvedFamily === 'codex' && pin.arms.includes('ecc-legacy')) {
|
||||
throw new Error('Codex real evaluation requires --arms without ecc-legacy; the pinned legacy skills arm is Claude-only');
|
||||
@@ -763,7 +773,8 @@ function runEvaluation({ repoRoot = DEFAULT_REPO_ROOT, corpus = loadCorpus(), re
|
||||
if (registration && !isDeepStrictEqual(registration, pin)) throw new Error('Registration pin mismatch');
|
||||
const injected = Boolean(provider);
|
||||
const liveProvider = provider || (resolvedFamily === 'claude'
|
||||
? createClaudeProvider({ allowRealProvider, executable, model, persistSessions: Boolean(artifactDir) })
|
||||
? createClaudeProvider({ allowRealProvider, allowCredentialedTools, executable, model,
|
||||
persistSessions: Boolean(artifactDir) })
|
||||
: createCodexProvider({ allowRealProvider, executable, model, effort, authHome }));
|
||||
const state = { calls: 0, metrics: [], maxCalls, callTimeoutMs, family: resolvedFamily,
|
||||
deadline: Date.now() + deadlineMs, provider: liveProvider,
|
||||
|
||||
@@ -5,7 +5,7 @@ const { preregister, runEvaluation, loadCorpus } = require('./ai-eval-lib');
|
||||
|
||||
function main(argv = process.argv.slice(2), injected = {}) {
|
||||
const flags = new Map();
|
||||
const switches = new Set(['--plan', '--allow-real-provider', '--help']);
|
||||
const switches = new Set(['--plan', '--allow-real-provider', '--allow-credentialed-tools', '--help']);
|
||||
const values = new Set(['--registration', '--model', '--executable', '--provider', '--auth-home', '--effort', '--repeats', '--max-calls', '--deadline-ms', '--artifact-dir', '--corpus', '--call-timeout-ms', '--arms']);
|
||||
for (let i = 0; i < argv.length; i++) {
|
||||
const flag = argv[i];
|
||||
@@ -14,10 +14,13 @@ function main(argv = process.argv.slice(2), injected = {}) {
|
||||
flags.set(flag, switches.has(flag) ? true : argv[++i]);
|
||||
}
|
||||
if (flags.has('--help')) {
|
||||
return { usage: 'ai-eval.js --plan [--corpus FILE] [--arms a,b] [--repeats N] [--model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL]] | --allow-real-provider --registration FILE --model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL (Codex only)] [--auth-home ABSOLUTE_DIR (Codex only)] [--corpus FILE] [--arms a,b] [--repeats N] [--max-calls N] [--deadline-ms N] [--call-timeout-ms N]. Claude auth: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, or the macOS Keychain login.' };
|
||||
return { usage: 'ai-eval.js --plan [--corpus FILE] [--arms a,b] [--repeats N] [--model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--effort LEVEL]] | --allow-real-provider --registration FILE --model MODEL --executable ABSOLUTE_PATH [--provider claude|codex] [--allow-credentialed-tools (Claude only)] [--effort LEVEL (Codex only)] [--auth-home ABSOLUTE_DIR (Codex only)] [--corpus FILE] [--arms a,b] [--repeats N] [--max-calls N] [--deadline-ms N] [--call-timeout-ms N]. Claude auth: CLAUDE_CODE_OAUTH_TOKEN, ANTHROPIC_API_KEY, or the macOS Keychain login.' };
|
||||
}
|
||||
if (flags.get('--provider') !== undefined && !['claude', 'codex'].includes(flags.get('--provider'))) throw new Error('Provider must be claude or codex');
|
||||
if (flags.get('--provider') === 'claude' && flags.has('--effort')) throw new Error('Reasoning effort applies only to the Codex provider');
|
||||
if (flags.has('--allow-credentialed-tools') && (!flags.has('--allow-real-provider') || flags.get('--provider') !== 'claude')) {
|
||||
throw new Error('Credentialed-tool opt-in requires a real Claude evaluation');
|
||||
}
|
||||
const repeats = flags.has('--repeats') ? Number(flags.get('--repeats')) : 1;
|
||||
const corpus = flags.has('--corpus') ? loadCorpus(flags.get('--corpus')) : undefined;
|
||||
const arms = flags.has('--arms') ? flags.get('--arms').split(',').map(a => a.trim()).filter(Boolean) : undefined;
|
||||
@@ -30,6 +33,7 @@ function main(argv = process.argv.slice(2), injected = {}) {
|
||||
if (!flags.has('--registration')) throw new Error('Evaluation requires a preregistration file');
|
||||
const registration = JSON.parse(fs.readFileSync(flags.get('--registration'), 'utf8'));
|
||||
return runEvaluation({ ...injected, registration, repeats, allowRealProvider: flags.has('--allow-real-provider'),
|
||||
allowCredentialedTools: flags.has('--allow-credentialed-tools'),
|
||||
executable: flags.get('--executable'), model: flags.get('--model'), family: flags.get('--provider'), effort: flags.get('--effort'), authHome: flags.get('--auth-home'),
|
||||
artifactDir: flags.get('--artifact-dir'), ...(corpus ? { corpus } : {}), ...(arms ? { arms } : {}),
|
||||
...(flags.has('--max-calls') ? { maxCalls: Number(flags.get('--max-calls')) } : {}),
|
||||
|
||||
@@ -136,7 +136,7 @@ node docker/context-profiles/ai-eval.js --plan \
|
||||
> registration.json
|
||||
|
||||
# 5. Run (requires your own Claude subscription login or API key).
|
||||
node docker/context-profiles/ai-eval.js --allow-real-provider \
|
||||
node docker/context-profiles/ai-eval.js --allow-real-provider --allow-credentialed-tools \
|
||||
--registration registration.json \
|
||||
--corpus docker/context-profiles/complex-corpus.json \
|
||||
--provider claude --model <model> --executable /absolute/path/to/claude \
|
||||
@@ -144,6 +144,11 @@ node docker/context-profiles/ai-eval.js --allow-real-provider \
|
||||
--artifact-dir /absolute/path/for/transcripts > report.json
|
||||
```
|
||||
|
||||
Claude task tools inherit the provider credential through the CLI process and can read it. Use
|
||||
`--allow-credentialed-tools` only with a trusted local corpus and credential. Without that
|
||||
explicit flag, real Claude task evaluation stops before a provider call; selection-only calls
|
||||
remain tool-free. This development evaluator does not provide a credential isolation boundary.
|
||||
|
||||
The registration digest binds the exact corpus, evaluator source, model, and
|
||||
executable; the run refuses to start if any of them drift, and aborts if the
|
||||
tree changes mid-run. `--artifact-dir` retains per-trial session transcripts
|
||||
|
||||
@@ -40,7 +40,7 @@ ecc profile start --state-root /absolute/dedicated/profile-store --native-root /
|
||||
|
||||
`resolve --state-root` uses the saved base, mode and exclusions. It rejects overrides and stale source generations. `mode` preserves the configured profile and explicit selections while recording the new mode transactionally.
|
||||
|
||||
A task input contains caller-assigned `sessionId`, `taskId`, positive integer `revision`, and `phase`. Optional fields are `query`, `explicitIds`, `proposedIds`, and `noWorkflow`. Increment revision for material task changes; keep it stable for rewording. Task prose is consumed locally and omitted from returned receipts.
|
||||
A task input contains caller-assigned `sessionId`, `taskId`, positive integer `revision`, and `phase`. Optional fields are `query`, `explicitIds`, `proposedIds`, and `noWorkflow`. Increment revision for material task changes. A changed query, including rewording, also invalidates selection reuse. Task prose is consumed locally and omitted from returned receipts.
|
||||
|
||||
```json
|
||||
{
|
||||
@@ -54,7 +54,7 @@ A task input contains caller-assigned `sessionId`, `taskId`, positive integer `r
|
||||
|
||||
Auto uses explicit user IDs first, then a completed pinned decision, an unambiguous ranked match, one cited skill name, or admitted agent-proposed IDs. Ambiguous free text shortlists up to five candidates for a bounded proposal. Manual uses explicit IDs; suggest emits a proposal without bodies. `--load` returns selected UTF-8 instructions and declared required resources, capped at 32,000 bytes across at most eight skills. `--task-input -` accepts one UTF-8 JSON object on standard input, capped at 65,536 bytes. These byte caps are output and transport bounds, not native tokenizer results.
|
||||
|
||||
Save the returned `selection.receipt` as a separate JSON document to use `--previous receipt.json`. `--expected-digest` can bind a load to a prior selection digest. Source, routing-policy version, profile, mode, exclusions, session, task revision and phase invalidate stale reuse. A pending proposal cannot be reused as a completed decision. Receipts are integrity checks for local operation, not an authorization signature.
|
||||
Save the returned `selection.receipt` as a separate JSON document to use `--previous receipt.json`. `--expected-digest` can bind a load to a prior selection digest. Source, trigger content, routing-policy version, profile, mode, exclusions, session, task revision, phase, and a digest of the query invalidate stale reuse. A pending proposal cannot be reused as a completed decision. Receipts are integrity checks for local operation, not an authorization signature.
|
||||
|
||||
An agent can call the resolver at task boundaries and read the returned context. This integration is prompt-advisory. Returning a body never grants tools, invokes shell interpolation, starts a native skill, changes hooks or installs dependencies. Native manual-only flags and authority-bearing metadata are checked before selection. Base profiles remain stable during task routing.
|
||||
|
||||
|
||||
@@ -43,6 +43,10 @@ function parseFlags(argv) {
|
||||
flags[arg.slice(2).replace(/-([a-z])/g, (_, c) => c.toUpperCase())] = argv[index += 1];
|
||||
} else throw new Error(`Unknown flag: ${arg}`);
|
||||
}
|
||||
if (!/^[1-9][0-9]*$/.test(String(flags.batch)) || !Number.isSafeInteger(Number(flags.batch))) {
|
||||
throw new Error('--batch must be a positive integer');
|
||||
}
|
||||
flags.batch = Number(flags.batch);
|
||||
return flags;
|
||||
}
|
||||
|
||||
|
||||
@@ -154,7 +154,9 @@ function resolveTaskContext({ repoRoot = DEFAULT_REPO_ROOT, task, profileId = 'l
|
||||
if (excluded.has(id)) throw new Error(`Context ID is excluded: ${id}`);
|
||||
});
|
||||
const taskBinding = { sessionId: task.sessionId, taskId: task.taskId, revision: task.revision, phase: task.phase };
|
||||
const bindingDigest = digestObject({ ...taskBinding, planDigest: plan.planDigest, routingPolicyVersion: ROUTING_POLICY_VERSION });
|
||||
const bindingDigest = digestObject({ ...taskBinding, planDigest: plan.planDigest,
|
||||
routingPolicyVersion: ROUTING_POLICY_VERSION, triggersDigest: digestObject(triggers),
|
||||
queryDigest: digestObject(task.query || '') });
|
||||
const reused = Boolean(previous && previous.bindingDigest === bindingDigest && !task.noWorkflow
|
||||
&& ['selected', 'none'].includes(previous.decision) && !explicitIds.length && !proposedIds.length);
|
||||
const admissible = id => {
|
||||
|
||||
@@ -195,9 +195,11 @@ function sameRepoIdentity(a, b) {
|
||||
if (!a || !b) return false;
|
||||
if (normalizeRepoPath(a) === normalizeRepoPath(b)) return true;
|
||||
try {
|
||||
const sa = fs.statSync(a);
|
||||
const sb = fs.statSync(b);
|
||||
return sa.ino !== 0 && sa.dev === sb.dev && sa.ino === sb.ino;
|
||||
// Windows file IDs can exceed Number.MAX_SAFE_INTEGER; rounded IDs may
|
||||
// otherwise make distinct files look identical.
|
||||
const sa = fs.statSync(a, { bigint: true });
|
||||
const sb = fs.statSync(b, { bigint: true });
|
||||
return sa.ino !== 0n && sa.dev === sb.dev && sa.ino === sb.ino;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -187,6 +187,13 @@ test('subscription lease copies private auth into the call home, returns refresh
|
||||
});
|
||||
assert.equal(fs.existsSync(path.join(codexHome, 'auth.json')), false);
|
||||
assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"new"}}');
|
||||
const staleTemp = path.join(authHome, `auth.json.${process.pid}.tmp`);
|
||||
fs.writeFileSync(staleTemp, 'stale', { mode: 0o600 });
|
||||
lease.run(codexHome, () => fs.writeFileSync(path.join(codexHome, 'auth.json'), '{"tokens":{"refresh_token":"latest"}}'));
|
||||
assert.equal(fs.existsSync(staleTemp), false);
|
||||
assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"new"}}');
|
||||
lease.run(codexHome, () => fs.writeFileSync(path.join(codexHome, 'auth.json'), '{"tokens":{"refresh_token":"latest"}}'));
|
||||
assert.equal(fs.readFileSync(path.join(authHome, 'auth.json'), 'utf8'), '{"tokens":{"refresh_token":"latest"}}');
|
||||
assert.throws(() => lease.run(codexHome, () => { throw new Error('provider crashed'); }), /crashed/);
|
||||
assert.equal(fs.existsSync(path.join(codexHome, 'auth.json')), false);
|
||||
fs.chmodSync(authHome, 0o755);
|
||||
@@ -334,7 +341,11 @@ test('Claude provider runs tool-free selection and permissioned tasks with a san
|
||||
const request = { phase: 'selection', input: 'request', cwd, timeoutMs: 5, maxBuffer: 1000,
|
||||
env: { PATH: '/bin', HOME: cwd, CLAUDE_CONFIG_DIR: path.join(cwd, 'cfg'), TMPDIR: '/tmp', CODEX_HOME: '/tmp/x', SECRET: 's' } };
|
||||
provider(request);
|
||||
provider({ ...request, phase: 'task' });
|
||||
assert.throws(() => provider({ ...request, phase: 'task' }), /credentialed-tool opt-in/);
|
||||
const credentialed = createClaudeProvider({ allowRealProvider: true, allowCredentialedTools: true,
|
||||
executable: process.execPath, model: 'pinned-model', oauthToken: 'test-token', tokenSource: null,
|
||||
execute(command, args, options) { calls.push({ args, options }); return { status: 0, stdout: claudeJson() }; } });
|
||||
credentialed({ ...request, phase: 'task' });
|
||||
assert.ok(calls[0].args.includes('--tools'));
|
||||
assert.ok(!calls[0].args.join(' ').includes('bypassPermissions'));
|
||||
assert.ok(calls[1].args.includes('--permission-mode') && calls[1].args.includes('bypassPermissions'));
|
||||
|
||||
@@ -104,14 +104,27 @@ test('authority-bearing metadata cannot become automatic invocation', () => with
|
||||
test('receipt pins source and task identity without retaining query text', () => withFixture(repoRoot => {
|
||||
const first = resolve(repoRoot, { proposedIds: ['skill:feature'], query: 'private task prose' });
|
||||
assert.ok(!JSON.stringify(first.receipt).includes('private task prose'));
|
||||
const second = resolve(repoRoot, { query: 'reworded' }, { previous: first.receipt });
|
||||
const second = resolve(repoRoot, { query: 'private task prose' }, { previous: first.receipt });
|
||||
assert.deepEqual(second.selectedIds, first.selectedIds);
|
||||
assert.equal(second.reused, true);
|
||||
const reworded = resolve(repoRoot, { query: 'reworded' }, { previous: first.receipt });
|
||||
assert.equal(reworded.reused, false);
|
||||
assert.notEqual(reworded.receipt.bindingDigest, first.receipt.bindingDigest);
|
||||
assert.throws(() => resolve(repoRoot, {}, { previous: { ...first.receipt, selectedIds: ['skill:shared'] } }), /receipt/);
|
||||
const changed = resolve(repoRoot, { sessionId: 'session-2' }, { previous: first.receipt });
|
||||
assert.equal(changed.reused, false);
|
||||
}));
|
||||
|
||||
test('trigger changes invalidate a pinned Auto receipt', () => withFixture(repoRoot => {
|
||||
const first = resolve(repoRoot, { proposedIds: ['skill:feature'], query: 'feature work' });
|
||||
write(repoRoot, 'manifests/context-packs/skill-triggers@1.json', JSON.stringify({
|
||||
schemaVersion: 1, triggers: { 'skill:feature': ['feature work'] },
|
||||
}));
|
||||
const second = resolve(repoRoot, { query: 'feature work' }, { previous: first.receipt });
|
||||
assert.equal(second.reused, false);
|
||||
assert.notEqual(second.receipt.bindingDigest, first.receipt.bindingDigest);
|
||||
}));
|
||||
|
||||
for (const [label, taskChanges, options] of [
|
||||
['task', { taskId: 'task-2' }, {}],
|
||||
['revision', { revision: 2 }, {}],
|
||||
|
||||
@@ -307,6 +307,17 @@ function runTests() {
|
||||
}
|
||||
})) passed++; else failed++;
|
||||
|
||||
if (test('sameRepoIdentity compares full-width filesystem IDs', () => {
|
||||
const originalStat = fs.statSync;
|
||||
try {
|
||||
fs.statSync = (file, options) => {
|
||||
assert.equal(options?.bigint, true);
|
||||
return { dev: 1n, ino: file.endsWith('first') ? 9007199254740993n : 9007199254740994n };
|
||||
};
|
||||
assert.ok(!utils.sameRepoIdentity('/missing/first', '/missing/second'));
|
||||
} finally { fs.statSync = originalStat; }
|
||||
})) passed++; else failed++;
|
||||
|
||||
// sanitizeSessionId tests
|
||||
console.log('\nsanitizeSessionId:');
|
||||
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
'use strict';
|
||||
|
||||
const assert = require('node:assert/strict');
|
||||
const path = require('node:path');
|
||||
const { spawnSync } = require('node:child_process');
|
||||
const test = require('node:test');
|
||||
|
||||
const command = path.resolve(__dirname, '../../scripts/dev/generate-skill-triggers.js');
|
||||
|
||||
test('trigger generation rejects batch sizes that cannot advance', () => {
|
||||
for (const value of ['0', '-1', 'NaN', '1.5', '9007199254740992']) {
|
||||
const result = spawnSync(process.execPath, [command, '--batch', value, '--dry-run'], {
|
||||
encoding: 'utf8', timeout: 5000,
|
||||
});
|
||||
assert.equal(result.status, 1, `${value}: ${result.stderr}`);
|
||||
assert.match(result.stderr, /--batch must be a positive integer/);
|
||||
}
|
||||
});
|
||||
Reference in New Issue
Block a user