Files
ECC/tests/scripts/instinct-cli-evolve-generate.test.js
T
8a97868b5b fix(continuous-learning): /evolve never produces skill or agent candidates (#2664)
* fix(continuous-learning): cluster instincts by keyword overlap in /evolve

`cmd_evolve` grouped instincts by exact string equality of the whole
normalized trigger sentence. Triggers are free-form sentences, so every
instinct landed in its own bucket and `skill_candidates` was always empty.
`agent_candidates` is derived from `skill_candidates`, so agents never
generated either — `/evolve --generate` could only ever emit commands.

Measured on a 42-instinct project: 42 instincts produced 42 unique cluster
keys, largest cluster size 1.

Group on keyword overlap instead. Jaccard is the wrong metric here — trigger
keyword sets average ~7 words, so even clearly related pairs top out around
0.33 — so this uses the overlap coefficient (shared / smaller set) at 0.5,
plus a floor of 2 shared keywords so one incidental word cannot pull
unrelated instincts together. The same 42 instincts now yield 4 clusters.

Also unify the command/agent slug used by the preview and the writer. The
preview called `.replace('a ', '')`, which strips "a " anywhere in the
string, mangling "extracting data from Reddit" into
`/extracting-datfrom-R` while `--generate` wrote `extracting-data-from.md`.
Both paths now share `_evolved_command_name()` / `_evolved_agent_name()`.

Adds tests/scripts/instinct-cli-evolve.test.js, which fails on the previous
implementation (0 clusters instead of 1; preview name `extracting-datfrom-R`)
and covers the negative cases so unrelated triggers still stay apart.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* docs(continuous-learning): correct clustering metric name in docstring

The docstring said "Jaccard" while the implementation uses the overlap
coefficient, which is the point of the change.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

* fix(continuous-learning): generate every evolve candidate and cut slugs on word boundaries

_generate_evolved() wrote only skill_candidates[:5], workflow_instincts[:5]
and agent_candidates[:3]. On a project with 36 command candidates that meant
5 files and no warning, so the output read as complete while 86% of the
candidates were dropped.

Generation is now unbounded by default and takes a --limit N flag for callers
that want a cap. A cap that truncates says so:

  Note: writing 3 of 36 command candidates (--limit 3); 33 skipped.

The analysis preview keeps showing five per kind but now names the remainder
("... and 31 more command candidates not shown") instead of presenting a
sample as the whole set.

Slugs were also cut with a hard slice, which split words mid-token and
produced /investigating-comple, /learning-about-compl and
/researching-mechanis. _truncate_slug() retreats to the last separator that
fits, and keeps the full head when the cut already lands on one, so
"analyzing large text files" stays /analyzing-large-text rather than losing
a word. A first word longer than the limit still falls back to a hard cut
because no boundary is available.

Shorter slugs collide more easily, and a collision used to mean one file
silently overwriting another. _assign_unique_slugs() suffixes duplicates
(-2, -3) and is called by both the preview and the writer over the same
ordered list, so advertised names and written names cannot drift apart.

Skill directory naming moved to _evolved_skill_name(); it previously used its
own inline slug expression, so it was the one truncation the shared helper
did not cover.

Adds tests/scripts/instinct-cli-evolve-generate.test.js: 7 cases covering
word-boundary cuts, the separator-aligned cut, unbounded generation, --limit
reporting, collision dedup, preview remainder and preview/writer agreement.
Six of the seven fail against the previous implementation.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
Co-authored-by: haelyra <49814733+haelyra@users.noreply.github.com>
2026-08-04 16:57:24 -04:00

250 lines
7.4 KiB
JavaScript

const assert = require('assert');
const fs = require('fs');
const os = require('os');
const path = require('path');
const { spawnSync } = require('child_process');
let passed = 0;
let failed = 0;
const repoRoot = path.resolve(__dirname, '..', '..');
const cliPath = path.join(
repoRoot,
'skills',
'continuous-learning-v2',
'scripts',
'instinct-cli.py'
);
function detectPython3() {
for (const bin of ['python3', 'python']) {
const r = spawnSync(bin, ['--version'], { encoding: 'utf8' });
if (r.status === 0 && /Python 3/.test(r.stdout + r.stderr)) return bin;
}
return null;
}
const PYTHON3 = detectPython3();
if (!PYTHON3) {
console.log('\n=== Testing instinct-cli.py evolve generation ===\n');
console.log(' - skipped: Python 3 not found in PATH');
console.log('\nPassed: 0');
console.log('Failed: 0');
process.exit(0);
}
function test(name, fn) {
try {
fn();
console.log(` ✓ ${name}`);
passed += 1;
} catch (error) {
console.log(` ✗ ${name}`);
console.log(` Error: ${error.message}`);
failed += 1;
}
}
function createTempDir() {
return fs.mkdtempSync(path.join(os.tmpdir(), 'ecc-instinct-cli-evolve-'));
}
function cleanupDir(dir) {
fs.rmSync(dir, { recursive: true, force: true });
}
function writeInstinct(root, id, trigger, confidence = 0.8, domain = 'workflow') {
const dir = path.join(root, 'instincts', 'personal');
fs.mkdirSync(dir, { recursive: true });
fs.writeFileSync(
path.join(dir, `${id}.yaml`),
[
'---',
`id: ${id}`,
`trigger: "${trigger}"`,
`confidence: ${confidence}`,
`domain: ${domain}`,
'---',
'',
`## Action`,
'',
`Action for ${id}.`,
'',
].join('\n')
);
}
// CLV2_NO_PROJECT pins the run to global scope, so seeded instincts live in
// <root>/instincts/personal and generated files land in <root>/evolved.
function runCli(root, args) {
return spawnSync(PYTHON3, [cliPath, ...args], {
cwd: repoRoot,
encoding: 'utf8',
env: {
...process.env,
CLV2_HOMUNCULUS_DIR: root,
CLV2_NO_PROJECT: '1',
HOME: path.join(root, 'home'),
USERPROFILE: path.join(root, 'home'),
CLAUDE_PROJECT_DIR: '',
},
});
}
function generatedCommands(root) {
const dir = path.join(root, 'evolved', 'commands');
if (!fs.existsSync(dir)) return [];
return fs.readdirSync(dir).sort();
}
// Eight unrelated workflow triggers: no two share enough keywords to cluster,
// so each one is its own command candidate.
const EIGHT_TRIGGERS = [
['run-tests', 'when running tests'],
['build-images', 'when building images'],
['deploy-services', 'when deploying services'],
['profile-memory', 'when profiling memory'],
['rotate-secrets', 'when rotating secrets'],
['tag-releases', 'when tagging releases'],
['prune-caches', 'when pruning caches'],
['review-requests', 'when reviewing pull requests'],
];
function seedEight(root) {
for (const [id, trigger] of EIGHT_TRIGGERS) {
writeInstinct(root, id, trigger);
}
}
console.log('\n=== Testing instinct-cli.py evolve generation ===\n');
test('generated command names are cut on a word boundary', () => {
const root = createTempDir();
try {
writeInstinct(root, 'archaeology', 'when investigating complex systems');
writeInstinct(root, 'codebases', 'when learning about complex codebases');
writeInstinct(root, 'large-text', 'when analyzing large text files');
const result = runCli(root, ['evolve', '--generate']);
assert.strictEqual(result.status, 0, result.stderr);
const names = generatedCommands(root);
// A hard slice used to yield investigating-comple.md and
// learning-about-compl.md, which read as typos.
assert.deepStrictEqual(names, [
'analyzing-large-text.md',
'investigating.md',
'learning-about.md',
]);
} finally {
cleanupDir(root);
}
});
test('a cut landing on a separator keeps the whole word', () => {
const root = createTempDir();
try {
writeInstinct(root, 'a', 'when analyzing large text files');
writeInstinct(root, 'b', 'when running tests');
writeInstinct(root, 'c', 'when building images');
assert.strictEqual(runCli(root, ['evolve', '--generate']).status, 0);
// "analyzing-large-text" is exactly the slug limit and ends on a word, so
// nothing further may be dropped.
assert.ok(generatedCommands(root).includes('analyzing-large-text.md'));
} finally {
cleanupDir(root);
}
});
test('every command candidate is generated, not just the first five', () => {
const root = createTempDir();
try {
seedEight(root);
const result = runCli(root, ['evolve', '--generate']);
assert.strictEqual(result.status, 0, result.stderr);
assert.strictEqual(
generatedCommands(root).length,
EIGHT_TRIGGERS.length,
'a fixed cap silently dropped candidates'
);
} finally {
cleanupDir(root);
}
});
test('--limit caps generation and reports what it skipped', () => {
const root = createTempDir();
try {
seedEight(root);
const result = runCli(root, ['evolve', '--generate', '--limit', '3']);
assert.strictEqual(result.status, 0, result.stderr);
assert.strictEqual(generatedCommands(root).length, 3);
assert.match(result.stdout, /writing 3 of 8 command candidates/);
assert.match(result.stdout, /5 skipped/);
} finally {
cleanupDir(root);
}
});
test('colliding slugs produce distinct files instead of overwriting', () => {
const root = createTempDir();
try {
// Both triggers trim to "investigating".
writeInstinct(root, 'first', 'when investigating complex systems');
writeInstinct(root, 'second', 'when investigating extraordinarily convoluted pipelines');
writeInstinct(root, 'third', 'when running tests');
assert.strictEqual(runCli(root, ['evolve', '--generate']).status, 0);
const names = generatedCommands(root);
assert.ok(names.includes('investigating.md'), `missing base name in ${names}`);
assert.ok(names.includes('investigating-2.md'), `missing deduped name in ${names}`);
assert.strictEqual(new Set(names).size, names.length);
} finally {
cleanupDir(root);
}
});
test('preview states how many candidates it left out', () => {
const root = createTempDir();
try {
seedEight(root);
const result = runCli(root, ['evolve']);
assert.strictEqual(result.status, 0, result.stderr);
assert.match(result.stdout, /COMMAND CANDIDATES \(8\)/);
assert.match(result.stdout, /and 3 more command candidates not shown/);
} finally {
cleanupDir(root);
}
});
test('preview names match the files --generate writes', () => {
const root = createTempDir();
try {
writeInstinct(root, 'first', 'when investigating complex systems');
writeInstinct(root, 'second', 'when investigating extraordinarily convoluted pipelines');
writeInstinct(root, 'third', 'when running tests');
const preview = runCli(root, ['evolve']);
assert.strictEqual(preview.status, 0, preview.stderr);
assert.match(preview.stdout, /\/investigating\b/);
assert.match(preview.stdout, /\/investigating-2\b/);
assert.strictEqual(runCli(root, ['evolve', '--generate']).status, 0);
const names = generatedCommands(root);
assert.ok(names.includes('investigating.md'));
assert.ok(names.includes('investigating-2.md'));
} finally {
cleanupDir(root);
}
});
console.log(`\nPassed: ${passed}`);
console.log(`Failed: ${failed}`);
process.exit(failed > 0 ? 1 : 0);