From 76d34cd066923c2c8d0bbdfaeb80ccfaed58b4d1 Mon Sep 17 00:00:00 2001 From: sv-tmueller Date: Thu, 1 Oct 2026 13:00:28 +0200 Subject: [PATCH 1/2] test: require each workflow meta to be a literal first statement The workflow runtime reads meta before running the script, so it has to be a plain literal at byte 0. The test also checks the literal against SPEC and TIER_MODELS so the copy cannot drift. Co-Authored-By: Claude Sonnet 5.5 --- .../__tests__/workflow-meta.test.mjs | 106 ++++++++++++++++++ 1 file changed, 106 insertions(+) create mode 100644 .claude/workflows/__tests__/workflow-meta.test.mjs diff --git a/.claude/workflows/__tests__/workflow-meta.test.mjs b/.claude/workflows/__tests__/workflow-meta.test.mjs new file mode 100644 index 0000000..7095437 --- /dev/null +++ b/.claude/workflows/__tests__/workflow-meta.test.mjs @@ -0,0 +1,106 @@ +/** + * Guard for the `meta` export of each workflow script (issue #424). + * + * The Claude Code workflow runtime reads `meta` before it runs the script, so + * it has to be a plain literal and the first statement of the file. It cannot + * be computed from SPEC. This test pins that shape for every workflow script + * and checks the literal stays in sync with SPEC and TIER_MODELS, which are + * checked in turn against the JSON spec and the adapter table elsewhere. + */ + +import { test, describe } from 'node:test' +import assert from 'node:assert/strict' +import { readFileSync, readdirSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { join, dirname } from 'node:path' +import { createContext, runInContext } from 'node:vm' + +const __dir = dirname(fileURLToPath(import.meta.url)) +const workflowsDir = join(__dir, '..') + +const META_START = 'export const meta = {' + +// Listed from disk so a new workflow cannot skip the check. +const WORKFLOW_FILES = readdirSync(workflowsDir).filter((f) => f.endsWith('.js')) + +// Return the source text of the first brace-balanced object literal that +// starts at or after `from`. +function braceMatch(src, from) { + let pos = src.indexOf('{', from) + const start = pos + let depth = 0 + while (pos < src.length) { + if (src[pos] === '{') depth++ + if (src[pos] === '}') depth-- + pos++ + if (depth === 0) break + } + return src.slice(start, pos) +} + +function evalLiteral(literal) { + return runInContext('(' + literal + ')', createContext({})) +} + +// Objects built in a vm context come from another realm, so deepStrictEqual +// fails on them. A JSON round-trip moves both sides into this realm. +const plain = (value) => JSON.parse(JSON.stringify(value)) + +describe('workflow scripts: meta is a literal first statement', () => { + test('the workflow scripts are found on disk', () => { + assert.ok(WORKFLOW_FILES.length >= 3, `found ${WORKFLOW_FILES.join(', ')}`) + }) + + for (const file of WORKFLOW_FILES) { + const src = readFileSync(join(workflowsDir, file), 'utf8') + + test(`${file}: starts at byte 0 with the meta export`, () => { + assert.ok( + src.startsWith(META_START), + `${file}: must begin with "${META_START}" (no header comment, no other statement above it)` + ) + }) + + test(`${file}: meta is a plain literal with no calls or references`, () => { + const idx = src.indexOf(META_START) + assert.notEqual(idx, -1, `${file}: no meta export found`) + const literal = braceMatch(src, idx) + // Evaluating in an empty context is not enough: builtins such as + // Object.freeze and Array.prototype.map are always there. So strip the + // quoted strings and the `key:` names, then require only punctuation + // to remain. That rejects calls, identifiers as values, arrows, spreads, + // template literals and comments. + const stripped = literal + .replace(/'[^'\\\n]*'/g, '') + .replace(/"[^"\\\n]*"/g, '') + .replace(/[A-Za-z_$][\w$]*\s*:/g, '') + assert.match( + stripped, + /^[\s{}[\],]*$/, + `${file}: meta must hold only quoted strings, object keys, braces, brackets and commas; left over: ${stripped.replace(/\s+/g, ' ').trim()}` + ) + assert.doesNotThrow(() => evalLiteral(literal), `${file}: meta does not evaluate in an empty context`) + }) + + test(`${file}: meta matches SPEC with tiers resolved through TIER_MODELS`, () => { + const metaIdx = src.indexOf(META_START) + assert.notEqual(metaIdx, -1, `${file}: no meta export found`) + const meta = evalLiteral(braceMatch(src, metaIdx)) + + const specIdx = src.indexOf('const SPEC = {') + assert.notEqual(specIdx, -1, `${file}: no SPEC found`) + const spec = evalLiteral(braceMatch(src, specIdx)) + + const modelsIdx = src.indexOf('const TIER_MODELS = {') + assert.notEqual(modelsIdx, -1, `${file}: no TIER_MODELS found`) + const models = evalLiteral(braceMatch(src, modelsIdx)) + + const expected = { + name: spec.name, + description: spec.description, + phases: spec.phases.map((p) => ({ title: p.title, detail: p.detail, model: models[p.tier] })), + } + assert.deepStrictEqual(plain(meta), plain(expected), `${file}: meta drifted from SPEC / TIER_MODELS`) + }) + } +}) From 1afc2840427db9279e63ffcf25c5b0ab54b8c0b6 Mon Sep 17 00:00:00 2001 From: sv-tmueller Date: Thu, 1 Oct 2026 13:01:14 +0200 Subject: [PATCH 2/2] fix: make each workflow meta a literal first statement The workflow runtime reads meta before it runs the script, so a meta computed from SPEC below it is not read reliably. Move meta to the top as a plain literal with the values the old expression produced. SPEC is unchanged. Bump plugin 2.9.0 -> 2.9.1. Co-Authored-By: Claude Sonnet 5.5 --- .claude/.claude-plugin/plugin.json | 2 +- .claude/workflows/tm-map-codebase.js | 26 ++++++++++++++--------- .claude/workflows/tm-review-changes.js | 28 ++++++++++++++----------- .claude/workflows/tm-review-codebase.js | 26 ++++++++++++++--------- 4 files changed, 49 insertions(+), 33 deletions(-) diff --git a/.claude/.claude-plugin/plugin.json b/.claude/.claude-plugin/plugin.json index 54abfb7..10d8549 100644 --- a/.claude/.claude-plugin/plugin.json +++ b/.claude/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "orchestrai", "displayName": "OrchestrAI", - "version": "2.9.0", + "version": "2.9.1", "description": "Orchestrator team: the role agents, the tm- operational skills, and the review workflows.", "author": { "name": "Thomas Mueller" diff --git a/.claude/workflows/tm-map-codebase.js b/.claude/workflows/tm-map-codebase.js index 05b6252..e2d1ca7 100644 --- a/.claude/workflows/tm-map-codebase.js +++ b/.claude/workflows/tm-map-codebase.js @@ -1,3 +1,19 @@ +export const meta = { + name: 'tm-map-codebase', + description: + 'Token-bounded full-repo map: a Sonnet scout splits the repo into N coherent areas (N sized to the repo, capped at a ceiling), one Sonnet worker maps each area (purpose, entry points, key modules, data and control flow, external dependencies, conventions), and one Opus critic synthesizes a dated architecture map. Models are pinned per stage in-script, so it never inherits the session model, and the agent count scales with repo size only up to a hard ceiling. This is a map, not a review: no findings, no severities, no recommendations.', + phases: [ + { title: 'Scout', detail: 'one Sonnet agent splits the repo into N areas (N <= ceiling)', model: 'sonnet' }, + { title: 'Map', detail: 'one Sonnet worker per area describes it', model: 'sonnet' }, + { title: 'Synthesize', detail: 'one Opus critic writes the consolidated map', model: 'opus' }, + ], +} +// meta is a plain literal and the first statement because the workflow runtime +// reads it before running the script. It repeats the name, description and +// phases of SPEC below, with each phase's tier resolved through TIER_MODELS. +// workflow-meta.test.mjs fails if the two drift. +// The model is carried here for display purposes only. + // Workflow spec (embedded; mirrors specs/tm-map-codebase.spec.json) // The spec encodes the fan-out as data: stages, tiers, parallelism, schemas, // and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase() @@ -51,16 +67,6 @@ const SPEC = { const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' } const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' } -export const meta = { - name: SPEC.name, - description: SPEC.description, - phases: SPEC.phases.map((p) => ({ - title: p.title, - detail: p.detail, - model: TIER_MODELS[p.tier], - })), -} - // Bounded by construction. The scout sizes the number of areas N to the repo, and // the script hard-clamps N to MAX_AREAS, so a run is 1 scout + N area workers + // 1 critic = N + 2 agents, with N <= MAX_AREAS. The count scales with repo size diff --git a/.claude/workflows/tm-review-changes.js b/.claude/workflows/tm-review-changes.js index 97a288c..c120978 100644 --- a/.claude/workflows/tm-review-changes.js +++ b/.claude/workflows/tm-review-changes.js @@ -1,3 +1,19 @@ +export const meta = { + name: 'tm-review-changes', + description: + 'Token-bounded code review: Sonnet workers review the diff across fixed dimensions, one Opus critic consolidates. Models are pinned per stage in-script, so it never inherits the session model or fans out unboundedly.', + phases: [ + { title: 'Review', detail: 'one Sonnet worker per dimension', model: 'sonnet' }, + { title: 'Verify', detail: 'one adversarial Sonnet worker per must-fix finding, capped', model: 'sonnet' }, + { title: 'Consolidate', detail: 'one Opus critic verifies and merges findings', model: 'opus' }, + ], +} +// meta is a plain literal and the first statement because the workflow runtime +// reads it before running the script. It repeats the name, description and +// phases of SPEC below, with each phase's tier resolved through TIER_MODELS. +// workflow-meta.test.mjs fails if the two drift. +// The model is carried here for display purposes only. + // Workflow spec (embedded; mirrors specs/tm-review-changes.spec.json) // The spec encodes the fan-out as data: stages, tiers, parallelism, schemas, // and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase() @@ -53,18 +69,6 @@ const SPEC = { const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' } const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' } -export const meta = { - name: SPEC.name, - description: SPEC.description, - phases: SPEC.phases.map((p) => ({ - title: p.title, - detail: p.detail, - // The adapter table resolves the tier to a concrete model for the - // Claude Code host. meta.phases carries the model for display purposes. - model: TIER_MODELS[p.tier], - })), -} - // Bounded by construction. The dimension list is fixed, there is no per-file // fan-out and no loop, so a run is DIMENSIONS.length Sonnet reviewers, plus one // Sonnet verifier per must-fix finding (at most min(must-fix count, MAX_VERIFY), diff --git a/.claude/workflows/tm-review-codebase.js b/.claude/workflows/tm-review-codebase.js index 1a42904..ceb74c4 100644 --- a/.claude/workflows/tm-review-codebase.js +++ b/.claude/workflows/tm-review-codebase.js @@ -1,3 +1,19 @@ +export const meta = { + name: 'tm-review-codebase', + description: + 'Token-bounded full-repo review: a Sonnet scout splits the repo into N coherent areas (N sized to the repo, capped at a ceiling), one Sonnet worker reviews each area plus one Sonnet architecture worker audits repo-wide structure, and one Opus critic verifies, writes a dated report, and consolidates. Models are pinned per stage in-script, so it never inherits the session model, and the agent count scales with repo size only up to a hard ceiling.', + phases: [ + { title: 'Scout', detail: 'one Sonnet agent splits the repo into N areas (N <= ceiling)', model: 'sonnet' }, + { title: 'Review', detail: 'one Sonnet worker per area plus one architecture worker', model: 'sonnet' }, + { title: 'Consolidate', detail: 'one Opus critic verifies, writes the report, consolidates', model: 'opus' }, + ], +} +// meta is a plain literal and the first statement because the workflow runtime +// reads it before running the script. It repeats the name, description and +// phases of SPEC below, with each phase's tier resolved through TIER_MODELS. +// workflow-meta.test.mjs fails if the two drift. +// The model is carried here for display purposes only. + // Workflow spec (embedded; mirrors specs/tm-review-codebase.spec.json) // The spec encodes the fan-out as data: stages, tiers, parallelism, schemas, // and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase() @@ -59,16 +75,6 @@ const SPEC = { const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' } const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' } -export const meta = { - name: SPEC.name, - description: SPEC.description, - phases: SPEC.phases.map((p) => ({ - title: p.title, - detail: p.detail, - model: TIER_MODELS[p.tier], - })), -} - // Bounded by construction. The scout sizes the number of areas N to the repo, and // the script hard-clamps N to MAX_AREAS, so a run is 1 scout + (N area workers + // 1 architecture worker) + 1 critic = N + 3 agents, with N <= MAX_AREAS. The