Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .claude/.claude-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"name": "orchestrai",
"displayName": "OrchestrAI",
"version": "2.9.0",
"version": "2.9.1",
"description": "Orchestrator team: the role agents, the tm- operational skills, and the review workflows.",
"author": {
"name": "Thomas Mueller"
Expand Down
106 changes: 106 additions & 0 deletions .claude/workflows/__tests__/workflow-meta.test.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
/**
* Guard for the `meta` export of each workflow script (issue #424).
*
* The Claude Code workflow runtime reads `meta` before it runs the script, so
* it has to be a plain literal and the first statement of the file. It cannot
* be computed from SPEC. This test pins that shape for every workflow script
* and checks the literal stays in sync with SPEC and TIER_MODELS, which are
* checked in turn against the JSON spec and the adapter table elsewhere.
*/

import { test, describe } from 'node:test'
import assert from 'node:assert/strict'
import { readFileSync, readdirSync } from 'node:fs'
import { fileURLToPath } from 'node:url'
import { join, dirname } from 'node:path'
import { createContext, runInContext } from 'node:vm'

const __dir = dirname(fileURLToPath(import.meta.url))
const workflowsDir = join(__dir, '..')

const META_START = 'export const meta = {'

// Listed from disk so a new workflow cannot skip the check.
const WORKFLOW_FILES = readdirSync(workflowsDir).filter((f) => f.endsWith('.js'))

// Return the source text of the first brace-balanced object literal that
// starts at or after `from`.
function braceMatch(src, from) {
let pos = src.indexOf('{', from)
const start = pos
let depth = 0
while (pos < src.length) {
if (src[pos] === '{') depth++
if (src[pos] === '}') depth--
pos++
if (depth === 0) break
}
return src.slice(start, pos)
}

function evalLiteral(literal) {
return runInContext('(' + literal + ')', createContext({}))
}

// Objects built in a vm context come from another realm, so deepStrictEqual
// fails on them. A JSON round-trip moves both sides into this realm.
const plain = (value) => JSON.parse(JSON.stringify(value))

describe('workflow scripts: meta is a literal first statement', () => {
test('the workflow scripts are found on disk', () => {
assert.ok(WORKFLOW_FILES.length >= 3, `found ${WORKFLOW_FILES.join(', ')}`)
})

for (const file of WORKFLOW_FILES) {
const src = readFileSync(join(workflowsDir, file), 'utf8')

test(`${file}: starts at byte 0 with the meta export`, () => {
assert.ok(
src.startsWith(META_START),
`${file}: must begin with "${META_START}" (no header comment, no other statement above it)`
)
})

test(`${file}: meta is a plain literal with no calls or references`, () => {
const idx = src.indexOf(META_START)
assert.notEqual(idx, -1, `${file}: no meta export found`)
const literal = braceMatch(src, idx)
// Evaluating in an empty context is not enough: builtins such as
// Object.freeze and Array.prototype.map are always there. So strip the
// quoted strings and the `key:` names, then require only punctuation
// to remain. That rejects calls, identifiers as values, arrows, spreads,
// template literals and comments.
const stripped = literal
.replace(/'[^'\\\n]*'/g, '')
.replace(/"[^"\\\n]*"/g, '')
.replace(/[A-Za-z_$][\w$]*\s*:/g, '')
assert.match(
stripped,
/^[\s{}[\],]*$/,
`${file}: meta must hold only quoted strings, object keys, braces, brackets and commas; left over: ${stripped.replace(/\s+/g, ' ').trim()}`
)
assert.doesNotThrow(() => evalLiteral(literal), `${file}: meta does not evaluate in an empty context`)
})

test(`${file}: meta matches SPEC with tiers resolved through TIER_MODELS`, () => {
const metaIdx = src.indexOf(META_START)
assert.notEqual(metaIdx, -1, `${file}: no meta export found`)
const meta = evalLiteral(braceMatch(src, metaIdx))

const specIdx = src.indexOf('const SPEC = {')
assert.notEqual(specIdx, -1, `${file}: no SPEC found`)
const spec = evalLiteral(braceMatch(src, specIdx))

const modelsIdx = src.indexOf('const TIER_MODELS = {')
assert.notEqual(modelsIdx, -1, `${file}: no TIER_MODELS found`)
const models = evalLiteral(braceMatch(src, modelsIdx))

const expected = {
name: spec.name,
description: spec.description,
phases: spec.phases.map((p) => ({ title: p.title, detail: p.detail, model: models[p.tier] })),
}
assert.deepStrictEqual(plain(meta), plain(expected), `${file}: meta drifted from SPEC / TIER_MODELS`)
})
}
})
26 changes: 16 additions & 10 deletions .claude/workflows/tm-map-codebase.js
Original file line number Diff line number Diff line change
@@ -1,3 +1,19 @@
export const meta = {
name: 'tm-map-codebase',
description:
'Token-bounded full-repo map: a Sonnet scout splits the repo into N coherent areas (N sized to the repo, capped at a ceiling), one Sonnet worker maps each area (purpose, entry points, key modules, data and control flow, external dependencies, conventions), and one Opus critic synthesizes a dated architecture map. Models are pinned per stage in-script, so it never inherits the session model, and the agent count scales with repo size only up to a hard ceiling. This is a map, not a review: no findings, no severities, no recommendations.',
phases: [
{ title: 'Scout', detail: 'one Sonnet agent splits the repo into N areas (N <= ceiling)', model: 'sonnet' },
{ title: 'Map', detail: 'one Sonnet worker per area describes it', model: 'sonnet' },
{ title: 'Synthesize', detail: 'one Opus critic writes the consolidated map', model: 'opus' },
],
}
// meta is a plain literal and the first statement because the workflow runtime
// reads it before running the script. It repeats the name, description and
// phases of SPEC below, with each phase's tier resolved through TIER_MODELS.
// workflow-meta.test.mjs fails if the two drift.
// The model is carried here for display purposes only.

// Workflow spec (embedded; mirrors specs/tm-map-codebase.spec.json)
// The spec encodes the fan-out as data: stages, tiers, parallelism, schemas,
// and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase()
Expand Down Expand Up @@ -51,16 +67,6 @@ const SPEC = {
const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' }
const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' }

export const meta = {
name: SPEC.name,
description: SPEC.description,
phases: SPEC.phases.map((p) => ({
title: p.title,
detail: p.detail,
model: TIER_MODELS[p.tier],
})),
}

// Bounded by construction. The scout sizes the number of areas N to the repo, and
// the script hard-clamps N to MAX_AREAS, so a run is 1 scout + N area workers +
// 1 critic = N + 2 agents, with N <= MAX_AREAS. The count scales with repo size
Expand Down
28 changes: 16 additions & 12 deletions .claude/workflows/tm-review-changes.js
Original file line number Diff line number Diff line change
@@ -1,3 +1,19 @@
export const meta = {
name: 'tm-review-changes',
description:
'Token-bounded code review: Sonnet workers review the diff across fixed dimensions, one Opus critic consolidates. Models are pinned per stage in-script, so it never inherits the session model or fans out unboundedly.',
phases: [
{ title: 'Review', detail: 'one Sonnet worker per dimension', model: 'sonnet' },
{ title: 'Verify', detail: 'one adversarial Sonnet worker per must-fix finding, capped', model: 'sonnet' },
{ title: 'Consolidate', detail: 'one Opus critic verifies and merges findings', model: 'opus' },
],
}
// meta is a plain literal and the first statement because the workflow runtime
// reads it before running the script. It repeats the name, description and
// phases of SPEC below, with each phase's tier resolved through TIER_MODELS.
// workflow-meta.test.mjs fails if the two drift.
// The model is carried here for display purposes only.

// Workflow spec (embedded; mirrors specs/tm-review-changes.spec.json)
// The spec encodes the fan-out as data: stages, tiers, parallelism, schemas,
// and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase()
Expand Down Expand Up @@ -53,18 +69,6 @@ const SPEC = {
const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' }
const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' }

export const meta = {
name: SPEC.name,
description: SPEC.description,
phases: SPEC.phases.map((p) => ({
title: p.title,
detail: p.detail,
// The adapter table resolves the tier to a concrete model for the
// Claude Code host. meta.phases carries the model for display purposes.
model: TIER_MODELS[p.tier],
})),
}

// Bounded by construction. The dimension list is fixed, there is no per-file
// fan-out and no loop, so a run is DIMENSIONS.length Sonnet reviewers, plus one
// Sonnet verifier per must-fix finding (at most min(must-fix count, MAX_VERIFY),
Expand Down
26 changes: 16 additions & 10 deletions .claude/workflows/tm-review-codebase.js
Original file line number Diff line number Diff line change
@@ -1,3 +1,19 @@
export const meta = {
name: 'tm-review-codebase',
description:
'Token-bounded full-repo review: a Sonnet scout splits the repo into N coherent areas (N sized to the repo, capped at a ceiling), one Sonnet worker reviews each area plus one Sonnet architecture worker audits repo-wide structure, and one Opus critic verifies, writes a dated report, and consolidates. Models are pinned per stage in-script, so it never inherits the session model, and the agent count scales with repo size only up to a hard ceiling.',
phases: [
{ title: 'Scout', detail: 'one Sonnet agent splits the repo into N areas (N <= ceiling)', model: 'sonnet' },
{ title: 'Review', detail: 'one Sonnet worker per area plus one architecture worker', model: 'sonnet' },
{ title: 'Consolidate', detail: 'one Opus critic verifies, writes the report, consolidates', model: 'opus' },
],
}
// meta is a plain literal and the first statement because the workflow runtime
// reads it before running the script. It repeats the name, description and
// phases of SPEC below, with each phase's tier resolved through TIER_MODELS.
// workflow-meta.test.mjs fails if the two drift.
// The model is carried here for display purposes only.

// Workflow spec (embedded; mirrors specs/tm-review-codebase.spec.json)
// The spec encodes the fan-out as data: stages, tiers, parallelism, schemas,
// and fallbacks. The JS renderer reads SPEC to drive agent()/parallel()/phase()
Expand Down Expand Up @@ -59,16 +75,6 @@ const SPEC = {
const TIER_MODELS = { judgment: 'opus', worker: 'sonnet', lead: 'opus' }
const TIER_EFFORTS = { judgment: 'xhigh', worker: 'high', lead: 'xhigh' }

export const meta = {
name: SPEC.name,
description: SPEC.description,
phases: SPEC.phases.map((p) => ({
title: p.title,
detail: p.detail,
model: TIER_MODELS[p.tier],
})),
}

// Bounded by construction. The scout sizes the number of areas N to the repo, and
// the script hard-clamps N to MAX_AREAS, so a run is 1 scout + (N area workers +
// 1 architecture worker) + 1 critic = N + 3 agents, with N <= MAX_AREAS. The
Expand Down
Loading