From c5d0bff5d2bda071c8227caa2d2e9b3ac4e5d23a Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Mon, 3 Aug 2026 12:24:35 -0600 Subject: [PATCH 1/2] feat(campaign): family-GEPA round 3 on OpenHands+Terminus2 dev pools --scenario-family oht2 retargets the analyst GEPA campaign at the TUNING-LEGAL OpenHands + Terminus2 dev pools: labeled-positive rows only, per-family seeded sample (12 train + 8 selection per family, disjoint), recorded in a committed split-plan that every run re-derives and verifies before spending. The optimized metric is the plain per-case judge composite; selection readouts report plain macro and pooled micro plus a per-family breakdown. --recipe engine runs one bounded GEPA engine; --recipe omni runs the official Omni composition on the same evaluation budget via the gepa-source venv. Family evaluations run at concurrency 3 (the provider seat 429-limits at ~5 concurrent); the mini-SWE round-2 path is unchanged and stays serial. --- .../oht2-family-gepa-20260803/split-plan.json | 312 ++++++ scripts/gepa-analyst-campaign.ts | 923 +++++++++++++----- 2 files changed, 1010 insertions(+), 225 deletions(-) create mode 100644 benchmarks/trace-analysis/oht2-family-gepa-20260803/split-plan.json diff --git a/benchmarks/trace-analysis/oht2-family-gepa-20260803/split-plan.json b/benchmarks/trace-analysis/oht2-family-gepa-20260803/split-plan.json new file mode 100644 index 00000000..7d14a685 --- /dev/null +++ b/benchmarks/trace-analysis/oht2-family-gepa-20260803/split-plan.json @@ -0,0 +1,312 @@ +{ + "kind": "agent-eval/gepa-analyst-family-split-plan", + "round": 3, + "scenarioFamily": "oht2", + "seed": 31, + "metric": "plain per-case judge composite (f1 averaged with critical-step accuracy when defined); no gold-mass weighting", + "spent": "Every traj_id under train and selection is GEPA-SPENT: certification splits must exclude these rows.", + "trainPerFamily": 12, + "selectionPerFamily": 8, + "pools": [ + { + "name": "openhands", + "labelsPath": "/dev/shm/ctb-openhands-dev-labels.json", + "labelsSha256": "2bd62a9d5ec785482611e1b02a1a829cf98c0777b6ae1ac9b1867b8fc9e4b8c6", + "durableLabelsPath": "/home/drew/bench-cache/ctb-20260801/oht2/ctb-openhands-dev-labels.json", + "rowCount": 464, + "positiveCount": 146 + }, + { + "name": "terminus2", + "labelsPath": "/dev/shm/ctb-terminus2-dev-labels.json", + "labelsSha256": "8388ffde9c1b3b3f8f04afb4356a3b10abefda6db9ecd0216f65003bf6251958", + "durableLabelsPath": "/home/drew/bench-cache/ctb-20260801/oht2/ctb-terminus2-dev-labels.json", + "rowCount": 188, + "positiveCount": 50 + } + ], + "train": [ + { + "trajId": "openhands-Anthropic__Claude-Sonnet-4-20250514-Thinking-fibonacci-server-89ec6d06", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "openhands-Anthropic__Claude-Sonnet-4-20250514-Thinking-intrusion-detection-a847d6de", + "family": "openhands", + "goldMass": 9, + "maxBlockWidth": 5, + "stratum": "wide" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-classifier-debug-0b0fdb32", + "family": "openhands", + "goldMass": 2, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-microsoft__vscode-153121-9abfbc86", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-build-tcc-qemu-0af21736", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-financial-document-processor-39cf7cca", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "openhands-Anthropic__Claude-Sonnet-4-20250514-Thinking-distribution-search-b2092e02", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-chem-rf-43d5d989", + "family": "openhands", + "goldMass": 5, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-microsoft__vscode-109750-0071ea66", + "family": "openhands", + "goldMass": 4, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-amuse-install-4d9725f6", + "family": "openhands", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-instance_ansible__ansible-6cc97447aac5816745278f3735af128afb255c81-v0f01c69f1e2528b935359cfe578530722bca2c59-83adb675", + "family": "openhands", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-Anthropic__Claude-Sonnet-4-20250514-Thinking-build-pmars-78f839aa", + "family": "openhands", + "goldMass": 2, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-polyglot-rust-c-9477650d", + "family": "terminus2", + "goldMass": 8, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-bank-trans-filter-991f9d5d", + "family": "terminus2", + "goldMass": 2, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-magsac-install-4c7fd435", + "family": "terminus2", + "goldMass": 8, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-lean4-proof-0dd8d068", + "family": "terminus2", + "goldMass": 5, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-predicate-pushdown-bench-0225294a", + "family": "terminus2", + "goldMass": 4, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-adaptive-rejection-sampler-63290128", + "family": "terminus2", + "goldMass": 5, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-install-windows-3.11-9b3a3ca5", + "family": "terminus2", + "goldMass": 9, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-tree-directory-parser-6c88f322", + "family": "terminus2", + "goldMass": 6, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-mteb-leaderboard-d9697907", + "family": "terminus2", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-circuit-fibsqrt-2c6a2717", + "family": "terminus2", + "goldMass": 6, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-broken-python-02d55149", + "family": "terminus2", + "goldMass": 3, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-build-pov-ray-4a3cb2c6", + "family": "terminus2", + "goldMass": 3, + "maxBlockWidth": 2, + "stratum": "narrow" + } + ], + "selection": [ + { + "trajId": "openhands-OpenAI__GPT-5-prettier__prettier-3515-0d82b41d", + "family": "openhands", + "goldMass": 3, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "openhands-OpenAI__GPT-5-instance_ansible__ansible-fb144c44144f8bd3542e71f5db62b6d322c7bd85-vba6da65a0f3baefda7a058ebbd0a8dcafb8512f5-57b35ea6", + "family": "openhands", + "goldMass": 5, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-fix-code-vulnerability-10de86e2", + "family": "openhands", + "goldMass": 3, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-DeepSeek__DeepSeek-V3.2-multi-source-data-merger-250e898f", + "family": "openhands", + "goldMass": 5, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-mui__material-ui-13778-83d0b364", + "family": "openhands", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-django__django-14351-058e4e93", + "family": "openhands", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-instance_ansible__ansible-942424e10b2095a173dbd78e7128f52f7995849b-v30a923fb5c164d6cd18280c02422f75e611e8fb2-83422c29", + "family": "openhands", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "openhands-OpenAI__GPT-5-microsoft__vscode-160342-be2f227f", + "family": "openhands", + "goldMass": 12, + "maxBlockWidth": 4, + "stratum": "wide" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-npm-conflict-resolution-1fb2a796", + "family": "terminus2", + "goldMass": 1, + "maxBlockWidth": 1, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-bn-fit-modify-e23f783a", + "family": "terminus2", + "goldMass": 3, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-regex-chess-f6300f5d", + "family": "terminus2", + "goldMass": 15, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-path-tracing-reverse-56141d64", + "family": "terminus2", + "goldMass": 2, + "maxBlockWidth": 2, + "stratum": "narrow" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-build-pov-ray-0022ddb2", + "family": "terminus2", + "goldMass": 9, + "maxBlockWidth": 6, + "stratum": "wide" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-model-extraction-relu-logits-e2e85d60", + "family": "terminus2", + "goldMass": 6, + "maxBlockWidth": 6, + "stratum": "wide" + }, + { + "trajId": "terminus2-DeepSeek__DeepSeek-V3.2-catch-me-if-you-can-7f73ac9f", + "family": "terminus2", + "goldMass": 7, + "maxBlockWidth": 3, + "stratum": "wide" + }, + { + "trajId": "terminus2-Anthropic__Claude-Sonnet-4-20250514-Thinking-solve-maze-challenge-27b84da7", + "family": "terminus2", + "goldMass": 6, + "maxBlockWidth": 2, + "stratum": "narrow" + } + ] +} diff --git a/scripts/gepa-analyst-campaign.ts b/scripts/gepa-analyst-campaign.ts index 13cd1142..0031e356 100644 --- a/scripts/gepa-analyst-campaign.ts +++ b/scripts/gepa-analyst-campaign.ts @@ -1,5 +1,5 @@ /** - * GEPA campaign over the recursive CodeTraceBench analyst instructions (round 2). + * GEPA campaign over the recursive CodeTraceBench analyst instructions. * * Surface: the full RLM instruction string. The output-contract block (the * subject grammar and block caps) is frozen: a candidate that does not contain @@ -7,46 +7,64 @@ * because a rewritten contract breaks the subject encoding and produces * parse-zero scores that are indistinguishable from bad analysis. * - * Data: the SPENT dev-32 and holdout-1 pools (32 labeled-positive scenarios). - * Both pools were burned by earlier measurement runs, so they are legal for - * training but can certify nothing; certification runs elsewhere on sealed - * splits via `analyst-benchmark --instructions-file`. The train/selection - * split (~20/~12, disjoint) is seeded and stratified by gold-block width — - * the widest contiguous run of labeled incorrect steps — so the selection - * split represents both wide-block and narrow-block regimes. + * Two scenario sources, selected by `--scenario-family`: * - * Feedback is micro-aligned two ways: - * 1. The score GEPA optimizes is the judge composite (per-case f1 averaged - * with critical-step accuracy when defined) multiplied by the case's - * gold-mass weight (labeled-issue count / pool max), so the mean GEPA - * climbs is proportional to a gold-mass-weighted average — the macro - * proxy closest to the micro benchmark headline. - * 2. Every evaluation returns per-case TP/FP/FN counts and missed issue ids - * through the artifact evidence, so reflection sees the count structure - * behind each scalar. The selection readout reports the weighted metric, - * the plain macro composite, plain macro f1, and pooled micro f1 - * side by side so they are never confused. + * Default (mini-SWE, round 2): the SPENT dev-32 and holdout-1 pools + * (32 labeled-positive scenarios), stratified by gold-block width, scored by + * the judge composite multiplied by the case's gold-mass weight. + * + * `--scenario-family oht2` (round 3): the TUNING-LEGAL OpenHands + Terminus2 + * dev pools. Labeled-positive rows only, both families mixed, per-family + * seeded sample: 12 train + 8 selection per family (24 train / 16 selection, + * disjoint). The selected traj_ids are GEPA-SPENT and recorded in a committed + * split-plan (`benchmarks/trace-analysis/oht2-family-gepa-20260803/`); + * certification cuts must exclude them, and every run re-derives the split and + * refuses to start if it drifts from the committed plan. The score is the + * plain per-case judge composite (f1 averaged with critical-step accuracy when + * defined) with no gold-mass weighting; selection readouts report plain macro + * and pooled micro side by side. `--recipe engine` (default) runs one bounded + * GEPA engine; `--recipe omni` runs GEPA's official Omni composition + * (two explore stages, then a continuation seeded by the explore winner) on + * the same total evaluation budget — Omni requires the GEPA source venv + * (`.venv-gepa-source`). + * + * Every training or selection evaluation returns per-case TP/FP/FN counts and + * missed issue ids through the artifact evidence, so reflection sees the count + * structure behind each scalar. * * Usage: - * ZAI_GLM_API_KEY=... pnpm exec tsx scripts/gepa-analyst-campaign.ts --plan # split only, no spend - * ZAI_GLM_API_KEY=... pnpm exec tsx scripts/gepa-analyst-campaign.ts --smoke - * # full run, detached, with archival on completion: - * setsid nohup bash -c 'ZAI_GLM_API_KEY=... pnpm exec tsx scripts/gepa-analyst-campaign.ts; \ - * rsync -a /dev/shm/mp-tg2-gepa-r2/ ~/bench-cache/ctb-20260801/gepa-run2/' \ - * > /dev/shm/mp-tg2-gepa-r2.log 2>&1 & + * # family split plan only, no spend, no venvs needed: + * pnpm exec tsx scripts/gepa-analyst-campaign.ts --scenario-family oht2 --plan + * # family smoke (3 scenarios, 3 evaluations, real provider — take the + * # measurement mutex first): + * until mkdir /tmp/ctb-llm-mutex.lock 2>/dev/null; do sleep 20; done; \ + * trap 'rmdir /tmp/ctb-llm-mutex.lock' EXIT; \ + * ZAI_GLM_API_KEY=... pnpm exec tsx scripts/gepa-analyst-campaign.ts \ + * --scenario-family oht2 --smoke + * # full family run, detached, with archival on completion: + * setsid nohup bash -c 'until mkdir /tmp/ctb-llm-mutex.lock 2>/dev/null; do sleep 20; done; \ + * trap "rmdir /tmp/ctb-llm-mutex.lock" EXIT; \ + * ZAI_GLM_API_KEY=... pnpm exec tsx scripts/gepa-analyst-campaign.ts --scenario-family oht2; \ + * rsync -a /dev/shm/mp-tg3-gepa-family-engine/ ~/bench-cache/ctb-20260801/gepa-family-engine/' \ + * > /dev/shm/mp-tg3-gepa-family-engine.log 2>&1 & */ -import { appendFileSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { appendFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { homedir } from 'node:os' import { dirname, join, resolve } from 'node:path' import { fileURLToPath } from 'node:url' import type { AnalystBenchmarkCase } from '../src/analyst/benchmark' import type { AnalystIssueExpectation } from '../src/analyst/benchmark' import { scoreAnalystFindings } from '../src/analyst/benchmark' +import { codeTraceBenchCase, type CodeTraceBenchRow } from '../src/analyst/benchmark-datasets' import { adaptPublicBenchmarkFindings } from '../src/analyst/benchmark-public-adapters' import { CODE_TRACE_BENCH_ANALYST_PROMPT, publicBenchmarkRlmInstructions, } from '../src/analyst/benchmark-public-prompt' -import { preparePublicAnalystBenchmark } from '../src/analyst/benchmark-public-data' +import { + loadPublicBenchmarkRows, + preparePublicAnalystBenchmark, +} from '../src/analyst/benchmark-public-data' import { sha256Digest } from '../src/analyst/benchmark-verification-artifacts' import { createDspyRlmTraceEngine } from '../src/analyst/dspy-rlm-engine' import { evidenceRefsFromRawFinding } from '../src/analyst/finding-signature' @@ -58,6 +76,10 @@ import { type TraceAnalystArtifact, type TraceAnalystScenario, } from '../src/campaign/analyst-surface' +import type { + GepaEngineRun, + GepaOptimizationRecipe, +} from '../src/campaign/gepa-optimization-method' import { gepaOptimizationMethod } from '../src/campaign/gepa-optimization-method' import type { OptimizationMethodInput, @@ -71,11 +93,33 @@ import { loadOptimizerExecutionOwner } from '../examples/_shared/optimizer-execu // ── Configuration ───────────────────────────────────────────────────── const SMOKE = process.argv.includes('--smoke') +const PLAN_ONLY = process.argv.includes('--plan') const RUN_DIR_ARG = argValue('--run-dir') const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..') -const RUN_DIR = RUN_DIR_ARG ?? (SMOKE ? '/dev/shm/mp-tg2-gepa-r2-smoke' : '/dev/shm/mp-tg2-gepa-r2') -const ROUND = 2 -const SEED = 23 + +const FAMILY_ARG = argValue('--scenario-family') +if (FAMILY_ARG !== undefined && FAMILY_ARG !== 'oht2') { + throw new Error(`--scenario-family supports only 'oht2', received '${FAMILY_ARG}'`) +} +const FAMILY = FAMILY_ARG === 'oht2' +const RECIPE_ARG = argValue('--recipe') +if (RECIPE_ARG !== undefined && !FAMILY) { + throw new Error('--recipe requires --scenario-family oht2') +} +if (RECIPE_ARG !== undefined && RECIPE_ARG !== 'engine' && RECIPE_ARG !== 'omni') { + throw new Error(`--recipe supports 'engine' or 'omni', received '${RECIPE_ARG}'`) +} +const RECIPE_KIND: 'engine' | 'omni' = RECIPE_ARG === 'omni' ? 'omni' : 'engine' + +const RUN_DIR = + RUN_DIR_ARG ?? + (FAMILY + ? `/dev/shm/mp-tg3-gepa-family-${RECIPE_KIND}${SMOKE ? '-smoke' : ''}` + : SMOKE + ? '/dev/shm/mp-tg2-gepa-r2-smoke' + : '/dev/shm/mp-tg2-gepa-r2') +const ROUND = FAMILY ? 3 : 2 +const SEED = FAMILY ? 31 : 23 const MODEL = 'glm-5.2' const BASE_URL = 'https://api.z.ai/api/coding/paas/v4' const PRICING = { inputUsdPerMillion: 0.6, outputUsdPerMillion: 2.2 } @@ -84,16 +128,34 @@ const MAX_EVALUATIONS = EVALUATIONS_OVERRIDE ? Number.parseInt(EVALUATIONS_OVERRIDE, 10) : SMOKE ? 3 - : 60 + : FAMILY + ? 50 + : 60 const COST_CEILING_USD = SMOKE ? 3 : 20 const MAX_PROPOSER_COST_USD = SMOKE ? 1 : 5 const GEPA_TIMEOUT_MS = SMOKE ? 3_600_000 : 28_800_000 const ANALYSIS_TIMEOUT_MS = 1_200_000 const EVAL_COST_PHASE = 'gepa.analyst-evaluation' +/** + * Concurrent analyses cap. The z.ai glm seat starts returning 429s at ~5 + * concurrent requests (measured), so the family campaign runs at most 3 + * evaluations in flight; the mini-SWE round stays strictly serial. + */ +const EVAL_CONCURRENCY = FAMILY ? 3 : 1 const DSPY_PYTHON = join(ROOT, 'clients/python/.venv/bin/python') -const GEPA_PYTHON = join(ROOT, 'clients/python/.venv-gepa/bin/python') +/** Omni needs GEPA's official composition functions, present only in the source venv. */ +const GEPA_PYTHON = join( + ROOT, + FAMILY && RECIPE_KIND === 'omni' + ? 'clients/python/.venv-gepa-source/bin/python' + : 'clients/python/.venv-gepa/bin/python', +) +const GEPA_VENV_HINT = + FAMILY && RECIPE_KIND === 'omni' + ? 'cd clients/python && UV_PROJECT_ENVIRONMENT=.venv-gepa-source uv sync --frozen --group gepa-source' + : 'cd clients/python && UV_PROJECT_ENVIRONMENT=.venv-gepa uv sync --frozen --group gepa-release' -/** Spent pools only. Both are training-legal and certification-illegal. */ +/** Spent mini-SWE pools. Both are training-legal and certification-illegal. */ const POOLS = [ { name: 'dev32', @@ -112,12 +174,41 @@ const POOLS = [ }, ] as const +/** + * TUNING-LEGAL family dev pools. The /dev/shm labels are the primaries; the + * bench-cache copies are byte-identical durable backups for restoring them. + * Rows selected into the split become GEPA-SPENT. + */ +const FAMILY_POOLS = [ + { + name: 'openhands', + labelsPath: '/dev/shm/ctb-openhands-dev-labels.json', + durableLabelsPath: join(homedir(), 'bench-cache/ctb-20260801/oht2/ctb-openhands-dev-labels.json'), + traceDir: '/dev/shm/ctb-oht2-traces-openhands', + artifactDir: join(homedir(), 'bench-cache/ctb-20260801/oht2/work/openhands/extracted'), + }, + { + name: 'terminus2', + labelsPath: '/dev/shm/ctb-terminus2-dev-labels.json', + durableLabelsPath: join(homedir(), 'bench-cache/ctb-20260801/oht2/ctb-terminus2-dev-labels.json'), + traceDir: '/dev/shm/ctb-oht2-traces-terminus2', + artifactDir: join(homedir(), 'bench-cache/ctb-20260801/oht2/work/terminus2/extracted'), + }, +] as const +const FAMILY_TRAIN_PER_FAMILY = 12 +const FAMILY_SELECTION_PER_FAMILY = 8 +/** Committed record of the GEPA-SPENT rows; runs refuse to start on drift. */ +const FAMILY_SPLIT_PLAN_PATH = join( + ROOT, + 'benchmarks/trace-analysis/oht2-family-gepa-20260803/split-plan.json', +) + /** A gold block of >= this many contiguous incorrect steps counts as wide. */ const WIDE_MIN_BLOCK_WIDTH = 3 const SELECTION_TOTAL = 12 const API_KEY = process.env.ZAI_GLM_API_KEY?.trim() -if (!API_KEY) throw new Error('ZAI_GLM_API_KEY is required') +if (!API_KEY && !PLAN_ONLY) throw new Error('ZAI_GLM_API_KEY is required') // ── Frozen output contract ──────────────────────────────────────────── @@ -130,8 +221,8 @@ if (!CONTRACT_BLOCK.trim() || `${CODE_TRACE_BENCH_ANALYST_PROMPT}\n${CONTRACT_BL throw new Error('frozen contract block extraction failed') } -// ── Scenarios (JSON-safe: compareOptimizationMethods structuredClones them, -// so live trace stores are resolved by scenario id inside the dispatch) ── +// ── Scenarios (JSON-safe: the optimizer bridge serializes them, so live +// trace stores are resolved by scenario id inside the dispatch) ── interface GepaAnalystScenario extends Scenario { kind: 'trace-analyst' @@ -164,6 +255,14 @@ interface GepaAnalystArtifact extends TraceAnalystArtifact { blockDiagnostics?: unknown } +interface PoolProvenance { + name: string + labelsPath: string + labelsSha256: string + filteredLabelsPath?: string + filteredLabelsSha256?: string +} + const storesById = new Map() function seededShuffle(items: readonly T[], seed: number): T[] { @@ -211,6 +310,137 @@ function goldBlockWidths(expectedIssues: readonly AnalystIssueExpectation[]): nu return widths } +// ── Family split: labeled-positive rows only, per-family seeded sample ── + +interface FamilyPlanRow { + trajId: string + family: string + goldMass: number + maxBlockWidth: number + stratum: 'wide' | 'narrow' +} + +interface FamilyPoolSplit { + name: string + labelsPath: string + labelsSha256: string + durableLabelsPath: string + rowCount: number + positiveCount: number + train: FamilyPlanRow[] + selection: FamilyPlanRow[] + rowsByTrajId: Map> +} + +async function computeFamilySplit(): Promise { + const pools: FamilyPoolSplit[] = [] + for (const [poolIndex, pool] of FAMILY_POOLS.entries()) { + const labelsSha256 = sha256Digest(readFileSync(pool.labelsPath, 'utf8')) + const rows = await loadPublicBenchmarkRows(pool.labelsPath) + const rowsByTrajId = new Map>() + const positives: FamilyPlanRow[] = [] + for (const row of rows) { + const benchmarkCase = codeTraceBenchCase(row as unknown as CodeTraceBenchRow, undefined) + const trajId = String(row.traj_id) + rowsByTrajId.set(trajId, row) + if (benchmarkCase.labelState !== 'positive') continue + const maxBlockWidth = Math.max(...goldBlockWidths(benchmarkCase.expectedIssues)) + positives.push({ + trajId, + family: pool.name, + goldMass: benchmarkCase.expectedIssues.length, + maxBlockWidth, + stratum: maxBlockWidth >= WIDE_MIN_BLOCK_WIDTH ? 'wide' : 'narrow', + }) + } + const needed = FAMILY_TRAIN_PER_FAMILY + FAMILY_SELECTION_PER_FAMILY + if (positives.length < needed) { + throw new Error( + `family pool '${pool.name}' has ${positives.length} labeled-positive rows; need ${needed}`, + ) + } + positives.sort((left, right) => left.trajId.localeCompare(right.trajId)) + const shuffled = seededShuffle(positives, SEED + poolIndex) + pools.push({ + name: pool.name, + labelsPath: pool.labelsPath, + labelsSha256, + durableLabelsPath: pool.durableLabelsPath, + rowCount: rows.length, + positiveCount: positives.length, + selection: shuffled.slice(0, FAMILY_SELECTION_PER_FAMILY), + train: shuffled.slice( + FAMILY_SELECTION_PER_FAMILY, + FAMILY_SELECTION_PER_FAMILY + FAMILY_TRAIN_PER_FAMILY, + ), + rowsByTrajId, + }) + } + return pools +} + +function familySplitPlanDocument(pools: readonly FamilyPoolSplit[]): Record { + return { + kind: 'agent-eval/gepa-analyst-family-split-plan', + round: ROUND, + scenarioFamily: 'oht2', + seed: SEED, + metric: + 'plain per-case judge composite (f1 averaged with critical-step accuracy when defined); no gold-mass weighting', + spent: + 'Every traj_id under train and selection is GEPA-SPENT: certification splits must exclude these rows.', + trainPerFamily: FAMILY_TRAIN_PER_FAMILY, + selectionPerFamily: FAMILY_SELECTION_PER_FAMILY, + pools: pools.map((pool) => ({ + name: pool.name, + labelsPath: pool.labelsPath, + labelsSha256: pool.labelsSha256, + durableLabelsPath: pool.durableLabelsPath, + rowCount: pool.rowCount, + positiveCount: pool.positiveCount, + })), + train: pools.flatMap((pool) => pool.train), + selection: pools.flatMap((pool) => pool.selection), + } +} + +/** The identity of a split: seed, source label digests, exact traj_id order. */ +function familySplitProjection(plan: Record): string { + const pools = (plan.pools as Array>).map((pool) => ({ + name: pool.name, + labelsSha256: pool.labelsSha256, + })) + const ids = (rows: unknown) => + (rows as Array>).map((row) => `${row.family}:${row.trajId}`) + return JSON.stringify({ + seed: plan.seed, + trainPerFamily: plan.trainPerFamily, + selectionPerFamily: plan.selectionPerFamily, + pools, + train: ids(plan.train), + selection: ids(plan.selection), + }) +} + +function verifyCommittedFamilySplitPlan(computed: Record): void { + if (!existsSync(FAMILY_SPLIT_PLAN_PATH)) { + throw new Error( + `committed family split plan is missing at ${FAMILY_SPLIT_PLAN_PATH} — ` + + 'run --scenario-family oht2 --plan and commit the result before spending', + ) + } + const committed = JSON.parse(readFileSync(FAMILY_SPLIT_PLAN_PATH, 'utf8')) as Record< + string, + unknown + > + if (familySplitProjection(committed) !== familySplitProjection(computed)) { + throw new Error( + `family split drifted from the committed plan at ${FAMILY_SPLIT_PLAN_PATH} — ` + + 'the labels, seed, or split constants changed; regenerate with --plan and re-commit deliberately', + ) + } +} + // ── Score ledger: every judge verdict lands here, keyed by candidate hash, // so the selection comparison is harvested without re-running anything. ── @@ -219,7 +449,7 @@ const SCORE_LEDGER_PATH = join(RUN_DIR, 'score-ledger.jsonl') interface LedgerScoreRow { scenarioId: string surfaceSha256: string - /** Gold-mass-weighted composite — the signal GEPA optimizes. */ + /** The signal GEPA optimizes: gold-mass-weighted in round 2, plain in family mode. */ composite: number /** Unweighted judge composite (f1, averaged with critical-step accuracy). */ rawComposite: number @@ -260,123 +490,286 @@ async function main(): Promise { costCeilingUsd: COST_CEILING_USD, }) - // ── Pool: positives from both spent splits, stratified by block width ── + // ── Pool: labeled positives, stratified; family mode splits per family ── const scenarios: GepaAnalystScenario[] = [] - const poolProvenance: Array<{ name: string; labelsPath: string; labelsSha256: string }> = [] - for (const pool of POOLS) { - const prepared = await preparePublicAnalystBenchmark({ - dataset: 'codetracebench', - labelsPath: pool.labelsPath, - traceDir: pool.traceDir, - artifactDir: pool.artifactDir, - limit: 32, - seed: 0, - }) - poolProvenance.push({ - name: pool.name, - labelsPath: pool.labelsPath, - labelsSha256: sha256Digest(readFileSync(pool.labelsPath, 'utf8')), - }) - for (const benchmarkCase of prepared.cases) { - if (benchmarkCase.labelState !== 'positive') continue - if (!benchmarkCase.input.traceStore) { - throw new Error(`prepared case '${benchmarkCase.id}' has no trace store`) + const poolProvenance: PoolProvenance[] = [] + let trainScenarios: GepaAnalystScenario[] + let selectionScenarios: GepaAnalystScenario[] + let stratification: Record + let maxGoldMass: number + + if (FAMILY) { + const split = await computeFamilySplit() + const planDocument = familySplitPlanDocument(split) + if (PLAN_ONLY) { + mkdirSync(dirname(FAMILY_SPLIT_PLAN_PATH), { recursive: true }) + writeFileSync(FAMILY_SPLIT_PLAN_PATH, `${JSON.stringify(planDocument, null, 2)}\n`) + writeFileSync(join(RUN_DIR, 'split-plan.json'), `${JSON.stringify(planDocument, null, 2)}\n`) + console.log(`[gepa-analyst-r${ROUND}] plan-only: wrote ${FAMILY_SPLIT_PLAN_PATH}`) + return + } + verifyCommittedFamilySplitPlan(planDocument) + writeFileSync(join(RUN_DIR, 'split-plan.json'), `${JSON.stringify(planDocument, null, 2)}\n`) + + // Smoke touches one train + one selection row from openhands and one train + // row from terminus2 — 3 scenarios, both families, both split roles. + const activeByPool = new Map( + split.map((pool) => [ + pool.name, + SMOKE + ? { + train: [pool.train[0]!], + selection: pool.name === 'openhands' ? [pool.selection[0]!] : [], + } + : { train: [...pool.train], selection: [...pool.selection] }, + ]), + ) + + const scenarioByTrajId = new Map() + for (const pool of split) { + const active = activeByPool.get(pool.name)! + const activeRows = [...active.train, ...active.selection] + if (activeRows.length === 0) continue + const filteredLabelsPath = join(RUN_DIR, 'pool-labels', `${pool.name}.json`) + mkdirSync(dirname(filteredLabelsPath), { recursive: true }) + writeFileSync( + filteredLabelsPath, + `${JSON.stringify( + activeRows.map((row) => { + const source = pool.rowsByTrajId.get(row.trajId) + if (!source) throw new Error(`family pool '${pool.name}' lost row '${row.trajId}'`) + return source + }), + null, + 2, + )}\n`, + ) + const poolConfig = FAMILY_POOLS.find((candidate) => candidate.name === pool.name)! + const prepared = await preparePublicAnalystBenchmark({ + dataset: 'codetracebench', + labelsPath: filteredLabelsPath, + traceDir: poolConfig.traceDir, + artifactDir: poolConfig.artifactDir, + limit: activeRows.length, + seed: 0, + }) + if (prepared.cases.length !== activeRows.length) { + throw new Error( + `family pool '${pool.name}' prepared ${prepared.cases.length} cases for ${activeRows.length} rows`, + ) } - if (storesById.has(benchmarkCase.id)) { - throw new Error(`scenario '${benchmarkCase.id}' appears in more than one pool`) + poolProvenance.push({ + name: pool.name, + labelsPath: pool.labelsPath, + labelsSha256: pool.labelsSha256, + filteredLabelsPath, + filteredLabelsSha256: prepared.labelsSha256, + }) + for (const benchmarkCase of prepared.cases) { + if (benchmarkCase.labelState !== 'positive') { + throw new Error(`family case '${benchmarkCase.id}' is not labeled positive`) + } + if (!benchmarkCase.input.traceStore) { + throw new Error(`prepared case '${benchmarkCase.id}' has no trace store`) + } + if (storesById.has(benchmarkCase.id)) { + throw new Error(`scenario '${benchmarkCase.id}' appears in more than one pool`) + } + storesById.set(benchmarkCase.id, benchmarkCase.input.traceStore) + const trajId = benchmarkCase.id.replace(/^codetrace:/, '') + const planRow = activeRows.find((row) => row.trajId === trajId) + if (!planRow) throw new Error(`prepared case '${benchmarkCase.id}' is not in the split plan`) + const scenario: GepaAnalystScenario = { + id: benchmarkCase.id, + kind: 'trace-analyst', + labelState: 'positive', + pool: pool.name, + trajectoryId: trajId, + expectedIssues: [...benchmarkCase.expectedIssues], + ...(benchmarkCase.labeledEvidence + ? { labeledEvidence: [...benchmarkCase.labeledEvidence] } + : {}), + goldMass: planRow.goldMass, + maxBlockWidth: planRow.maxBlockWidth, + stratum: planRow.stratum, + } + scenarios.push(scenario) + scenarioByTrajId.set(trajId, scenario) } - storesById.set(benchmarkCase.id, benchmarkCase.input.traceStore) - const widths = goldBlockWidths(benchmarkCase.expectedIssues) - const maxBlockWidth = Math.max(...widths) - scenarios.push({ - id: benchmarkCase.id, - kind: 'trace-analyst', - labelState: 'positive', - pool: pool.name, - trajectoryId: benchmarkCase.id.replace(/^codetrace:/, ''), - expectedIssues: [...benchmarkCase.expectedIssues], - ...(benchmarkCase.labeledEvidence - ? { labeledEvidence: [...benchmarkCase.labeledEvidence] } - : {}), - goldMass: benchmarkCase.expectedIssues.length, - maxBlockWidth, - stratum: maxBlockWidth >= WIDE_MIN_BLOCK_WIDTH ? 'wide' : 'narrow', + } + const resolveScenarios = (rows: readonly FamilyPlanRow[]) => + rows.map((row) => { + const scenario = scenarioByTrajId.get(row.trajId) + if (!scenario) throw new Error(`no prepared scenario for planned row '${row.trajId}'`) + return scenario }) + const plannedTrain = split.flatMap((pool) => activeByPool.get(pool.name)!.train) + const plannedSelection = split.flatMap((pool) => activeByPool.get(pool.name)!.selection) + trainScenarios = SMOKE + ? resolveScenarios(plannedTrain) + : seededShuffle(resolveScenarios(plannedTrain), SEED + 2) + selectionScenarios = SMOKE + ? resolveScenarios(plannedSelection) + : seededShuffle(resolveScenarios(plannedSelection), SEED + 3) + maxGoldMass = Math.max(...scenarios.map((scenario) => scenario.goldMass)) + const describeFamilySplit = (splitScenarios: readonly GepaAnalystScenario[]) => ({ + n: splitScenarios.length, + wide: splitScenarios.filter((scenario) => scenario.stratum === 'wide').length, + narrow: splitScenarios.filter((scenario) => scenario.stratum === 'narrow').length, + byFamily: Object.fromEntries( + split.map((pool) => [ + pool.name, + splitScenarios.filter((scenario) => scenario.pool === pool.name).length, + ]), + ), + goldMass: splitScenarios.reduce((total, scenario) => total + scenario.goldMass, 0), + }) + stratification = { + scheme: 'per-family seeded sample over labeled-positive rows', + trainPerFamily: FAMILY_TRAIN_PER_FAMILY, + selectionPerFamily: FAMILY_SELECTION_PER_FAMILY, + pools: split.map((pool) => ({ + name: pool.name, + rowCount: pool.rowCount, + positiveCount: pool.positiveCount, + })), + train: describeFamilySplit(trainScenarios), + selection: describeFamilySplit(selectionScenarios), } - } - if (scenarios.length !== 32) { - throw new Error(`expected 32 labeled-positive scenarios across pools, found ${scenarios.length}`) - } - const maxGoldMass = Math.max(...scenarios.map((scenario) => scenario.goldMass)) - const goldMassWeight = (scenario: Pick) => - scenario.goldMass / maxGoldMass + } else { + for (const pool of POOLS) { + const prepared = await preparePublicAnalystBenchmark({ + dataset: 'codetracebench', + labelsPath: pool.labelsPath, + traceDir: pool.traceDir, + artifactDir: pool.artifactDir, + limit: 32, + seed: 0, + }) + poolProvenance.push({ + name: pool.name, + labelsPath: pool.labelsPath, + labelsSha256: sha256Digest(readFileSync(pool.labelsPath, 'utf8')), + }) + for (const benchmarkCase of prepared.cases) { + if (benchmarkCase.labelState !== 'positive') continue + if (!benchmarkCase.input.traceStore) { + throw new Error(`prepared case '${benchmarkCase.id}' has no trace store`) + } + if (storesById.has(benchmarkCase.id)) { + throw new Error(`scenario '${benchmarkCase.id}' appears in more than one pool`) + } + storesById.set(benchmarkCase.id, benchmarkCase.input.traceStore) + const widths = goldBlockWidths(benchmarkCase.expectedIssues) + const maxBlockWidth = Math.max(...widths) + scenarios.push({ + id: benchmarkCase.id, + kind: 'trace-analyst', + labelState: 'positive', + pool: pool.name, + trajectoryId: benchmarkCase.id.replace(/^codetrace:/, ''), + expectedIssues: [...benchmarkCase.expectedIssues], + ...(benchmarkCase.labeledEvidence + ? { labeledEvidence: [...benchmarkCase.labeledEvidence] } + : {}), + goldMass: benchmarkCase.expectedIssues.length, + maxBlockWidth, + stratum: maxBlockWidth >= WIDE_MIN_BLOCK_WIDTH ? 'wide' : 'narrow', + }) + } + } + if (scenarios.length !== 32) { + throw new Error( + `expected 32 labeled-positive scenarios across pools, found ${scenarios.length}`, + ) + } + maxGoldMass = Math.max(...scenarios.map((scenario) => scenario.goldMass)) - const wide = seededShuffle( - scenarios.filter((scenario) => scenario.stratum === 'wide'), - SEED, - ) - const narrow = seededShuffle( - scenarios.filter((scenario) => scenario.stratum === 'narrow'), - SEED + 1, - ) - const selectionWide = Math.min( - wide.length, - Math.max(1, Math.round((SELECTION_TOTAL * wide.length) / scenarios.length)), - ) - const selectionNarrow = SELECTION_TOTAL - selectionWide - const fullSelection = [...wide.slice(0, selectionWide), ...narrow.slice(0, selectionNarrow)] - const fullTrain = [...wide.slice(selectionWide), ...narrow.slice(selectionNarrow)] - const trainScenarios = SMOKE - ? [wide[selectionWide]!, narrow[selectionNarrow]!] - : seededShuffle(fullTrain, SEED + 2) - const selectionScenarios = SMOKE ? [fullSelection[0]!] : seededShuffle(fullSelection, SEED + 3) - const describeSplit = (split: readonly GepaAnalystScenario[]) => ({ - n: split.length, - wide: split.filter((scenario) => scenario.stratum === 'wide').length, - narrow: split.filter((scenario) => scenario.stratum === 'narrow').length, - dev32: split.filter((scenario) => scenario.pool === 'dev32').length, - holdout1: split.filter((scenario) => scenario.pool === 'holdout1').length, - goldMass: split.reduce((total, scenario) => total + scenario.goldMass, 0), - }) - const stratification = { - wideMinBlockWidth: WIDE_MIN_BLOCK_WIDTH, - pool: describeSplit(scenarios), - train: describeSplit(trainScenarios), - selection: describeSplit(selectionScenarios), - maxGoldMass, - } - console.log(`[gepa-analyst-r2] stratification ${JSON.stringify(stratification)}`) + const wide = seededShuffle( + scenarios.filter((scenario) => scenario.stratum === 'wide'), + SEED, + ) + const narrow = seededShuffle( + scenarios.filter((scenario) => scenario.stratum === 'narrow'), + SEED + 1, + ) + const selectionWide = Math.min( + wide.length, + Math.max(1, Math.round((SELECTION_TOTAL * wide.length) / scenarios.length)), + ) + const selectionNarrow = SELECTION_TOTAL - selectionWide + const fullSelection = [...wide.slice(0, selectionWide), ...narrow.slice(0, selectionNarrow)] + const fullTrain = [...wide.slice(selectionWide), ...narrow.slice(selectionNarrow)] + trainScenarios = SMOKE + ? [wide[selectionWide]!, narrow[selectionNarrow]!] + : seededShuffle(fullTrain, SEED + 2) + selectionScenarios = SMOKE ? [fullSelection[0]!] : seededShuffle(fullSelection, SEED + 3) + const describeSplit = (split: readonly GepaAnalystScenario[]) => ({ + n: split.length, + wide: split.filter((scenario) => scenario.stratum === 'wide').length, + narrow: split.filter((scenario) => scenario.stratum === 'narrow').length, + dev32: split.filter((scenario) => scenario.pool === 'dev32').length, + holdout1: split.filter((scenario) => scenario.pool === 'holdout1').length, + goldMass: split.reduce((total, scenario) => total + scenario.goldMass, 0), + }) + stratification = { + wideMinBlockWidth: WIDE_MIN_BLOCK_WIDTH, + pool: describeSplit(scenarios), + train: describeSplit(trainScenarios), + selection: describeSplit(selectionScenarios), + maxGoldMass, + } - if (process.argv.includes('--plan')) { - const plan = { - kind: 'agent-eval/gepa-analyst-campaign-plan', - round: ROUND, - seed: SEED, - pools: poolProvenance, - stratification, - trainScenarios: trainScenarios.map((scenario) => ({ - id: scenario.id, - pool: scenario.pool, - stratum: scenario.stratum, - goldMass: scenario.goldMass, - maxBlockWidth: scenario.maxBlockWidth, - })), - selectionScenarios: selectionScenarios.map((scenario) => ({ - id: scenario.id, - pool: scenario.pool, - stratum: scenario.stratum, - goldMass: scenario.goldMass, - maxBlockWidth: scenario.maxBlockWidth, - })), - weighting: { - scheme: 'goldMass / maxGoldMass over the full 32-scenario pool', - maxGoldMass, - }, + if (PLAN_ONLY) { + const plan = { + kind: 'agent-eval/gepa-analyst-campaign-plan', + round: ROUND, + seed: SEED, + pools: poolProvenance, + stratification, + trainScenarios: trainScenarios.map((scenario) => ({ + id: scenario.id, + pool: scenario.pool, + stratum: scenario.stratum, + goldMass: scenario.goldMass, + maxBlockWidth: scenario.maxBlockWidth, + })), + selectionScenarios: selectionScenarios.map((scenario) => ({ + id: scenario.id, + pool: scenario.pool, + stratum: scenario.stratum, + goldMass: scenario.goldMass, + maxBlockWidth: scenario.maxBlockWidth, + })), + weighting: { + scheme: 'goldMass / maxGoldMass over the full 32-scenario pool', + maxGoldMass, + }, + } + writeFileSync(join(RUN_DIR, 'split-plan.json'), `${JSON.stringify(plan, null, 2)}\n`) + console.log(`[gepa-analyst-r${ROUND}] plan-only: wrote ${join(RUN_DIR, 'split-plan.json')}`) + return } - writeFileSync(join(RUN_DIR, 'split-plan.json'), `${JSON.stringify(plan, null, 2)}\n`) - console.log(`[gepa-analyst-r2] plan-only: wrote ${join(RUN_DIR, 'split-plan.json')}`) - return + } + console.log(`[gepa-analyst-r${ROUND}] stratification ${JSON.stringify(stratification)}`) + + /** + * Round 2 weights the composite by gold mass so the optimized mean tracks + * micro-F1. The family campaign scores the plain per-case composite: every + * case counts equally, and the weight is identically 1. + */ + const goldMassWeight = FAMILY + ? () => 1 + : (scenario: Pick) => scenario.goldMass / maxGoldMass + + if (!existsSync(DSPY_PYTHON)) { + throw new Error( + `${DSPY_PYTHON} is missing — run: cd clients/python && uv sync --frozen --extra dspy`, + ) + } + if (!existsSync(GEPA_PYTHON)) { + throw new Error(`${GEPA_PYTHON} is missing — run: ${GEPA_VENV_HINT}`) } // ── Engine + dispatch ─────────────────────────────────────────────── @@ -510,19 +903,22 @@ async function main(): Promise { // ── Judge: the shared trace-analyst quality judge, wrapped to // (a) score contract violations and execution failures as explicit zeros, - // (b) multiply the composite by the case's gold-mass weight so the mean - // GEPA climbs is micro-aligned, and + // (b) apply the round's weighting (gold mass in round 2, none in family + // mode), and // (c) append every verdict (raw + weighted + counts) to the score ledger. ── const baseJudge = traceAnalystQualityJudge() const zeroDimensions = Object.fromEntries(baseJudge.dimensions.map(({ key }) => [key, 0])) + const judgeVersionSuffix = FAMILY ? '+gepa-r3-family-plain' : '+gepa-r2-goldmass' const judge: JudgeConfig = { name: baseJudge.name, - ...(baseJudge.judgeVersion ? { judgeVersion: `${baseJudge.judgeVersion}+gepa-r2-goldmass` } : {}), + ...(baseJudge.judgeVersion + ? { judgeVersion: `${baseJudge.judgeVersion}${judgeVersionSuffix}` } + : {}), dimensions: [ ...baseJudge.dimensions, { key: 'raw_composite', description: 'unweighted judge composite' }, - { key: 'gold_mass_weight', description: 'labeled-issue count / pool max' }, + { key: 'gold_mass_weight', description: 'labeled-issue count / pool max; 1 in family mode' }, ], appliesTo: (scenario) => scenario.kind === 'trace-analyst', async score(input) { @@ -576,48 +972,101 @@ async function main(): Promise { const optimizerExecution = await loadOptimizerExecutionOwner(MODEL) + const round2Objective = + 'Rewrite the analysis policy of a recursive coding-trace analyst so it localizes every ' + + 'incorrect assistant step (contiguous failure blocks) more accurately. Per case, the ' + + 'quality score is the harmonic mean of labeled-issue recall and finding precision, ' + + 'averaged with critical-step accuracy when defined; that score is multiplied by the ' + + "case's gold-mass weight (labeled incorrect-step count / pool max), so cases with many " + + 'labeled incorrect steps carry proportionally more of the optimization signal — this ' + + 'matches the micro-F1 benchmark headline. Evaluation evidence carries per-case counts ' + + '(tp = labeled issues matched, fp = findings tied to no labeled issue, fn = labeled ' + + 'issues missed) plus the missed issue ids: use the counts, not just the scalar, to ' + + 'decide what to change — high fn concentrated in wide blocks means under-flagging block ' + + 'spans; high fp means over-flagging correct steps. The candidate is the complete ' + + 'instruction string: an analysis policy followed by a fixed output contract. The ' + + 'output-contract block must be preserved BYTE-IDENTICAL as the final section of the ' + + 'candidate — it defines the machine-parsed subject grammar (incorrect-steps--' + + '--consequence-) and the block caps, and any candidate missing it ' + + 'verbatim scores 0 without being executed. Edit only the policy text above the contract.' + const familyObjective = + 'Rewrite the analysis policy of a recursive coding-trace analyst so it localizes every ' + + 'incorrect assistant step (contiguous failure blocks) in long OpenHands and Terminus2 ' + + 'trajectories (median 48 steps, roughly double the trajectories the current policy was ' + + 'tuned on). Per case, the quality score is the harmonic mean of labeled-issue recall ' + + 'and finding precision, averaged with critical-step accuracy when defined; the ' + + 'optimized signal is the plain mean of that per-case score — every case counts ' + + 'equally. Measured failure profile of the current policy on these agent families: ' + + '(1) mislocalization dominates — nearly half the labeled gold mass is missed with ' + + 'every citation more than 2 steps away, yet crediting the existing citations at any ' + + 'distance would roughly double the score, so the analyst finds real incidents but ' + + 'reports the wrong steps; (2) under-enumeration — about 1.6 predicted blocks per run ' + + 'against about 3.0 labeled blocks per case, capping recall near 55% before any ' + + 'localization error; (3) early anchoring — predictions sit earlier in the trajectory ' + + 'than the labels (predicted start fraction p50 about 0.5 versus gold about 0.67): the ' + + 'analyst commits to the first convincing incident and never reaches the mid-late ' + + 'segment where the labels live. Evaluation evidence carries per-case counts (tp = ' + + 'labeled issues matched, fp = findings tied to no labeled issue, fn = labeled issues ' + + 'missed) plus the missed issue ids: use the counts, not just the scalar, to decide ' + + 'what to change. The candidate is the complete instruction string: an analysis policy ' + + 'followed by a fixed output contract. The output-contract block must be preserved ' + + 'BYTE-IDENTICAL as the final section of the candidate — it defines the machine-parsed ' + + 'subject grammar (incorrect-steps----consequence-) and the ' + + 'block caps, and any candidate missing it verbatim scores 0 without being executed. ' + + 'Edit only the policy text above the contract.' + + const engineRun = (maxEvaluations: number, runSeed: number): GepaEngineRun => ({ + engine: 'gepa', + maxEvaluations, + maxProposerCostUsd: MAX_PROPOSER_COST_USD, + maxConcurrency: EVAL_CONCURRENCY, + engineConfig: { + engine: { + capture_stdio: false, + max_workers: EVAL_CONCURRENCY, + parallel: EVAL_CONCURRENCY > 1, + raise_on_exception: true, + seed: runSeed, + }, + reflection: { reflection_minibatch_size: 3, skip_perfect_score: false }, + }, + }) + let recipe: GepaOptimizationRecipe + if (FAMILY && RECIPE_KIND === 'omni') { + // Same total evaluation budget as the engine recipe, split across GEPA's + // official Omni composition: two explore stages then a continuation that + // starts from the explore winner. maxWorkers 1 keeps the explore stages + // sequential so the global concurrency cap stays at EVAL_CONCURRENCY. + const exploreEvaluations = Math.max(1, Math.round(MAX_EVALUATIONS * 0.24)) + const continueEvaluations = MAX_EVALUATIONS - 2 * exploreEvaluations + if (continueEvaluations < 1) { + throw new Error( + `--max-evaluations ${MAX_EVALUATIONS} cannot fund the omni recipe (needs 2 explore stages + a continuation)`, + ) + } + recipe = { + kind: 'omni', + explore: [ + engineRun(exploreEvaluations, SEED + 101), + engineRun(exploreEvaluations, SEED + 102), + ], + continueWith: engineRun(continueEvaluations, SEED + 103), + maxWorkers: 1, + } + } else { + recipe = { kind: 'engine', run: engineRun(MAX_EVALUATIONS, SEED) } + } + const method = gepaOptimizationMethod({ - name: 'gepa-analyst-policy-r2', - objective: - 'Rewrite the analysis policy of a recursive coding-trace analyst so it localizes every ' + - 'incorrect assistant step (contiguous failure blocks) more accurately. Per case, the ' + - 'quality score is the harmonic mean of labeled-issue recall and finding precision, ' + - 'averaged with critical-step accuracy when defined; that score is multiplied by the ' + - "case's gold-mass weight (labeled incorrect-step count / pool max), so cases with many " + - 'labeled incorrect steps carry proportionally more of the optimization signal — this ' + - 'matches the micro-F1 benchmark headline. Evaluation evidence carries per-case counts ' + - '(tp = labeled issues matched, fp = findings tied to no labeled issue, fn = labeled ' + - 'issues missed) plus the missed issue ids: use the counts, not just the scalar, to ' + - 'decide what to change — high fn concentrated in wide blocks means under-flagging block ' + - 'spans; high fp means over-flagging correct steps. The candidate is the complete ' + - 'instruction string: an analysis policy followed by a fixed output contract. The ' + - 'output-contract block must be preserved BYTE-IDENTICAL as the final section of the ' + - 'candidate — it defines the machine-parsed subject grammar (incorrect-steps--' + - '--consequence-) and the block caps, and any candidate missing it ' + - 'verbatim scores 0 without being executed. Edit only the policy text above the contract.', + name: FAMILY ? `gepa-analyst-family-r3-${RECIPE_KIND}` : 'gepa-analyst-policy-r2', + objective: FAMILY ? familyObjective : round2Objective, background: `The frozen output contract (must appear verbatim at the end of every candidate):\n` + `${CONTRACT_BLOCK}`, - evaluationId: 'codetracebench-analyst-instructions-r2', - recipe: { - kind: 'engine', - run: { - engine: 'gepa', - maxEvaluations: MAX_EVALUATIONS, - maxProposerCostUsd: MAX_PROPOSER_COST_USD, - maxConcurrency: 1, - engineConfig: { - engine: { - capture_stdio: false, - max_workers: 1, - parallel: false, - raise_on_exception: true, - seed: SEED, - }, - reflection: { reflection_minibatch_size: 3, skip_perfect_score: false }, - }, - }, - }, + evaluationId: FAMILY + ? 'codetracebench-analyst-instructions-r3-oht2' + : 'codetracebench-analyst-instructions-r2', + recipe, optimizer: { model: MODEL, ...optimizerExecution, @@ -635,6 +1084,7 @@ async function main(): Promise { describeScenario: (scenario) => ({ id: scenario.id, trajectoryId: scenario.trajectoryId, + pool: scenario.pool, stratum: scenario.stratum, goldMass: scenario.goldMass, maxBlockWidth: scenario.maxBlockWidth, @@ -669,7 +1119,10 @@ async function main(): Promise { costLedger, } - console.log(`[gepa-analyst-r2] starting GEPA: maxEvaluations=${MAX_EVALUATIONS} runDir=${RUN_DIR}`) + console.log( + `[gepa-analyst-r${ROUND}] starting GEPA: recipe=${recipe.kind} ` + + `maxEvaluations=${MAX_EVALUATIONS} concurrency=${EVAL_CONCURRENCY} runDir=${RUN_DIR}`, + ) const started = Date.now() const result = await method.optimize(input) const winner = result.winnerSurface @@ -678,14 +1131,15 @@ async function main(): Promise { const baselineSha256 = sha256Digest(BASELINE) writeFileSync(join(RUN_DIR, 'winner-instructions.txt'), winner) console.log( - `[gepa-analyst-r2] optimize done in ${((Date.now() - started) / 60_000).toFixed(1)}min ` + + `[gepa-analyst-r${ROUND}] optimize done in ${((Date.now() - started) / 60_000).toFixed(1)}min ` + `evaluations=${result.provenance?.evaluationCount} winnerSha=${winnerSha256}`, ) // ── Selection comparison: harvest judged scores from the ledger; score any // missing (surface, scenario) pair directly through the same dispatch + - // judge. Reports the weighted metric (what GEPA optimized), the plain - // macro composite, plain macro f1, and pooled micro f1 side by side. ── + // judge. Reports the optimized metric alongside the plain macro + // composite, plain macro f1, and pooled micro f1, plus a per-pool + // breakdown so family results are readable per agent family. ── interface HarvestRow { rawComposite: number @@ -730,6 +1184,7 @@ async function main(): Promise { } const comparison: Array<{ scenarioId: string + pool: string stratum: string goldMass: number baseline: HarvestRow @@ -747,6 +1202,7 @@ async function main(): Promise { : (harvest.get(`${winnerSha256}:${scenario.id}`) ?? (await scoreDirect(winner, scenario))) comparison.push({ scenarioId: scenario.id, + pool: scenario.pool, stratum: scenario.stratum, goldMass: scenario.goldMass, baseline: baselineRow, @@ -758,14 +1214,40 @@ async function main(): Promise { } const mean = (values: number[]) => values.length === 0 ? 0 : values.reduce((total, value) => total + value, 0) / values.length + const comparisonBlock = (rows: readonly (typeof comparison)[number][]) => ({ + n: rows.length, + weighted: { + baselineMean: mean(rows.map((row) => row.baseline.weighted)), + winnerMean: mean(rows.map((row) => row.winner.weighted)), + deltaMean: mean(rows.map((row) => row.deltaWeighted)), + }, + plainMacroComposite: { + baselineMean: mean(rows.map((row) => row.baseline.rawComposite)), + winnerMean: mean(rows.map((row) => row.winner.rawComposite)), + deltaMean: mean(rows.map((row) => row.deltaRaw)), + }, + plainMacroF1: { + baselineMean: mean(rows.map((row) => row.baseline.f1)), + winnerMean: mean(rows.map((row) => row.winner.f1)), + deltaMean: mean(rows.map((row) => row.deltaF1)), + }, + microF1: { + baseline: microFromCounts(rows.map((row) => row.baseline.counts)), + winner: microFromCounts(rows.map((row) => row.winner.counts)), + }, + }) + const selectionPools = [...new Set(selectionScenarios.map((scenario) => scenario.pool))] const summary = { kind: 'agent-eval/gepa-analyst-campaign-summary', round: ROUND, + scenarioFamily: FAMILY ? 'oht2' : null, + recipeKind: recipe.kind, smoke: SMOKE, seed: SEED, model: MODEL, baseUrl: BASE_URL, runDir: RUN_DIR, + ...(FAMILY ? { splitPlanPath: FAMILY_SPLIT_PLAN_PATH } : {}), pools: poolProvenance, stratification, scenarioCounts: { @@ -775,46 +1257,37 @@ async function main(): Promise { }, trainScenarioIds: trainScenarios.map((scenario) => scenario.id), selectionScenarioIds: selectionScenarios.map((scenario) => scenario.id), - weighting: { - scheme: 'goldMass / maxGoldMass over the full 32-scenario pool', - maxGoldMass, - perScenario: Object.fromEntries( - scenarios.map((scenario) => [scenario.id, goldMassWeight(scenario)]), - ), - }, + weighting: FAMILY + ? { scheme: 'none — plain per-case judge composite (weight identically 1)' } + : { + scheme: 'goldMass / maxGoldMass over the full 32-scenario pool', + maxGoldMass, + perScenario: Object.fromEntries( + scenarios.map((scenario) => [scenario.id, goldMassWeight(scenario)]), + ), + }, maxEvaluations: MAX_EVALUATIONS, evaluationsUsed: result.provenance?.evaluationCount, + evalConcurrency: EVAL_CONCURRENCY, baselineSha256, winnerSha256, winnerChanged: winnerSha256 !== baselineSha256, // All metrics below are over the (spent) selection split; none certify. selectionComparison: { note: - 'paired per-scenario metrics over the selection split. weighted = gold-mass-weighted ' + - 'judge composite (the signal GEPA optimized); plainMacroComposite = unweighted judge ' + - 'composite; plainMacroF1 = mean per-scenario f1; microF1 = pooled per-step counts. ' + - 'Weighted and plain metrics are different quantities — never compare one arm on ' + - 'weighted against another on plain.', - n: comparison.length, - weighted: { - baselineMean: mean(comparison.map((row) => row.baseline.weighted)), - winnerMean: mean(comparison.map((row) => row.winner.weighted)), - deltaMean: mean(comparison.map((row) => row.deltaWeighted)), - }, - plainMacroComposite: { - baselineMean: mean(comparison.map((row) => row.baseline.rawComposite)), - winnerMean: mean(comparison.map((row) => row.winner.rawComposite)), - deltaMean: mean(comparison.map((row) => row.deltaRaw)), - }, - plainMacroF1: { - baselineMean: mean(comparison.map((row) => row.baseline.f1)), - winnerMean: mean(comparison.map((row) => row.winner.f1)), - deltaMean: mean(comparison.map((row) => row.deltaF1)), - }, - microF1: { - baseline: microFromCounts(comparison.map((row) => row.baseline.counts)), - winner: microFromCounts(comparison.map((row) => row.winner.counts)), - }, + 'paired per-scenario metrics over the selection split. weighted = the signal GEPA ' + + 'optimized (gold-mass-weighted judge composite in round 2; identical to the plain ' + + 'composite in family mode); plainMacroComposite = unweighted judge composite; ' + + 'plainMacroF1 = mean per-scenario f1; microF1 = pooled per-step counts. Weighted ' + + 'and plain metrics are different quantities — never compare one arm on weighted ' + + 'against another on plain.', + ...comparisonBlock(comparison), + perPool: Object.fromEntries( + selectionPools.map((pool) => [ + pool, + comparisonBlock(comparison.filter((row) => row.pool === pool)), + ]), + ), perScenario: comparison, }, methodCost: result.cost, @@ -829,12 +1302,12 @@ async function main(): Promise { writeFileSync(join(RUN_DIR, 'summary.json'), `${JSON.stringify(summary, null, 2)}\n`) console.log(JSON.stringify(summary.selectionComparison, null, 2)) console.log( - `[gepa-analyst-r2] total ledger cost=$${costLedger.summary().totalCostUsd.toFixed(4)} ` + + `[gepa-analyst-r${ROUND}] total ledger cost=$${costLedger.summary().totalCostUsd.toFixed(4)} ` + `summary=${join(RUN_DIR, 'summary.json')}`, ) } main().catch((error) => { - console.error('[gepa-analyst-r2] FAILED:', error) + console.error('[gepa-analyst] FAILED:', error) process.exitCode = 1 }) From 22011c4fd0ff410d2c9a5d12ad8f65819248805a Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Mon, 3 Aug 2026 13:16:04 -0600 Subject: [PATCH 2/2] =?UTF-8?q?fix(campaign):=20family-GEPA=20budget=20geo?= =?UTF-8?q?metry=20=E2=80=94=20serial=20metric=20calls=20with=20one-iterat?= =?UTF-8?q?ion=20callback=20headroom?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit gepa 0.1.4 checks its metric-call budget once per iteration-loop pass, so after a passing check it can still spend a full iteration (parent minibatch + proposed minibatch + accepted full-valset evaluation). When the callback's hard cap equals GEPA's own target, that boundary overshoot draws an HTTP 429 from the callback and raise_on_exception kills the run — measured on both smoke shapes. The family recipes now give the callback cap exactly one worst-case iteration burst of headroom above the upstream target, using the bridge's maxConcurrency derate as the headroom knob while engine parallel=false keeps execution strictly serial (also holding the z.ai seat far below its ~5-concurrent 429 limit). The family cost ceiling rises to $30 to keep ~2x margin over the larger worst-case call count, and the smoke prefers unsolved trajectories so it demonstrates the nonzero-score path. The mini-SWE round-2 recipe is byte-identical to its completed run. --- scripts/gepa-analyst-campaign.ts | 118 ++++++++++++++++++++++++------- 1 file changed, 93 insertions(+), 25 deletions(-) diff --git a/scripts/gepa-analyst-campaign.ts b/scripts/gepa-analyst-campaign.ts index 0031e356..cdb166f2 100644 --- a/scripts/gepa-analyst-campaign.ts +++ b/scripts/gepa-analyst-campaign.ts @@ -124,6 +124,8 @@ const MODEL = 'glm-5.2' const BASE_URL = 'https://api.z.ai/api/coding/paas/v4' const PRICING = { inputUsdPerMillion: 0.6, outputUsdPerMillion: 2.2 } const EVALUATIONS_OVERRIDE = argValue('--max-evaluations') +/** In family mode this is the upstream evaluation target GEPA optimizes with, + * not the callback's hard cap — see the budget-geometry note at the recipe. */ const MAX_EVALUATIONS = EVALUATIONS_OVERRIDE ? Number.parseInt(EVALUATIONS_OVERRIDE, 10) : SMOKE @@ -131,17 +133,17 @@ const MAX_EVALUATIONS = EVALUATIONS_OVERRIDE : FAMILY ? 50 : 60 -const COST_CEILING_USD = SMOKE ? 3 : 20 +// Family analyses run on trajectories roughly twice mini-SWE length, and the +// callback cap intentionally exceeds the upstream target by one iteration +// burst, so the family ceiling carries ~2x headroom over expected spend — a +// runaway guard, not a target. +const COST_CEILING_USD = SMOKE ? 3 : FAMILY ? 30 : 20 const MAX_PROPOSER_COST_USD = SMOKE ? 1 : 5 const GEPA_TIMEOUT_MS = SMOKE ? 3_600_000 : 28_800_000 const ANALYSIS_TIMEOUT_MS = 1_200_000 const EVAL_COST_PHASE = 'gepa.analyst-evaluation' -/** - * Concurrent analyses cap. The z.ai glm seat starts returning 429s at ~5 - * concurrent requests (measured), so the family campaign runs at most 3 - * evaluations in flight; the mini-SWE round stays strictly serial. - */ -const EVAL_CONCURRENCY = FAMILY ? 3 : 1 +/** GEPA reflection minibatch size; also a term of the budget-overshoot bound. */ +const REFLECTION_MINIBATCH_SIZE = 3 const DSPY_PYTHON = join(ROOT, 'clients/python/.venv/bin/python') /** Omni needs GEPA's official composition functions, present only in the source venv. */ const GEPA_PYTHON = join( @@ -514,13 +516,22 @@ async function main(): Promise { // Smoke touches one train + one selection row from openhands and one train // row from terminus2 — 3 scenarios, both families, both split roles. + // Zero-finding runs concentrate on solved trajectories on these families, + // so the smoke prefers unsolved rows: it must demonstrate the nonzero-score + // path, not just the plumbing. + const preferUnsolved = (pool: FamilyPoolSplit, rows: readonly FamilyPlanRow[]) => { + const unsolved = rows.find( + (row) => pool.rowsByTrajId.get(row.trajId)?.solved === false, + ) + return unsolved ?? rows[0]! + } const activeByPool = new Map( split.map((pool) => [ pool.name, SMOKE ? { - train: [pool.train[0]!], - selection: pool.name === 'openhands' ? [pool.selection[0]!] : [], + train: [preferUnsolved(pool, pool.train)], + selection: pool.name === 'openhands' ? [preferUnsolved(pool, pool.selection)] : [], } : { train: [...pool.train], selection: [...pool.selection] }, ]), @@ -1015,28 +1026,52 @@ async function main(): Promise { 'block caps, and any candidate missing it verbatim scores 0 without being executed. ' + 'Edit only the policy text above the contract.' - const engineRun = (maxEvaluations: number, runSeed: number): GepaEngineRun => ({ + /** + * Family budget geometry — how one engine run's numbers are derived. + * + * gepa 0.1.4 checks its metric-call budget only once per iteration-loop + * pass, so after a passing check it can still spend one full iteration: + * parent minibatch + proposed minibatch + (on acceptance) a full valset + * evaluation. If the callback's hard cap equals GEPA's own target, that + * boundary overshoot hits the cap, the callback returns HTTP 429, and + * raise_on_exception kills the whole run (measured on both smoke shapes). + * + * The fix is headroom, not tolerance: the callback cap exceeds the + * upstream target by exactly one worst-case iteration burst. The bridge + * derates the upstream limit by maxConcurrency - 1, so maxConcurrency + * serves here as the headroom knob only — engineConfig.engine.parallel is + * false, execution stays strictly serial (every gepa parallel path gates + * on that flag, and serial calls also keep the z.ai seat far below its + * ~5-concurrent 429 limit), and the injected max_workers value is inert. + * + * Cost stays bounded by the callback cap (upstream + burst) and the run + * cost ceiling; GEPA stops at the first boundary at or past its target. + */ + const boundaryBurst = 2 * REFLECTION_MINIBATCH_SIZE + selectionScenarios.length + const familyEngineRun = (upstreamEvaluations: number, runSeed: number): GepaEngineRun => ({ engine: 'gepa', - maxEvaluations, + maxEvaluations: upstreamEvaluations + boundaryBurst, maxProposerCostUsd: MAX_PROPOSER_COST_USD, - maxConcurrency: EVAL_CONCURRENCY, + maxConcurrency: boundaryBurst + 1, engineConfig: { engine: { capture_stdio: false, - max_workers: EVAL_CONCURRENCY, - parallel: EVAL_CONCURRENCY > 1, + parallel: false, raise_on_exception: true, seed: runSeed, }, - reflection: { reflection_minibatch_size: 3, skip_perfect_score: false }, + reflection: { + reflection_minibatch_size: REFLECTION_MINIBATCH_SIZE, + skip_perfect_score: false, + }, }, }) let recipe: GepaOptimizationRecipe if (FAMILY && RECIPE_KIND === 'omni') { - // Same total evaluation budget as the engine recipe, split across GEPA's - // official Omni composition: two explore stages then a continuation that - // starts from the explore winner. maxWorkers 1 keeps the explore stages - // sequential so the global concurrency cap stays at EVAL_CONCURRENCY. + // Same total upstream evaluation budget as the engine recipe, split + // across GEPA's official Omni composition: two explore stages then a + // continuation that starts from the explore winner. maxWorkers 1 keeps + // the explore stages sequential. const exploreEvaluations = Math.max(1, Math.round(MAX_EVALUATIONS * 0.24)) const continueEvaluations = MAX_EVALUATIONS - 2 * exploreEvaluations if (continueEvaluations < 1) { @@ -1047,14 +1082,37 @@ async function main(): Promise { recipe = { kind: 'omni', explore: [ - engineRun(exploreEvaluations, SEED + 101), - engineRun(exploreEvaluations, SEED + 102), + familyEngineRun(exploreEvaluations, SEED + 101), + familyEngineRun(exploreEvaluations, SEED + 102), ], - continueWith: engineRun(continueEvaluations, SEED + 103), + continueWith: familyEngineRun(continueEvaluations, SEED + 103), maxWorkers: 1, } + } else if (FAMILY) { + recipe = { kind: 'engine', run: familyEngineRun(MAX_EVALUATIONS, SEED) } } else { - recipe = { kind: 'engine', run: engineRun(MAX_EVALUATIONS, SEED) } + recipe = { + kind: 'engine', + run: { + engine: 'gepa', + maxEvaluations: MAX_EVALUATIONS, + maxProposerCostUsd: MAX_PROPOSER_COST_USD, + maxConcurrency: 1, + engineConfig: { + engine: { + capture_stdio: false, + max_workers: 1, + parallel: false, + raise_on_exception: true, + seed: SEED, + }, + reflection: { + reflection_minibatch_size: REFLECTION_MINIBATCH_SIZE, + skip_perfect_score: false, + }, + }, + }, + } } const method = gepaOptimizationMethod({ @@ -1121,7 +1179,8 @@ async function main(): Promise { console.log( `[gepa-analyst-r${ROUND}] starting GEPA: recipe=${recipe.kind} ` + - `maxEvaluations=${MAX_EVALUATIONS} concurrency=${EVAL_CONCURRENCY} runDir=${RUN_DIR}`, + `maxEvaluations=${MAX_EVALUATIONS}${FAMILY ? ` boundaryBurst=${boundaryBurst}` : ''} ` + + `runDir=${RUN_DIR}`, ) const started = Date.now() const result = await method.optimize(input) @@ -1268,7 +1327,16 @@ async function main(): Promise { }, maxEvaluations: MAX_EVALUATIONS, evaluationsUsed: result.provenance?.evaluationCount, - evalConcurrency: EVAL_CONCURRENCY, + ...(FAMILY + ? { + budgetGeometry: { + upstreamEvaluationTarget: MAX_EVALUATIONS, + boundaryBurst, + callbackCapPerRun: MAX_EVALUATIONS + boundaryBurst, + execution: 'serial (engine parallel=false)', + }, + } + : {}), baselineSha256, winnerSha256, winnerChanged: winnerSha256 !== baselineSha256,