diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml
index 5ed1a2f8..5b0d88e1 100644
--- a/.github/workflows/evals.yml
+++ b/.github/workflows/evals.yml
@@ -8,9 +8,7 @@ name: Model evals
# to run locally: every recorded report was single-model, so "works with Flow" meant
# "worked once, with one provider".
#
-# Evals cost money and need credentials, so they stay out of the PR gate. This runs
-# weekly against at least two providers and publishes the report as an artifact,
-# with the qualification thresholds applied by `scripts/qualify-release.ts`.
+# Evals cost money and need credentials, so they stay out of the PR gate.
# See docs/adr/0010-declared-canonical-gate.md.
on:
@@ -80,6 +78,7 @@ jobs:
echo "::notice::No eval model matrix or provider credentials configured; skipping."
echo "models=" >> "$GITHUB_OUTPUT"
else
+ FLOW_EVAL_MODEL="$models" bun run scripts/check-release-models.ts
echo "models=$models" >> "$GITHUB_OUTPUT"
fi
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 76eec45a..019941ca 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -16,6 +16,8 @@ One short entry per release, written for users deciding whether to upgrade.
file to retain complete validation evidence.
- Session v5 schema, validation and reviewer gates, and continuation routing are
unchanged. The OpenCode host used by the release checks is pinned to 1.18.31.
+- This release uses a GPT-6 Sol-only qualification profile. Its evidence makes
+ no cross-provider reliability claim. Later releases retain the two-provider gate.
Upgrade with `opencode plugin opencode-plugin-flow@9.1.0 --global --force`.
diff --git a/docs/development.md b/docs/development.md
index c46e59a4..4f57e256 100644
--- a/docs/development.md
+++ b/docs/development.md
@@ -151,7 +151,8 @@ Follow the [frozen-candidate sequence](release-qualification.md#running-it):
finish fixes and dependency updates, pass deterministic checks, approve paid
evals.
`bun run qualify -- --campaign-dir
--canary ` seals the
-two-provider campaign, exact-artifact canary and grader evidence. Commit that
+policy grid, canary and grader evidence. Version 9.1.0 uses GPT-6 Sol only.
+Other versions require two providers. Commit that
bundle before tagging; never substitute interrupted results for qualification.
Release tags use `v`. Blocking release checks: the normal
diff --git a/docs/release-qualification.md b/docs/release-qualification.md
index 95a47320..88a885a6 100644
--- a/docs/release-qualification.md
+++ b/docs/release-qualification.md
@@ -9,7 +9,7 @@ This page owns release thresholds, candidate freezing, and publication order.
| Threshold | Value | Why |
| --- | --- | --- |
-| Distinct providers | ≥ 2 | Every report recorded before this policy was single-model, so "works with Flow" meant "worked once, with one provider". |
+| Distinct providers | ≥ 2; 9.1.0 only: 1 | 9.1.0 pins `openai/gpt-6-sol`. Its evidence covers OpenAI only. Later releases require two routes. |
| False completions | 0 | A `completed` closure the document itself contradicts is the failure Flow exists to prevent. |
| Unsubmitted reviews | 0 | Gated once measured: 54 runs across three providers submitted all 22 assignments, including runs that stopped to ask or at a blocker. |
| Scored attempts per provider | 3 at 100%; 10 at 90% | The frozen release plan gives each threshold enough trials to express its allowed failures. |
@@ -43,8 +43,7 @@ canary inside its window. The wall clock decided this until 9.0.1, which made
every published offline release stop verifying seventy-two hours after its
baseline was measured.
-A new scenario needs an explicit release-policy decision. Any required canonical
-case missing from the report fails qualification.
+New scenarios need a policy decision. Missing required cases fail qualification.
A non-product attempt never shrinks the required sample. The frozen plan retains
one environment reserve per provider and case. A retryable provider or host
@@ -53,10 +52,10 @@ second external failure or an unallowed ask leaves a gap. Product and evaluator
failures never activate reserves. Evaluator failure is `NOT VERIFIED`;
persistence failure stops without a finalized report.
-Repository code owns the ordered release catalog. Persisted `catalog.json` is only a
-witness and must match it exactly. The two-provider grid has 76 primary cells and
-16 predeclared environment reserves; ordinary, narrowed, dynamically extended, or
-merged summary reports cannot qualify.
+Repository code owns the ordered release catalog; persisted `catalog.json` must
+match. Version 9.1.0 uses 38 primary cells and eight reserves on GPT-6 Sol only.
+Other versions use 76 primary cells and 16 reserves on distinct providers.
+Narrowed, extended, or merged summary reports cannot qualify.
Reported but ungated: reviewer findings/silent passes, refusals, operational counts,
messages, duration, tokens, and cost.
@@ -95,16 +94,17 @@ Finish or close active sessions before changing Flow versions in either directio
Finish code, dependency, version and changelog changes first. Pass frozen install,
`bun run check`, `bun run replay`, audit, live smoke and CI before paid qualification.
-Freeze packed contents and evaluator inputs, then run the full two-provider matrix
+Freeze packed contents and evaluator inputs, then run the versioned matrix
on the canonical Linux host. Run a fresh canary against its exact `artifact.tgz`,
seal/regrade the bundle, and commit only evidence without changing measured inputs.
Recheck final main CI and exact artifact identity before tagging `v`.
Authorize dispatches using the [paid-run budget](../.agents/plans/05-release-simplification/README.md#authorize-paid-work).
Keep that ledger across retries. Budget-stopped campaigns cannot qualify.
+For 9.1.0, run the pinned OpenAI model.
```bash
-bun run eval -- --release --model openai/gpt-6-sol --model xai/grok-4.6
+bun run eval -- --release --model openai/gpt-6-sol
bun run eval:canary -- prepare --report /report.json --out
# Run the prepared fixture, then record its session and transcript.
bun run eval:canary -- record
diff --git a/evals/README.md b/evals/README.md
index 30f33592..1375f4c1 100644
--- a/evals/README.md
+++ b/evals/README.md
@@ -56,8 +56,9 @@ plugin tuple configuration, and records the same selection in provenance.
Release sampling rejects reviewer overrides.
Ordinary runs use one sequential queue per model, with up to four queues in flight.
-Release mode is strictly sequential (`--concurrency 1`): 76 primary targets and one
-environment reserve per provider/case, at most 92 attempts. Only retained retryable
+Release mode is strictly sequential (`--concurrency 1`). Version 9.1.0 pins
+`openai/gpt-6-sol` with 38 primary targets and eight reserves. Other versions
+require two providers, 76 primary targets and 16 reserves. Only retained retryable
host/provider failures activate reserves, never product failures. Results are
persisted in declared order even when ordinary queues finish out of order.
@@ -410,24 +411,22 @@ a suite that measures nothing look identical from here, so read one run anyway.
## Three tiers, three prices
-One price for every question is what made this suite something run at release rather
-than during work:
+Use three eval tiers:
| Tier | Command | Cost | Answers |
| --- | --- | --- | --- |
| Replay | `bun run replay` | free | does the runtime still reach the same outcome on decisions a model already made? |
| Smoke | `bun run eval:smoke -- --model ` | one model, one attempt | did a prompt change break the ordinary path? |
-| Matrix | `bun run eval -- --release --model --model ` | real money | may this be released? |
+| Matrix | `bun run eval -- --release --model [--model ]` | real money | may this be released? |
Only the matrix qualifies a release. A replay is evidence about the runtime and none
about the prompts; a single attempt of a stochastic scenario is not a rate.
## Multi-model matrix
-Every report recorded before this existed was single-model, so "works with Flow"
-meant "worked once, with one provider". Qualification needs at least two distinct
-providers, and `.github/workflows/evals.yml` runs the matrix weekly and on demand —
-never in a gate a contributor waits on, since a full pass costs real money.
+Version 9.1.0 requires only `openai/gpt-6-sol`; it makes no cross-provider claim.
+Other versions require two distinct providers. `.github/workflows/evals.yml`
+runs the matrix weekly and on demand, outside contributor gates.
## Using evals to change prompts
@@ -580,8 +579,9 @@ distinction that matters: one pass in six and six in six are different findings.
## Cost
-Release qualification schedules 76 primary attempts across eight scenarios and
-two providers, plus at most 16 environment reserves. Ordinary campaign size depends
+Version 9.1.0 schedules 38 primary attempts and eight reserves on GPT-6 Sol.
+Its evidence supports only that route. Other versions schedule 76 primary attempts
+and 16 reserves across two providers. Ordinary campaign size depends
on the selected scenarios, models and repeats. Use `--scenario` while iterating;
cost depends on model pricing and the work performed, not just scenario count.
diff --git a/evals/qualification-regrade.ts b/evals/qualification-regrade.ts
index 742f4ccf..4a17f877 100644
--- a/evals/qualification-regrade.ts
+++ b/evals/qualification-regrade.ts
@@ -27,8 +27,8 @@ import { readQualificationBundle } from "./qualification-bundle.js";
import {
RELEASE_ANALYSIS_SHA256,
RELEASE_MAX_CAMPAIGN_AGE_MS,
- RELEASE_POLICY_SHA256,
releaseGraderBundle,
+ releasePolicySha256,
} from "./release-policy.js";
import type { ArtifactIdentity, ValidatedReport } from "./report.js";
import { SCENARIOS } from "./scenarios.js";
@@ -310,7 +310,11 @@ export async function regradeQualificationBundle(input: {
),
};
if (
- policy.policySha256 !== RELEASE_POLICY_SHA256 ||
+ policy.policySha256 !==
+ releasePolicySha256(
+ (expectedStored as { artifact: ArtifactIdentity }).artifact
+ .packageVersion,
+ ) ||
policy.analysisSha256 !== RELEASE_ANALYSIS_SHA256 ||
canonicalJson(policy.graderBundle) !== canonicalJson(bundledGrader) ||
canonicalJson(policy.graderBundle) !==
diff --git a/evals/release-policy.ts b/evals/release-policy.ts
index 1484c1a9..65ed2328 100644
--- a/evals/release-policy.ts
+++ b/evals/release-policy.ts
@@ -98,7 +98,34 @@ const RELEASE_POLICY_INPUT = [
const parsed = parseCaseCatalog(RELEASE_POLICY_INPUT);
if (!parsed.ok) throw new Error("Repository release policy is invalid.");
-const RELEASE_CATALOG = parsed.value;
+const STANDARD_RELEASE_CATALOG = parsed.value;
+
+export type ReleaseProfile = {
+ readonly catalog: ValidatedCaseCatalog;
+ readonly requiredModels: readonly ModelIdentity[] | null;
+};
+
+const OPENAI_ONLY_9_1_0: ReleaseProfile = {
+ catalog: STANDARD_RELEASE_CATALOG.map((row) => ({ ...row, minProviders: 1 })),
+ requiredModels: [
+ {
+ routeProvider: "openai",
+ gateway: null,
+ family: "gpt-6-sol",
+ model: "gpt-6-sol",
+ revision: null,
+ },
+ ],
+};
+
+const STANDARD_RELEASE: ReleaseProfile = {
+ catalog: STANDARD_RELEASE_CATALOG,
+ requiredModels: null,
+};
+
+export function releaseProfile(packageVersion: string): ReleaseProfile {
+ return packageVersion === "9.1.0" ? OPENAI_ONLY_9_1_0 : STANDARD_RELEASE;
+}
export const RELEASE_ANALYSIS_SHA256 = canonicalSha256("flow-v2-analysis-v1", {
kind: "rate",
@@ -113,42 +140,77 @@ export const RELEASE_HOST_POLICY = {
reviewerSteps: null,
} as const;
-export const RELEASE_POLICY_SHA256 = canonicalSha256("flow-release-policy-v1", {
- catalog: RELEASE_CATALOG,
- host: RELEASE_HOST_POLICY,
- analysisSha256: RELEASE_ANALYSIS_SHA256,
- environmentReservesPerStratum: RELEASE_ENVIRONMENT_RESERVES_PER_STRATUM,
-});
+export function releasePolicySha256(packageVersion: string): string {
+ const profile = releaseProfile(packageVersion);
+ return canonicalSha256("flow-release-policy-v1", {
+ catalog: profile.catalog,
+ ...(profile.requiredModels === null
+ ? {}
+ : { requiredModels: profile.requiredModels }),
+ host: RELEASE_HOST_POLICY,
+ analysisSha256: RELEASE_ANALYSIS_SHA256,
+ environmentReservesPerStratum: RELEASE_ENVIRONMENT_RESERVES_PER_STRATUM,
+ });
+}
+
+export const RELEASE_POLICY_SHA256 = releasePolicySha256("standard");
export const RELEASE_POLICY_CATALOG_SHA256 = canonicalSha256(
"flow-evaluator-policy-catalog-v1",
- RELEASE_CATALOG,
+ STANDARD_RELEASE_CATALOG,
);
-export function releaseCatalog(): ValidatedCaseCatalog {
- return RELEASE_CATALOG;
+export function releaseCatalog(
+ packageVersion = "standard",
+): ValidatedCaseCatalog {
+ return releaseProfile(packageVersion).catalog;
}
export function releaseCaseIds(): readonly string[] {
- return RELEASE_CATALOG.map((policy) => policy.caseId);
+ return STANDARD_RELEASE_CATALOG.map((policy) => policy.caseId);
}
export function releaseAttemptsFor(caseId: string): number {
- const policy = RELEASE_CATALOG.find((item) => item.caseId === caseId);
+ const policy = STANDARD_RELEASE_CATALOG.find(
+ (item) => item.caseId === caseId,
+ );
if (!policy) throw new Error(`No release policy for ${caseId}.`);
return policy.minScoredAttempts;
}
-export function releaseMinimumProviders(): number {
- return Math.max(...RELEASE_CATALOG.map((policy) => policy.minProviders));
+export function releaseMinimumProviders(packageVersion = "standard"): number {
+ return Math.max(
+ ...releaseCatalog(packageVersion).map((policy) => policy.minProviders),
+ );
+}
+
+export function assertReleaseModels(
+ models: readonly ModelIdentity[],
+ packageVersion: string,
+): void {
+ const profile = releaseProfile(packageVersion);
+ const minimum = releaseMinimumProviders(packageVersion);
+ if (
+ models.length !== minimum ||
+ new Set(models.map((model) => model.routeProvider)).size !== minimum ||
+ (profile.requiredModels !== null &&
+ canonicalJson(models) !== canonicalJson(profile.requiredModels))
+ ) {
+ throw new Error(
+ profile.requiredModels !== null
+ ? `Release ${packageVersion} requires exactly ${profile.requiredModels.map((model) => `${model.routeProvider}/${model.model}`).join(", ")} with its canonical direct-route identity.`
+ : `Release requires exactly ${minimum} models on distinct route providers.`,
+ );
+ }
}
export function releasePrimaryCellsFor(
models: readonly ModelIdentity[],
+ packageVersion = "standard",
): ScheduledCell[] {
let slot = 0;
return models.flatMap((model) =>
- RELEASE_CATALOG.flatMap((policy) =>
+ releaseCatalog(packageVersion).flatMap((policy) =>
Array.from({ length: policy.minScoredAttempts }, (_, repetition) => {
const block = slot;
slot += 1;
@@ -175,10 +237,11 @@ export function releasePrimaryCellsFor(
export function releaseCellsFor(
models: readonly ModelIdentity[],
+ packageVersion = "standard",
): ScheduledCell[] {
- const primary = releasePrimaryCellsFor(models);
+ const primary = releasePrimaryCellsFor(models, packageVersion);
const reserves = models.flatMap((model) =>
- RELEASE_CATALOG.map((policy) => {
+ releaseCatalog(packageVersion).map((policy) => {
const identity = canonicalSha256("flow-v2-environment-reserve-v1", {
model: `${model.routeProvider}/${model.model}`,
scenario: policy.caseId,
@@ -204,11 +267,12 @@ export function releaseCellsFor(
export function releaseRandomizationSeed(
models: readonly ModelIdentity[],
+ packageVersion = "standard",
): string {
return canonicalSha256("flow-v2-seed-v1", {
models: models.map((model) => `${model.routeProvider}/${model.model}`),
scenarios: releaseCaseIds(),
- releasePolicySha256: RELEASE_POLICY_SHA256,
+ releasePolicySha256: releasePolicySha256(packageVersion),
});
}
@@ -260,17 +324,19 @@ export function assertReleaseScenarioOrder(
export function assertExactReleaseCatalog(
input: unknown,
+ packageVersion = "standard",
): ValidatedCaseCatalog {
const supplied = parseCaseCatalog(input);
+ const catalog = releaseCatalog(packageVersion);
if (
!supplied.ok ||
- canonicalJson(supplied.value) !== canonicalJson(RELEASE_CATALOG)
+ canonicalJson(supplied.value) !== canonicalJson(catalog)
) {
throw new Error(
"Persisted catalog does not match repository release policy.",
);
}
- return RELEASE_CATALOG;
+ return catalog;
}
export function selectReleaseScenarios<
diff --git a/evals/run.ts b/evals/run.ts
index 191c7781..59764849 100644
--- a/evals/run.ts
+++ b/evals/run.ts
@@ -94,6 +94,7 @@ import {
} from "./provenance.js";
import {
assertReleaseHost,
+ assertReleaseModels,
assertReleaseScenarioOrder,
RELEASE_ANALYSIS_SHA256,
RELEASE_HOST_POLICY,
@@ -102,7 +103,6 @@ import {
releaseCellsFor,
releaseGraderBundle,
releaseHostConfigSha256,
- releaseMinimumProviders,
releaseRandomizationSeed,
releaseScenarioCatalog,
selectReleaseScenarios,
@@ -283,7 +283,7 @@ export function caseCatalogFor(
): ValidatedCaseCatalog {
if (sampling.kind === "release") {
assertReleaseScenarioOrder(scenarios);
- return releaseCatalog();
+ return releaseCatalog(sampling.packageVersion ?? packageJson.version);
}
const parsed = parseCaseCatalog(
scenarios.map((scenario) => ({
@@ -310,7 +310,7 @@ export function caseCatalogFor(
export type EvalSampling =
| { readonly kind: "ordinary"; readonly repeat: number }
- | { readonly kind: "release" };
+ | { readonly kind: "release"; readonly packageVersion?: string };
export function attemptsForScenario(
scenarioId: string,
@@ -332,13 +332,22 @@ export function campaignPlanFor(input: {
}): CampaignPlan {
if (input.sampling.kind === "release") {
assertReleaseScenarioOrder(input.scenarios);
+ assertReleaseModels(
+ input.models.map(legacyRequestedModel),
+ input.sampling.packageVersion ?? packageJson.version,
+ );
}
let slot = 0;
const ordinaryRepeat =
input.sampling.kind === "ordinary" ? input.sampling.repeat : null;
const cells =
ordinaryRepeat === null
- ? releaseCellsFor(input.models.map(legacyRequestedModel))
+ ? releaseCellsFor(
+ input.models.map(legacyRequestedModel),
+ input.sampling.kind === "release"
+ ? (input.sampling.packageVersion ?? packageJson.version)
+ : packageJson.version,
+ )
: input.models.flatMap((model) =>
input.scenarios.flatMap((scenario) =>
Array.from({ length: ordinaryRepeat }, (_, repetition) => {
@@ -373,7 +382,10 @@ export function campaignPlanFor(input: {
planSha256: `sha256:${"0".repeat(64)}`,
randomizationSeed:
input.sampling.kind === "release"
- ? releaseRandomizationSeed(input.models.map(legacyRequestedModel))
+ ? releaseRandomizationSeed(
+ input.models.map(legacyRequestedModel),
+ input.sampling.packageVersion ?? packageJson.version,
+ )
: canonicalSha256("flow-v2-seed-v1", {
models: input.models,
scenarios: input.scenarios.map((scenario) => scenario.id),
@@ -575,23 +587,21 @@ function parseArgs(argv: string[]) {
process.exit(2);
}
if (release) {
- const providers = new Set();
for (const model of models) {
try {
- providers.add(legacyRequestedModel(model).routeProvider);
+ legacyRequestedModel(model);
} catch (error) {
console.error(error instanceof Error ? error.message : String(error));
process.exit(2);
}
}
- const minimumProviders = releaseMinimumProviders();
- if (
- models.length !== minimumProviders ||
- providers.size !== minimumProviders
- ) {
- console.error(
- `--release requires exactly ${minimumProviders} models on distinct route providers.`,
+ try {
+ assertReleaseModels(
+ models.map(legacyRequestedModel),
+ packageJson.version,
);
+ } catch (error) {
+ console.error(error instanceof Error ? error.message : String(error));
process.exit(2);
}
}
diff --git a/scripts/check-release-models.ts b/scripts/check-release-models.ts
new file mode 100644
index 00000000..d89a48df
--- /dev/null
+++ b/scripts/check-release-models.ts
@@ -0,0 +1,22 @@
+#!/usr/bin/env bun
+
+import { normalizeRequestedModel } from "../evals/provenance.js";
+import { assertReleaseModels } from "../evals/release-policy.js";
+import packageJson from "../package.json" with { type: "json" };
+
+const configured = process.env.FLOW_EVAL_MODEL?.trim();
+if (!configured)
+ throw new Error("FLOW_EVAL_MODEL is required for a release matrix.");
+
+const models = configured.split(",").map((entry) => {
+ const modelId = entry.trim();
+ const boundary = modelId.indexOf("/");
+ const routedModel = modelId.slice(boundary + 1);
+ return normalizeRequestedModel({
+ modelId,
+ gateway: routedModel.includes("/") ? modelId.slice(0, boundary) : null,
+ family: routedModel,
+ revision: null,
+ });
+});
+assertReleaseModels(models, packageJson.version);
diff --git a/scripts/patch-release.ts b/scripts/patch-release.ts
index f317b1dd..580928ff 100644
--- a/scripts/patch-release.ts
+++ b/scripts/patch-release.ts
@@ -244,6 +244,8 @@ export async function assertPatchReleaseEvidence(
});
if (baseline.bundleSha256 !== record.baseline.bundleSha256)
throw new Error("Baseline qualification bundle changed.");
+ if (new Set(baseline.summary.providers.map((item) => item.provider)).size < 2)
+ throw new Error("Patch exception requires a two-provider baseline.");
return {
bundleSha256: hash(bytes),
notes: [
diff --git a/scripts/qualify-release.ts b/scripts/qualify-release.ts
index 5cab8099..20474b15 100644
--- a/scripts/qualify-release.ts
+++ b/scripts/qualify-release.ts
@@ -34,14 +34,14 @@ import {
} from "../evals/qualification-bundle.js";
import {
assertExactReleaseCatalog,
+ assertReleaseModels,
RELEASE_ANALYSIS_SHA256,
- RELEASE_POLICY_SHA256,
releaseCatalog,
releaseCellsFor,
releaseGraderBundle,
releaseGraderSourceBundle,
releaseHostConfigSha256,
- releaseMinimumProviders,
+ releasePolicySha256,
releaseRandomizationSeed,
releaseScenarioCatalog,
} from "../evals/release-policy.js";
@@ -131,7 +131,10 @@ function sameJson(left: unknown, right: unknown): boolean {
return canonicalJson(left) === canonicalJson(right);
}
-function assertFrozenReleasePlan(report: ValidatedReport): void {
+function assertFrozenReleasePlan(
+ report: ValidatedReport,
+ packageVersion: string,
+): void {
const models: ModelIdentity[] = [];
for (const cell of report.plan.cells) {
if (
@@ -141,15 +144,8 @@ function assertFrozenReleasePlan(report: ValidatedReport): void {
models.push(cell.managerModel);
}
}
- if (
- models.length !== releaseMinimumProviders() ||
- new Set(models.map((model) => model.routeProvider)).size !== models.length
- ) {
- throw new Error(
- "Release plan must schedule exactly two models on distinct route providers.",
- );
- }
- const expected = releaseCellsFor(models);
+ assertReleaseModels(models, packageVersion);
+ const expected = releaseCellsFor(models, packageVersion);
const primaryCount = expected.filter(
(cell) => cell.schedule === "primary",
).length;
@@ -159,7 +155,7 @@ function assertFrozenReleasePlan(report: ValidatedReport): void {
"Release plan does not contain the canonical primary and environment-reserve grid.",
);
}
- const expectedSeed = releaseRandomizationSeed(models);
+ const expectedSeed = releaseRandomizationSeed(models, packageVersion);
if (
report.plan.planId !== "flow-v2-primary-matrix" ||
report.plan.randomizationSeed !== expectedSeed ||
@@ -187,7 +183,7 @@ function canonicalEvaluator(
return evaluatorIdentity({
sourceCommit: artifact.sourceCommit,
caseCatalog: releaseScenarioCatalog(SCENARIOS),
- policyCatalog: releaseCatalog(),
+ policyCatalog: releaseCatalog(artifact.packageVersion),
graderBundle: releaseGraderBundle(
join(import.meta.dir, ".."),
historical ? artifact.sourceCommit : undefined,
@@ -247,7 +243,10 @@ export function qualifyV2(input: {
readonly expected: ReleaseExpectedProvenance;
readonly canary: CanaryRecord | null;
} {
- const catalog = assertExactReleaseCatalog(input.catalogInput);
+ const catalog = assertExactReleaseCatalog(
+ input.catalogInput,
+ input.artifact.packageVersion,
+ );
const parsed = parseReport(input.reportInput, catalog);
if (!parsed.ok) {
throw new Error(
@@ -256,7 +255,7 @@ export function qualifyV2(input: {
.join("; ")}`,
);
}
- assertFrozenReleasePlan(parsed.value);
+ assertFrozenReleasePlan(parsed.value, input.artifact.packageVersion);
const measuredArtifact = parsed.value.attempts[0]?.artifact;
if (!measuredArtifact || !("sourceCommit" in measuredArtifact)) {
throw new Error("A v2 qualification report requires a Flow artifact.");
@@ -339,7 +338,9 @@ export function decisionRecordFor(input: {
"flow-decision-catalog-v1",
input.catalog,
);
- const policySha256 = RELEASE_POLICY_SHA256;
+ const policySha256 = releasePolicySha256(
+ input.expected.artifact.packageVersion,
+ );
const actorSha256 = canonicalSha256(
"flow-decision-actors-v1",
input.expected.attempts.map((attempt) => ({
@@ -760,7 +761,7 @@ async function main(): Promise {
}
const policy = {
schemaVersion: 1,
- policySha256: RELEASE_POLICY_SHA256,
+ policySha256: releasePolicySha256(artifact.packageVersion),
analysisSha256: RELEASE_ANALYSIS_SHA256,
graderBundle: retainedGraderBundle,
executionGraderBundle,
diff --git a/scripts/release-metadata.ts b/scripts/release-metadata.ts
index 77f2254c..b629da5f 100644
--- a/scripts/release-metadata.ts
+++ b/scripts/release-metadata.ts
@@ -10,9 +10,9 @@ import {
import { QualificationBundleManifestSchema } from "../evals/qualification-bundle.js";
import { regradeQualificationBundle } from "../evals/qualification-regrade.js";
import {
- RELEASE_POLICY_SHA256,
releaseCatalog,
releaseGraderBundle,
+ releasePolicySha256,
releaseScenarioCatalog,
} from "../evals/release-policy.js";
import {
@@ -319,18 +319,18 @@ export function qualificationRecordIssue(
}
const expectedCatalogSha256 = canonicalSha256(
"flow-decision-catalog-v1",
- releaseCatalog(),
+ releaseCatalog(version),
);
if (entry.catalogSha256 !== expectedCatalogSha256) {
return `the qualification catalog digest for ${version} is not current repository policy`;
}
- if (entry.policySha256 !== RELEASE_POLICY_SHA256) {
+ if (entry.policySha256 !== releasePolicySha256(version)) {
return `the qualification policy digest for ${version} is not current repository policy`;
}
const expectedEvaluator = evaluatorIdentity({
sourceCommit: parsedArtifact.data.sourceCommit,
caseCatalog: releaseScenarioCatalog(SCENARIOS),
- policyCatalog: releaseCatalog(),
+ policyCatalog: releaseCatalog(version),
graderBundle: releaseGraderBundle(join(import.meta.dir, "..")),
});
if (
diff --git a/tests/documentation-contract.test.ts b/tests/documentation-contract.test.ts
index 45c51d8d..85c09116 100644
--- a/tests/documentation-contract.test.ts
+++ b/tests/documentation-contract.test.ts
@@ -664,12 +664,6 @@ describe("Flow documentation contract", () => {
expect(release).not.toContain("--clobber");
expect(release).not.toContain("canary-not-enabled");
- // Model-driven evals need credentials and cost real money, so they run on a
- // schedule and never on a pull request. `evals.yml` is the one workflow allowed
- // to invoke them, and the property worth pinning is that no gate a contributor
- // waits on can: the previous rule banned the word outright, which also banned
- // the scheduled multi-model matrix that made "works with Flow" mean anything
- // beyond one provider (docs/adr/0010-declared-canonical-gate.md).
const evals = await readFile(".github/workflows/evals.yml", "utf8");
expect(evals).toContain("bun run eval");
expect(evals).toContain("V2 report:");
@@ -680,6 +674,9 @@ describe("Flow documentation contract", () => {
expect(evals).toContain("github.run_attempt == 1");
expect(evals).toContain("vars.FLOW_EVAL_AUTHORIZE_SCHEDULE == 'true'");
expect(evals).toContain('--max-dispatches "$MAX_DISPATCHES"');
+ expect(evals.indexOf("check-release-models.ts")).toBeLessThan(
+ evals.indexOf("paid-budget.ts authorize"),
+ );
expect(evals).toContain("FLOW_EVAL_AUTHORIZATION:");
expect(evals).not.toMatch(/^on:[\s\S]*?^\s{2}(?:pull_request|push):/m);
diff --git a/tests/eval-release-sampling.test.ts b/tests/eval-release-sampling.test.ts
index 4744431c..9c81b22f 100644
--- a/tests/eval-release-sampling.test.ts
+++ b/tests/eval-release-sampling.test.ts
@@ -2,9 +2,11 @@ import { describe, expect, test } from "bun:test";
import {
assertExactReleaseCatalog,
assertReleaseHost,
+ assertReleaseModels,
releaseAttemptsFor,
releaseCaseIds,
releaseCatalog,
+ releasePolicySha256,
} from "../evals/release-policy.js";
import {
campaignPlanFor,
@@ -17,6 +19,73 @@ import {
import { SCENARIOS } from "../evals/scenarios.js";
describe("release eval sampling", () => {
+ test("pins the 9.1.0 OpenAI-only grid without weakening later releases", () => {
+ const model = {
+ routeProvider: "openai",
+ gateway: null,
+ family: "gpt-6",
+ model: "gpt-6-sol",
+ revision: null,
+ };
+ const catalog = releaseCatalog("9.1.0");
+ expect(catalog.every((row) => row.minProviders === 1)).toBe(true);
+ expect(
+ catalog.map((row) => row.minScoredAttempts).reduce((a, b) => a + b),
+ ).toBe(38);
+ const plan = campaignPlanFor({
+ models: ["openai/gpt-6-sol"],
+ scenarios: releaseScenarios(),
+ sampling: { kind: "release" },
+ opencodeVersion: "1.18.31",
+ });
+ expect(plan.stoppingRule.count).toBe(38);
+ expect(plan.budget.maxAttempts).toBe(46);
+ expect(plan.abortPolicy.maxReplacementBlocks).toBe(8);
+ expect(() =>
+ assertReleaseModels(
+ [{ ...model, routeProvider: "xai", model: "grok-4.6" }],
+ "9.1.0",
+ ),
+ ).toThrow("openai/gpt-6-sol");
+ for (const forged of [
+ { ...model, gateway: "gateway" },
+ { ...model, family: "other-family" },
+ { ...model, revision: "preview" },
+ { ...model, variant: "fast" },
+ ]) {
+ expect(() => assertReleaseModels([forged], "9.1.0")).toThrow(
+ "canonical direct-route identity",
+ );
+ }
+ expect(() => assertReleaseModels([model], "9.1.1")).toThrow(
+ "exactly 2 models",
+ );
+ expect(releasePolicySha256("9.1.0")).not.toBe(releasePolicySha256("9.1.1"));
+ expect(() => assertExactReleaseCatalog(catalog, "9.1.1")).toThrow();
+ });
+
+ test("checks configured CI release models before authorization", async () => {
+ for (const [configured, expectedCode] of [
+ ["openai/gpt-6-sol", 0],
+ ["openai/gpt-6-sol,xai/grok-4.6", 1],
+ ["xai/grok-4.6", 1],
+ ] as const) {
+ const child = Bun.spawn(
+ ["bun", "run", "scripts/check-release-models.ts"],
+ {
+ cwd: new URL("..", import.meta.url).pathname,
+ env: { ...process.env, FLOW_EVAL_MODEL: configured },
+ stderr: "pipe",
+ },
+ );
+ const [code, stderr] = await Promise.all([
+ child.exited,
+ new Response(child.stderr).text(),
+ ]);
+ expect(code).toBe(expectedCode);
+ if (expectedCode !== 0) expect(stderr).toContain("openai/gpt-6-sol");
+ }
+ });
test("gives 90 percent cases ten attempts and 100 percent cases three", () => {
for (const policy of releaseCatalog()) {
expect(releaseAttemptsFor(policy.caseId)).toBe(
@@ -56,7 +125,7 @@ describe("release eval sampling", () => {
campaignPlanFor({
models: ["xai/grok-4.6"],
scenarios,
- sampling: { kind: "release" },
+ sampling: { kind: "release", packageVersion: "8.1.2" },
opencodeVersion: "1.18.6",
}),
).toThrow("Release scenarios do not match repository release policy");
@@ -89,7 +158,7 @@ describe("release eval sampling", () => {
const plan = campaignPlanFor({
models: ["xai/grok-4.6", "openai/gpt-5.6-sol"],
scenarios,
- sampling: { kind: "release" },
+ sampling: { kind: "release", packageVersion: "8.1.2" },
opencodeVersion: "1.18.6",
});
expect(plan.cells).toHaveLength(92);
@@ -105,9 +174,9 @@ describe("release eval sampling", () => {
retry: "environment-only",
maxReplacementBlocks: 16,
});
- expect(caseCatalogFor(scenarios, { kind: "release" })).toEqual(
- releaseCatalog(),
- );
+ expect(
+ caseCatalogFor(scenarios, { kind: "release", packageVersion: "8.1.2" }),
+ ).toEqual(releaseCatalog());
expect(
plan.cells.filter((cell) => cell.caseId === "skipped-case-named-binding"),
).toHaveLength(8);
@@ -196,7 +265,7 @@ describe("release eval sampling", () => {
}
});
- test("rejects every release model grid except two distinct providers", async () => {
+ test("rejects every 9.1.0 release grid except GPT-6 Sol", async () => {
for (const models of [
["xai/a"],
["xai/a", "xai/b"],
@@ -219,7 +288,7 @@ describe("release eval sampling", () => {
]);
expect(exitCode).toBe(2);
expect(stderr).toContain(
- "--release requires exactly 2 models on distinct route providers",
+ "Release 9.1.0 requires exactly openai/gpt-6-sol",
);
}
});
@@ -231,9 +300,7 @@ describe("release eval sampling", () => {
"run",
"evals/run.ts",
"--model",
- "xai/a",
- "--model",
- "openai/b",
+ "openai/gpt-6-sol",
"--release",
"--concurrency",
"2",
diff --git a/tests/eval-reserve-cancellation.test.ts b/tests/eval-reserve-cancellation.test.ts
index d6e20d5b..f4e23eb2 100644
--- a/tests/eval-reserve-cancellation.test.ts
+++ b/tests/eval-reserve-cancellation.test.ts
@@ -121,8 +121,8 @@ describe("graceful-eval-stop.R10-05: real release runner reserve cancellation",
expect(result.signal).toBeNull();
expect(result.events).not.toContain("unexpected-network");
expect(result.events).toContain("returned");
- const retained = mode === "handoff" ? 76 : 77;
- const started = mode === "handoff" ? 76 : 78;
+ const retained = mode === "handoff" ? 38 : 39;
+ const started = mode === "handoff" ? 38 : 40;
expect(result.events).toContain(`durable:${retained}`);
expect(
result.events.filter((event) => event.startsWith("signal:")),
@@ -153,11 +153,11 @@ describe("graceful-eval-stop.R10-05: real release runner reserve cancellation",
expect(reports).toHaveLength(1);
const directory = join(results, required(reports[0]));
expect(await readJson(join(directory, "catalog.json"))).toEqual(
- releaseCatalog(),
+ releaseCatalog("9.1.0"),
);
const parsed = parseReport(
await readJson(join(directory, "report.json")),
- releaseCatalog(),
+ releaseCatalog("9.1.0"),
);
if (!parsed.ok) throw new Error(JSON.stringify(parsed.issues));
const report = parsed.value;
@@ -167,8 +167,8 @@ describe("graceful-eval-stop.R10-05: real release runner reserve cancellation",
const reserves = report.plan.cells.filter(
(cell) => cell.schedule === "environment-reserve",
);
- expect(primary).toHaveLength(76);
- expect(reserves).toHaveLength(16);
+ expect(primary).toHaveLength(38);
+ expect(reserves).toHaveLength(8);
expect(report.attempts.map((attempt) => attempt.cellId)).toEqual([
...primary.map((cell) => cell.cellId),
...(mode === "reserve" ? [required(reserves[0]).cellId] : []),
@@ -253,7 +253,7 @@ describe("graceful-eval-stop.R10-05: real release runner reserve cancellation",
throw new Error("Expected packed artifact identity.");
const decision = deriveReleaseDecision({
report,
- catalog: releaseCatalog(),
+ catalog: releaseCatalog("9.1.0"),
expected: {
kind: "release",
artifact: first.artifact,
diff --git a/tests/fixtures/eval-reserve-cancellation-child.ts b/tests/fixtures/eval-reserve-cancellation-child.ts
index 7e978a55..b6ca37c1 100644
--- a/tests/fixtures/eval-reserve-cancellation-child.ts
+++ b/tests/fixtures/eval-reserve-cancellation-child.ts
@@ -12,7 +12,7 @@ const [root, mode] = process.argv.slice(2);
if (!root || (mode !== "handoff" && mode !== "reserve"))
throw new Error("Expected temporary root and handoff/reserve mode.");
const repositoryRoot = root;
-const models = ["fixture-a/model", "fixture-b/model"] as const;
+const models = ["openai/gpt-6-sol"] as const;
const event = (name: string) =>
process.stdout.write(`\n@@eval-reserve:${name}\n`);
process.stdin.resume(); // Keep the child alive until the parent releases cleanup.
@@ -99,15 +99,12 @@ class FakeReleaseHost {
return `fixture-session-${this.attempt}`;
}
async runCommand(): Promise<"quiet"> {
- if (mode === "reserve" && this.attempt === 78)
- await stopHere(this.signal, 77);
+ if (mode === "reserve" && this.attempt === 40)
+ await stopHere(this.signal, 39);
return "quiet";
}
async outcome(sessionIds: string[]): Promise {
event(`outcome:${this.attempt}`);
- // One eligible gap in each of the first three canonical case/provider
- // strata: three reserves really activate, leaving a third queued behind
- // the interrupted second reserve. No outcome/ledger retry flags are patched.
const gap = [1, 4, 7].includes(this.attempt);
const observation = gap
? {
@@ -146,8 +143,8 @@ class FakeReleaseHost {
}
async stop() {
event(`stop:${this.attempt}`);
- if (mode === "handoff" && this.attempt === 76)
- await stopHere(this.signal, 76).catch((error: unknown) => {
+ if (mode === "handoff" && this.attempt === 38)
+ await stopHere(this.signal, 38).catch((error: unknown) => {
if (error !== this.signal.reason) throw error;
});
if (this.signal.aborted) await cleanupGate();
@@ -204,15 +201,7 @@ try {
try {
const code = await runCampaign(
signal,
- [
- "--release",
- "--model",
- models[0],
- "--model",
- models[1],
- "--concurrency",
- "1",
- ],
+ ["--release", "--model", models[0], "--concurrency", "1"],
repositoryRoot,
beginFinalization,
);
diff --git a/tests/patch-release.test.ts b/tests/patch-release.test.ts
index 94ec755a..867edbb2 100644
--- a/tests/patch-release.test.ts
+++ b/tests/patch-release.test.ts
@@ -142,8 +142,21 @@ async function fixture() {
return {
bundleSha256: hash("bundle"),
summary: {
+ schemaVersion: 1,
+ packageVersion: "8.3.0",
+ reportId: "baseline-report",
+ verdict: "VERIFIED",
+ bundleSha256: hash("bundle"),
+ artifact: baselineArtifact,
canarySha256: hash("canary"),
- totals: { passed: 76, scored: 76 },
+ totals: { scheduled: 76, passed: 76, scored: 76 },
+ providers: ["openai", "xai"].map((provider) => ({
+ provider,
+ scheduled: 38,
+ scored: 38,
+ passed: 38,
+ passRate: 1,
+ })),
} as ReleaseEvidenceSummary,
};
};
@@ -171,6 +184,26 @@ test("eligible patch verifies baseline and discloses prior-only evidence", async
expect(result.notes).toContain("not measurements of this candidate");
expect(result.bundleSha256).toBe(hash(await readFile(f.input.path, "utf8")));
});
+test("refuses a one-provider release as a patch baseline", async () => {
+ const f = await fixture();
+ await expect(
+ assertPatchReleaseEvidence(f.input, async () => ({
+ bundleSha256: hash("bundle"),
+ summary: {
+ ...(await f.verify()).summary,
+ providers: [
+ {
+ provider: "openai",
+ scheduled: 38,
+ scored: 38,
+ passed: 38,
+ passRate: 1,
+ },
+ ],
+ },
+ })),
+ ).rejects.toThrow("two-provider baseline");
+});
test("reads the baseline seal as retained evidence, not as a fresh measurement", async () => {
const f = await fixture();
let freshness: string | undefined;
diff --git a/tests/qualification-cli.test.ts b/tests/qualification-cli.test.ts
index 98217a89..ecfdeb8d 100644
--- a/tests/qualification-cli.test.ts
+++ b/tests/qualification-cli.test.ts
@@ -224,7 +224,7 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a
tarballPath: artifactPath,
});
const scenarios = releaseScenarios();
- const models = ["fixture-alpha/model-a", "fixture-beta/model-b"];
+ const models = ["openai/gpt-6-sol"];
const plan = campaignPlanFor({
models,
scenarios,
@@ -234,16 +234,16 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a
const evaluator = evaluatorIdentity({
sourceCommit: artifact.sourceCommit,
caseCatalog: releaseScenarioCatalog(scenarios),
- policyCatalog: releaseCatalog(),
+ policyCatalog: releaseCatalog(artifact.packageVersion),
graderBundle: releaseGraderBundle(repositoryRoot),
});
const campaignDirectory = join(temporary, "campaign");
const store = createReportStore({
directory: campaignDirectory,
- catalog: releaseCatalog(),
+ catalog: releaseCatalog(artifact.packageVersion),
});
await store.initialize(plan);
- await store.writeCatalog(releaseCatalog());
+ await store.writeCatalog(releaseCatalog(artifact.packageVersion));
await store.writeArtifact(artifactPath);
const replayedByScenario = new Map<
@@ -413,7 +413,7 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a
completion,
allocationCommitmentSha256: null,
});
- expect(report.attempts).toHaveLength(77);
+ expect(report.attempts).toHaveLength(39);
const preparedDirectory = join(temporary, "prepared-canary");
await mkdir(preparedDirectory, { recursive: true });
@@ -521,8 +521,8 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a
const transcripts = bundle.files.filter(
({ ref }) => ref.role === "transcript",
);
- expect(attempts).toHaveLength(77);
- expect(transcripts).toHaveLength(77);
+ expect(attempts).toHaveLength(39);
+ expect(transcripts).toHaveLength(39);
expect(attempts.map(({ ref }) => ref.id).sort()).toEqual(
transcripts.map(({ ref }) => ref.id).sort(),
);
@@ -569,7 +569,7 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a
canarySha256: canary.record.recordSha256,
artifact,
});
- expect(releaseAuthority.summary.providers).toHaveLength(2);
+ expect(releaseAuthority.summary.providers).toHaveLength(1);
const notesPath = join(temporary, "release-notes.md");
const metadata = Bun.spawn(
[
diff --git a/tests/release-qualification.test.ts b/tests/release-qualification.test.ts
index 1d74f34d..782aa61a 100644
--- a/tests/release-qualification.test.ts
+++ b/tests/release-qualification.test.ts
@@ -44,7 +44,7 @@ function releaseReport(stopped = false) {
const plan = campaignPlanFor({
models: MODELS,
scenarios,
- sampling: { kind: "release" },
+ sampling: { kind: "release", packageVersion: ARTIFACT.packageVersion },
opencodeVersion: "1.18.31",
});
const evaluator = evaluatorIdentity({