Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
67 commits
Select commit Hold shift + click to select a range
f4f69aa
docs(lab): freeze CL-08 automation contract
Wibias Aug 11, 2026
bb2c1c0
feat(lab): add bounded automation orchestration
Wibias Aug 11, 2026
6b3d944
feat(lab): wire trusted live-route automation
Wibias Aug 11, 2026
3d6b7f2
test(lab): harden automation authority and lifecycle
Wibias Aug 11, 2026
e4ba266
test(lab): capture CL-08 review regressions
Wibias Aug 11, 2026
a866f1d
fix(lab): close CL-08 automation policy schema
Wibias Aug 11, 2026
05e052f
fix(lab): harden automation state persistence
Wibias Aug 11, 2026
baaedf0
fix(lab): make automation budgets rolling and attempt-safe
Wibias Aug 11, 2026
16286a6
fix(lab): bound CL-08 queue and dispatch selection
Wibias Aug 11, 2026
1b1eb79
fix(lab): scope automation freshness to exact evidence identity
Wibias Aug 11, 2026
6713465
fix(lab): revalidate CL-08 dispatch contracts
Wibias Aug 11, 2026
5227aa0
fix(lab): serialize automation scheduling state
Wibias Aug 11, 2026
60921ef
fix(lab): reconcile automation queue on policy changes
Wibias Aug 11, 2026
b84fdd2
fix(lab): carry dispatch identity enforcement context
Wibias Aug 11, 2026
02aa4f2
fix(lab): enforce queued identity without breaking low-level test seams
Wibias Aug 11, 2026
1a14ec2
fix(lab): close scheduler lifecycle and stale-plan races
Wibias Aug 11, 2026
17fb662
fix(lab): ignore non-authoritative automation update fields
Wibias Aug 11, 2026
3110fd1
fix(lab): rebuild missing projection before scheduled planning
Wibias Aug 11, 2026
247ec04
test(lab): use canonical queued live runs
Wibias Aug 11, 2026
ec51713
fix(lab): complete CL-08 CLI execution controls
Wibias Aug 11, 2026
6706ac6
fix(lab): harden automation review bounds
Wibias Aug 11, 2026
a697421
fix(lab): reject zero concurrency limits
Wibias Aug 11, 2026
ab7e689
fix(lab): bound cooldown persistence
Wibias Aug 11, 2026
b8d9f87
fix(lab): retain rolling budget evidence
Wibias Aug 11, 2026
2eea0a6
fix(lab): reject stale run cursors
Wibias Aug 11, 2026
7ae55c6
fix(lab): harden OAuth live-route dispatch
Wibias Aug 11, 2026
68c938f
fix(lab): fence automation state ownership
Wibias Aug 11, 2026
32878f5
fix(lab): isolate scheduler ownership and cancellation
Wibias Aug 11, 2026
271eabe
fix(lab): make freshness exact and bounded
Wibias Aug 11, 2026
4f12978
fix(lab): harden automation management API
Wibias Aug 11, 2026
e250560
fix(lab): preserve CLI automation intent
Wibias Aug 11, 2026
f355104
test(lab): cover CodeRabbit automation regressions
Wibias Aug 11, 2026
1ae9b4a
fix(lab): prune cooldowns during recovery
Wibias Aug 11, 2026
c441591
test(lab): make rolling budget regression deterministic
Wibias Aug 11, 2026
7fbc18f
docs(lab): align CL-08 headings and budget reason
Wibias Aug 11, 2026
e04dce6
test(lab): strengthen automation authority regressions
Wibias Aug 11, 2026
36bcc80
test(lab): cover automation management pagination and cancel routes
Wibias Aug 11, 2026
f13489a
test(lab): assert automation HTTP error codes
Wibias Aug 11, 2026
4c4f8cf
fix(lab): guarantee cancellation backoff
Wibias Aug 11, 2026
2390c20
test(lab): prove cancellation backoff with zero failure cooldown
Wibias Aug 11, 2026
27e7394
fix(lab): reserve cooldown capacity before dispatch
Wibias Aug 11, 2026
7c09d51
fix(lab): fence dispatch deps and cooldown reservations
Wibias Aug 11, 2026
568f670
fix(lab): publish state locks atomically
Wibias Aug 11, 2026
7cdb374
fix(lab): add bounded exact freshness query
Wibias Aug 11, 2026
b7990ae
test(lab): cover final CL-08 review blockers
Wibias Aug 11, 2026
5cad25a
fix(lab): close final CL-08 review blockers
Wibias Aug 11, 2026
e991457
test(lab): prove bounded exact freshness lookup
Wibias Aug 11, 2026
e2bf73e
test(lab): initialize persisted-state fixture dirs
Wibias Aug 11, 2026
253a109
test(lab): reproduce cancellation history persistence overflow
Wibias Aug 11, 2026
f7598f0
fix(lab): fail closed when cooldown persistence saturates
Wibias Aug 11, 2026
7607cf6
fix(lab): keep persisted run history within hard ceiling
Wibias Aug 11, 2026
82f7e57
test(lab): carry parent-owned fabric isolation budgets to child
Wibias Aug 11, 2026
382b72d
test(lab): add guarded fabric isolation limit override
Wibias Aug 11, 2026
a806664
test(lab): scale synthetic producer timing to parent budgets
Wibias Aug 11, 2026
b2a9093
test(lab): shorten CL-07 isolation timing budgets
Wibias Aug 11, 2026
9e5b6e3
test(lab): track current projection schema version
Wibias Aug 11, 2026
e533234
test(lab): assert fail-closed cooldown saturation
Wibias Aug 11, 2026
dd04b8a
test(lab): reproduce maintainer lifecycle blockers
Wibias Aug 11, 2026
6220ea7
fix(lab): scope runtime resources to server owners
Wibias Aug 11, 2026
1998e97
fix(lab): release server-owned runtime resources
Wibias Aug 11, 2026
07a15cc
fix(lab): commit automation config atomically
Wibias Aug 11, 2026
40b7539
fix(lab): publish policy and routes as one config
Wibias Aug 11, 2026
ba393fd
fix(lab): bind scheduler authority to runtime owner
Wibias Aug 11, 2026
8a37c28
test(lab): exercise server-owned runtime cleanup
Wibias Aug 11, 2026
23c8f9b
fix(lab): keep CLI policy writes on committed config
Wibias Aug 11, 2026
1492c29
chore(lab): remove CL-08 plan trailing whitespace
Wibias Aug 11, 2026
bfaad5d
test(lab): read committed automation config generation
Wibias Aug 11, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1,019 changes: 1,019 additions & 0 deletions devlog/_plan/260807_compatibility_lab/008_cl08_automation.md

Large diffs are not rendered by default.

213 changes: 148 additions & 65 deletions src/cli/lab.ts
Original file line number Diff line number Diff line change
@@ -1,9 +1,10 @@
/**
* `ocx lab` — read-only Compatibility Lab inspection (CL-04).
* `ocx lab` — Compatibility Lab inspection and explicit CL-08 automation controls.
*
* Local SQLite projection reads; no daemon, network, probes, or rebuilds.
* Read commands use the local SQLite projection. Automation mutations and manual runs are
* explicit operator actions; read commands never start probes or scheduler ticks.
*/
import { getConfigDir } from "../config";
import { getConfigDir, readConfigDiagnostics } from "../config";
import {
ARTIFACT_CLASSES,
EVIDENCE_LAYERS,
Expand Down Expand Up @@ -43,6 +44,23 @@ import {
takeIntegerOption,
takeOption,
} from "./runtime-api";
import {
buildLabAutomationStatus,
enqueueManualLabRun,
reconcileLabAutomationQueue,
setLabAutomationDispatchDeps,
startLabAutomationScheduler,
stopLabAutomationScheduler,
} from "../lab/automation/orchestrator";
import { loadLabAutomationState } from "../lab/automation/persistence";
import {
loadLabAutomationConfig,
saveLabAutomationPolicyConfig,
} from "../lab/automation/config-persistence";
import { planManualLabRun } from "../lab/automation/planner";
import { listLabAutomationRuns } from "../lab/automation/runs-query";
import { LabAutomationError, type LabAutomationLayer } from "../lab/automation/types";
import { createProductionLabRouteExecutor } from "../lib/lab-live-route-production";

const USAGE = `Usage:
ocx lab status [--json]
Expand All @@ -54,7 +72,12 @@ const USAGE = `Usage:
ocx lab event <eventId> [--json]
ocx lab artifacts [--status <s>] [--artifact-class <c>] [--limit <n>] [--cursor <c>] [--json]
ocx lab artifact <digest> [--json]
ocx lab catalog [--layer <layer>] [--suite <id>] [--json]`;
ocx lab catalog [--layer <layer>] [--suite <id>] [--json]
ocx lab automation status [--json]
ocx lab automation enable [--protocol] [--live] [--json]
ocx lab automation disable [--json]
ocx lab automation runs [--limit <n>] [--cursor <c>] [--json]
ocx lab run --layer <layer> --scenario <id> [--provider <name>] [--model <id>] [--json]`;

const ARTIFACT_STATUSES = ["present", "corrupt", "purged_unavailable"] as const;
type ArtifactStatus = (typeof ARTIFACT_STATUSES)[number];
Expand All @@ -74,6 +97,7 @@ function labErrorMessage(err: unknown): string {
if (err instanceof LabProjectionUnavailableError) return "lab projection is not available";
if (err instanceof LabProjectionIncompatibleError) return "lab projection schema or spec version is incompatible";
if (err instanceof InvalidCursorError) return "invalid cursor";
if (err instanceof LabAutomationError && err.code === "invalid_cursor") return "invalid cursor";
return "lab read failed";
}

Expand Down Expand Up @@ -163,6 +187,28 @@ function catalogLines(scenarios: ReturnType<typeof queryLabCatalogEntries>): str
return lines.length > 0 ? lines : ["No catalog scenarios"];
}

function automationStatusLines(status: ReturnType<typeof buildLabAutomationStatus>): string[] {
return [
`Automation enabled: ${status.policy.enabled}`,
`Scheduler intent: ${status.policy.enabled ? "enabled" : "disabled"}`,
`Layers: protocol=${status.policy.layers.protocolConformance} live=${status.policy.layers.liveRouteCompatibility} task=${status.policy.layers.taskEffectiveness}`,
`Queued: ${status.counters.queued} | Running: ${status.counters.running}`,
`Run budget remaining: ${status.counters.remainingRunBudget} | Live request budget: ${status.counters.remainingLiveRequestBudget}`,
];
}

function automationCliStatus(status: ReturnType<typeof buildLabAutomationStatus>) {
const { schedulerRunning: _processLocalSchedulerState, ...stable } = status;
return { ...stable, schedulerIntentEnabled: status.policy.enabled };
}

function runListLines(page: ReturnType<typeof listLabAutomationRuns>): string[] {
const lines = page.items.map((row) =>
`${row.runId} ${row.state} ${row.evidenceLayer} ${row.scenarioId} trigger=${row.trigger}`,
);
return lines.length > 0 ? lines : ["No automation runs"];
}

export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): Promise<number> {
return runCliAction(async () => {
const configDir = deps.configDir ?? getConfigDir();
Expand Down Expand Up @@ -199,14 +245,7 @@ export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): P
const limit = takeIntegerOption(rest, "--limit", { min: 1 });
const cursor = takeOption(rest, "--cursor");
rejectArgs(rest, USAGE);
const page = queryLabVerdicts({
subjectId,
layer,
suiteId,
verdict,
from,
to,
}, cursor, limit, configDir);
const page = queryLabVerdicts({ subjectId, layer, suiteId, verdict, from, to }, cursor, limit, configDir);
printData(page, wantsJson, verdictLines(page));
return;
}
Expand All @@ -231,52 +270,23 @@ export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): P
}
case "observations": {
const subjectId = takeOption(rest, "--subject");
const layer = takeEnumOption<EvidenceLayer>(
rest,
"--layer",
EVIDENCE_LAYERS,
"--layer must be a supported evidence layer",
);
const layer = takeEnumOption<EvidenceLayer>(rest, "--layer", EVIDENCE_LAYERS, "--layer must be a supported evidence layer");
const suiteId = takeOption(rest, "--suite");
const scenarioId = takeOption(rest, "--scenario");
const outcome = takeEnumOption<ObservationOutcome>(
rest,
"--outcome",
OUTCOMES,
"--outcome must be a supported observation outcome",
);
const executionMode = takeEnumOption<ExecutionMode>(
rest,
"--execution-mode",
EXECUTION_MODES,
"--execution-mode must be a supported execution mode",
);
const outcome = takeEnumOption<ObservationOutcome>(rest, "--outcome", OUTCOMES, "--outcome must be a supported observation outcome");
const executionMode = takeEnumOption<ExecutionMode>(rest, "--execution-mode", EXECUTION_MODES, "--execution-mode must be a supported execution mode");
const from = takeIntegerOption(rest, "--from", { min: 0 });
const to = takeIntegerOption(rest, "--to", { min: 0 });
assertRange(from, to);
const limit = takeIntegerOption(rest, "--limit", { min: 1 });
const cursor = takeOption(rest, "--cursor");
rejectArgs(rest, USAGE);
const page = queryLabObservations({
subjectId,
layer,
suiteId,
scenarioId,
outcome,
executionMode,
from,
to,
}, cursor, limit, configDir);
const page = queryLabObservations({ subjectId, layer, suiteId, scenarioId, outcome, executionMode, from, to }, cursor, limit, configDir);
printData(page, wantsJson, observationLines(page));
return;
}
case "events": {
const eventKind = takeEnumOption<LabEventKind>(
rest,
"--event-kind",
EVENT_KINDS,
"--event-kind must be a supported lab event kind",
);
const eventKind = takeEnumOption<LabEventKind>(rest, "--event-kind", EVENT_KINDS, "--event-kind must be a supported lab event kind");
const subjectId = takeOption(rest, "--subject");
const from = takeIntegerOption(rest, "--from", { min: 0 });
const to = takeIntegerOption(rest, "--to", { min: 0 });
Expand Down Expand Up @@ -309,18 +319,8 @@ export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): P
return;
}
case "artifacts": {
const status = takeEnumOption<ArtifactStatus>(
rest,
"--status",
ARTIFACT_STATUSES,
"--status must be present, corrupt, or purged_unavailable",
);
const artifactClass = takeEnumOption<ArtifactClass>(
rest,
"--artifact-class",
ARTIFACT_CLASSES,
"--artifact-class must be a supported artifact class",
);
const status = takeEnumOption<ArtifactStatus>(rest, "--status", ARTIFACT_STATUSES, "--status must be present, corrupt, or purged_unavailable");
const artifactClass = takeEnumOption<ArtifactClass>(rest, "--artifact-class", ARTIFACT_CLASSES, "--artifact-class must be a supported artifact class");
const limit = takeIntegerOption(rest, "--limit", { min: 1 });
const cursor = takeOption(rest, "--cursor");
rejectArgs(rest, USAGE);
Expand All @@ -339,18 +339,101 @@ export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): P
return;
}
case "catalog": {
const layer = takeEnumOption<EvidenceLayer>(
rest,
"--layer",
EVIDENCE_LAYERS,
"--layer must be a supported evidence layer",
);
const layer = takeEnumOption<EvidenceLayer>(rest, "--layer", EVIDENCE_LAYERS, "--layer must be a supported evidence layer");
const suiteId = takeOption(rest, "--suite");
rejectArgs(rest, USAGE);
const scenarios = queryLabCatalogEntries({ layer, suiteId });
printData({ scenarios }, wantsJson, catalogLines(scenarios));
return;
}
case "automation": {
const [automationSub = "status", ...automationRest] = rest;
switch (automationSub) {
case "status": {
rejectArgs(automationRest, USAGE);
const status = buildLabAutomationStatus(configDir);
printData(automationCliStatus(status), wantsJson, automationStatusLines(status));
return;
}
case "enable": {
const enableProtocol = takeFlag(automationRest, "--protocol");
const enableLive = takeFlag(automationRest, "--live");
rejectArgs(automationRest, USAGE);
const selectionExplicit = enableProtocol || enableLive;
const current = loadLabAutomationConfig(configDir).policy;
const policy = {
...current,
enabled: true,
layers: {
protocolConformance: selectionExplicit ? enableProtocol : current.layers.protocolConformance,
liveRouteCompatibility: selectionExplicit ? enableLive : current.layers.liveRouteCompatibility,
taskEffectiveness: false,
},
taskEffectivenessBackgroundEnabled: false,
};
saveLabAutomationPolicyConfig(policy, configDir);
reconcileLabAutomationQueue(configDir);
startLabAutomationScheduler(configDir);
const status = buildLabAutomationStatus(configDir);
printData(automationCliStatus(status), wantsJson, automationStatusLines(status));
return;
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.
case "disable": {
rejectArgs(automationRest, USAGE);
const policy = { ...loadLabAutomationConfig(configDir).policy, enabled: false };
saveLabAutomationPolicyConfig(policy, configDir);
reconcileLabAutomationQueue(configDir);
stopLabAutomationScheduler(configDir);
const status = buildLabAutomationStatus(configDir);
printData(automationCliStatus(status), wantsJson, automationStatusLines(status));
return;
}
case "runs": {
const limit = takeIntegerOption(automationRest, "--limit", { min: 1 });
const cursor = takeOption(automationRest, "--cursor");
rejectArgs(automationRest, USAGE);
const page = listLabAutomationRuns(loadLabAutomationState(configDir), limit ?? 50, cursor);
printData(page, wantsJson, runListLines(page));
return;
}
default:
throw new CliUsageError(`unknown automation subcommand: ${automationSub}`, USAGE);
}
}
case "run": {
const layer = takeEnumOption<LabAutomationLayer>(
rest,
"--layer",
["protocol_conformance", "live_route_compatibility", "task_effectiveness"] as const,
"--layer must be protocol_conformance, live_route_compatibility, or task_effectiveness",
);
const scenarioId = takeOption(rest, "--scenario");
const providerName = takeOption(rest, "--provider");
const modelId = takeOption(rest, "--model");
rejectArgs(rest, USAGE);
if (!layer || !scenarioId) throw new CliUsageError("--layer and --scenario are required", USAGE);
const configSnapshot = readConfigDiagnostics().config;
if (layer === "live_route_compatibility") {
const loadConfig = () => configSnapshot;
setLabAutomationDispatchDeps({
configDir,
loadConfig,
routeExecutor: createProductionLabRouteExecutor({ configDir, loadConfig }),
});
}
const planned = planManualLabRun({
evidenceLayer: layer,
scenarioId,
providerName,
modelId,
config: configSnapshot,
configDir,
});
const record = await enqueueManualLabRun(planned, configDir);
if (!record) throw new CliUsageError("manual run enqueue failed", USAGE);
printData({ run: record }, wantsJson, [`Manual run ${record.runId} -> ${record.state}`]);
return;
}
default:
throw new CliUsageError(`unknown lab subcommand: ${sub}`, USAGE);
}
Expand All @@ -364,4 +447,4 @@ export async function handleLabCommand(argv: string[], deps: LabCliDeps = {}): P
});
}

export const LAB_USAGE = USAGE;
export const LAB_USAGE = USAGE;
78 changes: 78 additions & 0 deletions src/lab/automation/budgets.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
import { LAB_AUTOMATION_HARD_MAX } from "./constants";
import type { LabAutomationPolicyV1, LabAutomationStateV1 } from "./types";

export function rollBudgetWindow(state: LabAutomationStateV1, now: number): LabAutomationStateV1 {
if (now - state.budgetWindowStartedAt < LAB_AUTOMATION_HARD_MAX.budgetWindowMs) return state;
return {
...state,
budgetWindowStartedAt: now,
runsThisHour: 0,
liveRequestsThisHour: 0,
};
}

function rollingStartedRuns(state: LabAutomationStateV1, now: number): number {
const cutoff = now - LAB_AUTOMATION_HARD_MAX.budgetWindowMs;
return state.runs.filter((run) => typeof run.startedAt === "number" && run.startedAt > cutoff && run.startedAt <= now).length;
}

function rollingStartedLiveRuns(state: LabAutomationStateV1, now: number): number {
const cutoff = now - LAB_AUTOMATION_HARD_MAX.budgetWindowMs;
return state.runs.filter((run) =>
run.evidenceLayer === "live_route_compatibility"
&& typeof run.startedAt === "number"
&& run.startedAt > cutoff
&& run.startedAt <= now
).length;
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.

function legacyCounterActive(state: LabAutomationStateV1, now: number): boolean {
return now - state.budgetWindowStartedAt < LAB_AUTOMATION_HARD_MAX.budgetWindowMs;
}

export function runBudgetRemaining(
policy: LabAutomationPolicyV1,
state: LabAutomationStateV1,
now = Date.now(),
): number {
const rolling = rollingStartedRuns(state, now);
const legacy = legacyCounterActive(state, now) ? state.runsThisHour : 0;
return Math.max(0, policy.maxRunsPerHour - Math.max(rolling, legacy));
}

export function liveRequestBudgetRemaining(
policy: LabAutomationPolicyV1,
state: LabAutomationStateV1,
now = Date.now(),
): number {
const rolling = rollingStartedLiveRuns(state, now);
const legacy = legacyCounterActive(state, now) ? state.liveRequestsThisHour : 0;
return Math.max(0, policy.maxLiveRequestsPerHour - Math.max(rolling, legacy));
}

export function isRunBudgetExhausted(
policy: LabAutomationPolicyV1,
state: LabAutomationStateV1,
now = Date.now(),
): boolean {
return runBudgetRemaining(policy, state, now) <= 0;
}

export function isLiveRequestBudgetExhausted(
policy: LabAutomationPolicyV1,
state: LabAutomationStateV1,
now = Date.now(),
): boolean {
return liveRequestBudgetRemaining(policy, state, now) <= 0;
}

/** Legacy persisted counters remain for schema compatibility; startedAt records are rolling authority. */
export function recordRunBudgetUse(state: LabAutomationStateV1, liveRequest: boolean): LabAutomationStateV1 {
return {
...state,
runsThisHour: Math.min(LAB_AUTOMATION_HARD_MAX.maxRunsPerHour, state.runsThisHour + 1),
liveRequestsThisHour: liveRequest
? Math.min(LAB_AUTOMATION_HARD_MAX.maxLiveRequestsPerHour, state.liveRequestsThisHour + 1)
: state.liveRequestsThisHour,
};
}
Loading
Loading