diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 0741ad1c..d6585099 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -159,61 +159,34 @@ jobs: # zero matched tests is a FAILURE, and affected components with no recorded # fixture are named in a ::warning (a visible coverage gap, never a quiet # pass). PRs touching nothing under containers/ exit in ~30s. - affected-replay: - name: Replay affected (changed components) + affected: + name: Test affected (changed components) runs-on: ubuntu-latest if: github.event_name == 'pull_request' - timeout-minutes: 40 + timeout-minutes: 60 steps: - uses: actions/checkout@v6.0.3 with: fetch-depth: 0 # the base SHA must be resolvable for REF= fleet-hash - - name: Select fixtures for affected components + - name: Select affected components id: sel env: BASE: ${{ github.event.pull_request.base.sha }} - run: | - set -eo pipefail - changed=$(comm -13 \ - <(REF="$BASE" bash containers/scripts/fleet-hash.sh | LC_ALL=C sort) \ - <(bash containers/scripts/fleet-hash.sh | LC_ALL=C sort) | cut -f1) - benches=$(sed -n 's/^benchmark-//p' <<< "$changed") - agents=$(sed -n 's/^agent-//p' <<< "$changed") - echo "affected: $(tr '\n' ' ' <<< "$changed")" - FIX=tests/run/replay/fixtures - picks=""; uncovered="" - pick() { # $1 = fixture-name regex; no match = "" (|| true absorbs the - basename -a "$FIX"/*.traces.jsonl | grep -E "$1" | LC_ALL=C sort | head -1 || true - } # grep miss, which pipefail would otherwise turn into an abort) - for b in $benches; do - f=$(pick "^${b}-[0-9]+-[a-z0-9-]+\.traces\.jsonl$") - if [ -n "$f" ]; then picks="$picks $f"; else uncovered="$uncovered benchmark/$b"; fi - done - for a in $agents; do - f=$(pick "^[a-z0-9-]+-[0-9]+-${a}\.traces\.jsonl$") - if [ -n "$f" ]; then picks="$picks $f"; else uncovered="$uncovered agent/$a"; fi - done - [ -z "$uncovered" ] || echo "::warning::affected components with no replay fixture (untested by this gate):$uncovered" - # Fixture name -> test ident; dedup; cap (a core change affects dozens - # of leaves — three fixtures exercise the cascade without an hour-long PR). - tests=$(tr ' ' '\n' <<< "$picks" | grep . | sed 's/\.traces\.jsonl$//' | tr '-' '_' | sed 's/^/replay_/' \ - | LC_ALL=C sort -u | head -3 | tr '\n' ' ' || true) - echo "tests=${tests% }" >> "$GITHUB_OUTPUT" - echo "selected: ${tests:-}" + run: tests/select-affected.sh - name: Install Rust toolchain - if: steps.sel.outputs.tests != '' + if: steps.sel.outputs.any == 'true' uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable - name: Cache cargo - if: steps.sel.outputs.tests != '' + if: steps.sel.outputs.any == 'true' uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 with: shared-key: replay - name: Log in to GHCR - if: steps.sel.outputs.tests != '' + if: steps.sel.outputs.any == 'true' uses: docker/login-action@v4 with: registry: ghcr.io @@ -235,6 +208,47 @@ jobs: echo "::error::machine-selected fixtures matched 0 tests: $TESTS"; exit 1 fi + # The changed agents' own suite. AGENTS_SMOKE_BUILD makes it build from + # local source (the nightly path) — a PR's agent image does not exist in + # the registry yet, which is exactly what we need to test. + - name: Smoke the affected agents + if: ${{ !cancelled() && steps.sel.outputs.agents != '' }} + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} + AGENTS_SMOKE_BUILD: "1" + IDS: ${{ steps.sel.outputs.agents }} + run: | + set -eo pipefail + # shellcheck disable=SC2086 # IDS is a space-separated ident list + cargo test -p eval-containers-tests-run --test agents \ + -- --ignored --nocapture --test-threads=1 --exact $IDS \ + | tee agents.log + if grep -qE "running 0 tests" agents.log; then + echo "::error::machine-selected agent smoke matched 0 tests: $IDS"; exit 1 + fi + + # The changed benchmarks' gold-solution check: the oracle must still score + # 1.0, so a broken grader or a moved dataset fails the PR that broke it. + # Per-task benchmarks are included — the suite covers each with one pinned + # task. Coverage comes from the harness (ORACLE_LIST), never re-derived + # here: a benchmark it deliberately cannot oracle (terminal-bench, whose + # upstream gold drifted) warns instead of failing the PR. + - name: Oracle the affected benchmarks + if: ${{ !cancelled() && steps.sel.outputs.oracle != '' }} + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} + CAND: ${{ steps.sel.outputs.oracle }} + run: | + set -eo pipefail + # shellcheck disable=SC2086 # CAND is a space-separated name list + only=$(tests/select-affected.sh oracle-filter $CAND | tee /dev/stderr | tail -1) + [ -n "$only" ] || { echo "no covered benchmark to oracle"; exit 0; } + ORACLE_ONLY="$only" cargo test -p eval-containers-tests-run --test oracle \ + -- --ignored --nocapture --test-threads=1 | tee oracle.log + if grep -qE "running 0 tests" oracle.log; then + echo "::error::oracle suite matched 0 tests for: $only"; exit 1 + fi + # Per-PR gate: cheapest full-stack fixture (bigcodebench + zerostack) on the real # stack (otelcol → bifrost → runner), so a gateway/otelcol regression fails the PR. replay-e2e: diff --git a/tests/run/oracle/test.rs b/tests/run/oracle/test.rs index 35e32cbf..46d9b55b 100644 --- a/tests/run/oracle/test.rs +++ b/tests/run/oracle/test.rs @@ -84,8 +84,11 @@ async fn benchmarks_are_oracle_solvable() { // Collect every oracle check as (label, argv) up front so the heavy part — // each check `--local`-builds the benchmark image — can be fanned across a // matrix. ORACLE_SHARDS = matrix size, ORACLE_SHARD = this job's index; each - // job runs the round-robin slice `i % SHARDS == SHARD`. ORACLE_ONLY= - // narrows to matching labels (single-benchmark validation). All three unset + // job runs the round-robin slice `i % SHARDS == SHARD`. ORACLE_ONLY is a + // comma/space-separated list of substrings and narrows to labels matching + // ANY of them (the per-PR affected gate passes the changed benchmarks); a + // list that matches nothing is a hard error, never a green no-op — a + // machine-generated filter must not pass vacuously. All three unset // (the default) => SHARDS=1, SHARD=0, no filter => every check runs, so a // plain `cargo test --test oracle -- --ignored` is unchanged. let mut checks: Vec<(String, Vec)> = Vec::new(); @@ -124,11 +127,27 @@ async fn benchmarks_are_oracle_solvable() { )); } + checks.sort(); + // ORACLE_LIST=1 prints the coverage this suite has and returns — the + // per-PR gate asks the harness what it can oracle instead of + // re-deriving it (per-task benchmarks are covered by one pinned task + // each; a few, like terminal-bench, are deliberately uncovered). + if std::env::var_os("ORACLE_LIST").is_some() { + for (label, _) in &checks { + println!("oracle-label: {label}"); + } + return; + } + let only = std::env::var("ORACLE_ONLY").unwrap_or_default(); - if !only.is_empty() { - checks.retain(|(label, _)| label.contains(&only)); + let wanted: Vec<&str> = only.split([',', ' ']).filter(|s| !s.is_empty()).collect(); + if !wanted.is_empty() { + checks.retain(|(label, _)| wanted.iter().any(|w| label.contains(w))); + assert!( + !checks.is_empty(), + "ORACLE_ONLY={only:?} matched no benchmark — refusing to pass vacuously" + ); } - checks.sort(); // deterministic order so each shard's slice is stable let shard: usize = std::env::var("ORACLE_SHARD") .ok() .and_then(|s| s.parse().ok()) diff --git a/tests/select-affected.sh b/tests/select-affected.sh new file mode 100755 index 00000000..df511e51 --- /dev/null +++ b/tests/select-affected.sh @@ -0,0 +1,116 @@ +#!/usr/bin/env bash +# tests/select-affected.sh — pick the tests a change actually needs. +# +# Affected components come from the build-input hashes: the targets whose hash +# differs between BASE and HEAD (containers/scripts/fleet-hash.sh). That is a +# pure git computation — no registry, no daemon — which is what lets the per-PR +# gate use it (.agents/verification/RULES.md rule 1). Base-image cascades are +# free: a core/ change moves every dependent leaf's bases component. +# +# Selection is machine-generated, so it never passes vacuously: a component +# with no test of its own is named in a ::warning, and a cap that drops +# candidates says what it dropped. The callers fail on "running 0 tests". +# +# select-affected.sh [base-ref] emit tests/agents/oracle/any (below) +# select-affected.sh oracle-filter … keep only benchmarks the oracle covers +# +# Outputs (to $GITHUB_OUTPUT when set, else stdout): +# tests replay test idents agents agent_smoke idents +# oracle benchmark candidates any 'true' if anything was selected +# Env: BASE (base ref; default origin/main), REPLAY_CAP=3, SMOKE_CAP=2 +set -eo pipefail +cd "$(dirname "$0")/.." + +FIX=tests/run/replay/fixtures +# Warnings are collected, not printed inline: cap() runs inside command +# substitution (a subshell), so a warning echoed there would be captured into +# the value being computed, and one sent to stderr would not be rendered as a +# workflow command. A file survives the subshell; the flush at the end puts +# every warning on stdout, where GitHub parses it. +WARNF=$(mktemp); trap 'rm -f "$WARNF"' EXIT +warn() { echo "::warning::$1" >> "$WARNF"; } +out() { if [ -n "${GITHUB_OUTPUT:-}" ]; then echo "$1=$2" >> "$GITHUB_OUTPUT"; else echo "$1=$2"; fi; } +# Keep the first $1 entries of stdin, announcing what that drops — a cap that +# hides candidates would read as "covered" when it is not. +cap() { + local all n; all=$(cat); n=$(grep -c . <<< "$all" || true) + [ "$n" -le "$1" ] || warn "$2: $n affected, testing the first $1 ($(tr '\n' ' ' <<< "$all" | cut -c1-200))" + head -"$1" <<< "$all" +} + +# ── oracle-filter: intersect candidates with the suite's declared coverage ── +# Coverage is the harness's business (per-task benchmarks are covered by one +# pinned task each; a few, like terminal-bench, are deliberately uncovered), so +# ask it rather than re-deriving the rules here. +if [ "${1:-}" = "oracle-filter" ]; then + shift + labels=$(ORACLE_LIST=1 cargo test -p eval-containers-tests-run --test oracle \ + -- --ignored --nocapture 2>/dev/null | sed -n 's/^oracle-label: //p') + [ -n "$labels" ] || { echo "::error::ORACLE_LIST produced no labels"; exit 1; } + covered=""; uncovered="" + for b in "$@"; do + if grep -qE "^${b}( \(task |$)" <<< "$labels"; then covered="$covered$b"$'\n' + else uncovered="$uncovered $b"; fi + done + [ -z "$uncovered" ] || warn "no oracle check exists for:$uncovered (unverified by this gate)" + # Each check --local-builds a benchmark image (per-task ones pull a GB base). + filtered=$(grep . <<< "$covered" | cap "${ORACLE_CAP:-2}" oracle | tr '\n' ',' | sed 's/,$//') + cat "$WARNF" # flush before the value, so stdout's last line is it + printf '%s\n' "$filtered" + exit 0 +fi + +# ── affected set ──────────────────────────────────────────────────────────── +BASE="${1:-${BASE:-origin/main}}" +# Diff against the MERGE BASE, not the base branch's tip: a PR is responsible +# for what it changed, not for what main changed after it branched (GitHub's +# pull_request.base.sha is the tip, so comparing to it selects unrelated work). +BASE=$(git merge-base "$BASE" HEAD 2>/dev/null || echo "$BASE") +changed=$(comm -13 \ + <(REF="$BASE" containers/scripts/fleet-hash.sh | LC_ALL=C sort) \ + <(containers/scripts/fleet-hash.sh | LC_ALL=C sort) | cut -f1) +benches=$(sed -n 's/^benchmark-//p' <<< "$changed") +agents=$(sed -n 's/^agent-//p' <<< "$changed") +echo "affected: $(tr '\n' ' ' <<< "$changed")" + +# ── replay: one recorded fixture per affected component ───────────────────── +pick() { basename -a "$FIX"/*.traces.jsonl | grep -E "$1" | LC_ALL=C sort | head -1 || true; } +picks=""; uncovered="" +for b in $benches; do + f=$(pick "^${b}-[0-9]+-[a-z0-9-]+\.traces\.jsonl$") + if [ -n "$f" ]; then picks="$picks$f"$'\n'; else uncovered="$uncovered benchmark/$b"; fi +done +for a in $agents; do + f=$(pick "^[a-z0-9-]+-[0-9]+-${a}\.traces\.jsonl$") + if [ -n "$f" ]; then picks="$picks$f"$'\n'; else uncovered="$uncovered agent/$a"; fi +done +[ -z "$uncovered" ] || warn "affected components with no replay fixture (untested by this gate):$uncovered" +tests=$(grep . <<< "$picks" | sed 's/\.traces\.jsonl$//' | tr '-' '_' | sed 's/^/replay_/' \ + | LC_ALL=C sort -u | cap "${REPLAY_CAP:-3}" replay | tr '\n' ' ' || true) + +# ── agent smoke: the changed agents' own suite ────────────────────────────── +# --exact idents downstream, so agent_claude_code never matches _rtk. bob has +# no agent_smoke! by design (IBM-backend-tied; see tests/run/agents/test.rs). +smoke="" +for a in $agents; do + id="agent_$(tr '-' '_' <<< "$a")" + if grep -q "agent_smoke!($id," tests/run/agents/test.rs; then smoke="$smoke$id"$'\n' + else warn "agent $a has no agent_smoke! test (untested by this gate)"; fi +done +smoke=$(grep . <<< "$smoke" | LC_ALL=C sort -u | cap "${SMOKE_CAP:-2}" "agent smoke" | tr '\n' ' ' || true) + +# ── oracle: every changed benchmark is a candidate; oracle-filter decides ─── +orc=$(grep . <<< "$benches" | LC_ALL=C sort -u | tr '\n' ' ' || true) + +# Normalize before deciding: an empty list still leaves a separator space +# behind (head emits a newline, tr turns it into ' '), which would make the +# "anything to do?" flag true for a change that selected nothing. +tests=$(tr -s ' ' <<< "$tests" | sed 's/^ *//; s/ *$//') +smoke=$(tr -s ' ' <<< "$smoke" | sed 's/^ *//; s/ *$//') +orc=$(tr -s ' ' <<< "$orc" | sed 's/^ *//; s/ *$//') + +cat "$WARNF" # every warning, on stdout, outside any substitution +out tests "$tests" +out agents "$smoke" +out oracle "$orc" +out any "$([ -z "${tests}${smoke}${orc}" ] && echo false || echo true)"