Skip to content

fix(g2-followup): tiebreak newest-first journal scans on event_id #712

fix(g2-followup): tiebreak newest-first journal scans on event_id

fix(g2-followup): tiebreak newest-first journal scans on event_id #712

Workflow file for this run

name: CI
on:
push:
branches: [main]
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: true
jobs:
lint:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: pip install -e ".[dev]"
- name: Ruff check
run: ruff check src/ tests/ scripts/
- name: Ruff format check
run: ruff format --check src/ tests/ scripts/
- name: Type check
run: mypy src/ --ignore-missing-imports
schema-check:
runs-on: ubuntu-latest
needs: lint
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
fetch-depth: 2
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: pip install -e ".[dev]"
- name: Check schema evolution
run: python scripts/check_schema_evolution.py
test-unit:
runs-on: ubuntu-latest
needs: lint
permissions:
contents: read
id-token: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: |
pip install -e ".[dev,cloud]"
pip install -e "./sdk"
pip install -e "./integrations[mcp]"
- name: Prepare pytest temp directory
run: mkdir -p .tmp
- name: Run unit and property tests with coverage
run: |
# Full src/sdk baseline floor; changed-code coverage stays at 80% via Codecov patch status.
# --cov-branch turns the floor into a combined line+branch metric — local baseline
# 2026-05-25 is 62% (7716 lines / 2010 branches measured on HEAD `22b1be9`), so the
# 60% gate stays passing with a 2pp cushion. Raise the gate once that cushion grows.
python -m pytest tests/unit/ tests/property/ -v --tb=short --cov=src --cov=sdk --cov-branch --cov-report=xml --cov-report=term-missing --cov-fail-under=60
- name: Run quality validators coverage gate
run: |
python -m pytest tests/unit/test_validators.py -v --tb=short --cov=src.quality.validators --cov-report=term-missing --cov-fail-under=90
- name: Run freshness monitor coverage gate
run: |
python -m pytest tests/unit/test_freshness_monitor.py -v --tb=short --cov=src.quality.monitors.freshness_monitor --cov-report=term-missing --cov-fail-under=90
- name: Run event producer coverage gate
run: |
python -m pytest tests/unit/test_event_producer.py -v --tb=short --cov=src.ingestion.producers.event_producer --cov-report=term-missing --cov-fail-under=90
- name: Run SQL guard coverage gate
run: |
# Security-critical NL->SQL guard: validate_nl_sql (SELECT-only, no DML,
# tenant table allow-list, recursive-CTE shadow reject). test_sql_guard
# plus test_sql_guard_mutation give full module coverage (100% local);
# the 90% gate keeps a 10pp regression cushion.
python -m pytest tests/unit/test_sql_guard.py tests/unit/test_sql_guard_mutation.py -v --tb=short --cov=src.serving.semantic_layer.sql_guard --cov-report=term-missing --cov-fail-under=90
- name: Run rate limiter coverage gate
run: |
# Security-critical sliding-window rate limiter (Redis + in-memory
# fail-open fallback); local module coverage is 98% (only the optional
# redis auto-construct line is env-gated), so the 90% gate keeps a
# cushion on a mutmut target.
python -m pytest tests/unit/test_rate_limiter.py -v --tb=short --cov=src.serving.api.rate_limiter --cov-report=term-missing --cov-fail-under=90
- name: Run auth manager coverage gate
run: |
# Security-critical auth manager (key match/verify, tenant isolation,
# rate-limit/failed-auth windows, rotation grace) and a mutmut target;
# the gate runs its dedicated unit files. Module coverage is 94% so the
# 90% gate keeps a cushion; the remaining gap is the platform-divergent
# SIGHUP handler and bcrypt rotation paths the integration/e2e auth
# suites cover.
#
# NOTE: unlike the other per-module gates this uses `coverage run` +
# `coverage report --include`, NOT `pytest --cov=<module>`. The auth
# manager pulls in duckdb (usage table), and pytest-cov's source
# instrumentation of a duckdb-importing module trips duckdb's lazy
# `_duckdb._sqltypes` import at COLLECTION time, both locally and on CI
# runners. `coverage run` imports duckdb normally and avoids the break.
python -m coverage run -m pytest tests/unit/test_auth.py tests/unit/test_auth_manager_pure_logic.py tests/unit/test_auth_manager_memory_bounds.py tests/unit/test_auth_hashed_key_guidance.py tests/unit/test_auth_argon2_lookup.py -p no:schemathesis
python -m coverage report --include="*/serving/api/auth/manager.py" --show-missing --fail-under=90
- name: Run key rotation coverage gate
run: |
# Security-critical key-rotation lifecycle (create/rotate/revoke,
# grace-period scheduling, rotation status) and a mutmut target. Like
# the auth manager gate it pulls in duckdb, so it uses coverage run +
# coverage report --include (not pytest --cov) to avoid the
# duckdb _duckdb._sqltypes collection break. Module coverage is 93%.
python -m coverage run -m pytest tests/unit/test_key_rotation.py -p no:schemathesis
python -m coverage report --include="*/serving/api/auth/key_rotation.py" --show-missing --fail-under=90
- name: Run outbox coverage gate
run: |
# Security/reliability-critical at-least-once outbox dispatch loop
# (delivery, retry/backoff, poison-to-failed, mark-sent transactions)
# and a mutmut target. Imports duckdb, so it uses coverage run +
# coverage report --include like the auth gates. Module coverage is
# 92% across the two dedicated unit files.
python -m coverage run -m pytest tests/unit/test_outbox_processor.py tests/unit/test_outbox_connection_guard.py -p no:schemathesis
python -m coverage report --include="*/processing/outbox.py" --show-missing --fail-under=90
- name: Run query package coverage gate
run: |
# The NL->SQL orchestration surface (engine, entity/metric/NL query
# mixins, SQL builder) and a mutmut target set; the old single-file
# query_engine.py is a re-export shim, so the gate spans the whole
# query package. The engine imports duckdb, so it uses coverage run +
# coverage report --include like the auth/outbox gates. Package
# coverage is 97% across the six dedicated unit files; the gap is
# the OTel span-recording branches the integration suites cover.
# test_pipeline_events_scan.py covers QueryEngine.fetch_pipeline_events
# (the backend event scan the webhook dispatcher and SSE delegate to).
python -m coverage run -m pytest tests/unit/test_query_engine.py tests/unit/test_query_engine_injection.py tests/unit/test_query_engine_mixin_contracts.py tests/unit/test_paginated_nl_query.py tests/unit/test_query_package_logic.py tests/unit/test_pipeline_events_scan.py -p no:schemathesis
python -m coverage report --include="*/serving/semantic_layer/query/*" --show-missing --fail-under=90
- name: Upload coverage
uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0
with:
files: coverage.xml
use_oidc: true
fail_ci_if_error: false
test-integration:
runs-on: ubuntu-latest
needs: lint
services:
kafka:
image: confluentinc/cp-kafka:7.7.0
ports:
- 9092:9092
env:
KAFKA_NODE_ID: 1
KAFKA_PROCESS_ROLES: broker,controller
KAFKA_CONTROLLER_QUORUM_VOTERS: 1@localhost:29093
KAFKA_LISTENERS: PLAINTEXT://0.0.0.0:9092,CONTROLLER://0.0.0.0:29093
KAFKA_ADVERTISED_LISTENERS: PLAINTEXT://localhost:9092
KAFKA_CONTROLLER_LISTENER_NAMES: CONTROLLER
KAFKA_LISTENER_SECURITY_PROTOCOL_MAP: PLAINTEXT:PLAINTEXT,CONTROLLER:PLAINTEXT
KAFKA_OFFSETS_TOPIC_REPLICATION_FACTOR: 1
CLUSTER_ID: "CITestCluster01"
clickhouse:
# Live coverage for the ClickHouse serving backend's sqlglot
# transpile path (H-C2); test_clickhouse_backend_live.py skips
# itself when CLICKHOUSE_LIVE_HOST is absent.
image: clickhouse/clickhouse-server:25.3
ports:
- 8123:8123
env:
CLICKHOUSE_USER: agentflow
CLICKHOUSE_PASSWORD: agentflow
CLICKHOUSE_DB: agentflow
postgres:
# Live coverage for PostgresControlPlaneStore (ADR 0010 slice 5);
# test_control_plane_postgres_live.py skips itself when
# AGENTFLOW_TEST_PG_DSN is absent.
image: postgres:17
ports:
- 5432:5432
env:
POSTGRES_USER: agentflow
POSTGRES_PASSWORD: agentflow
POSTGRES_DB: agentflow
options: >-
--health-cmd "pg_isready -U agentflow"
--health-interval 5s
--health-timeout 5s
--health-retries 10
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: |
pip install -e ".[dev,cloud,postgres]"
pip install -e "./sdk"
- name: Prepare pytest temp directory
run: mkdir -p .tmp
- name: Wait for Kafka
run: |
timeout 30 bash -c 'until nc -z localhost 9092; do sleep 1; done'
- name: Wait for ClickHouse
run: |
timeout 60 bash -c 'until curl -sf http://localhost:8123/ping; do sleep 1; done'
- name: Run integration tests
env:
CLICKHOUSE_LIVE_HOST: localhost
CLICKHOUSE_LIVE_PORT: "8123"
CLICKHOUSE_LIVE_USER: agentflow
CLICKHOUSE_LIVE_PASSWORD: agentflow
CLICKHOUSE_LIVE_DATABASE: agentflow
AGENTFLOW_TEST_PG_DSN: postgresql://agentflow:agentflow@localhost:5432/agentflow
run: pytest tests/integration/ -v --tb=short
helm-schema-live:
runs-on: ubuntu-latest
needs: lint
timeout-minutes: 8
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: pip install -e ".[dev]"
- uses: azure/setup-helm@9bc31f4ebc9c6b171d7bfbaa5d006ae7abdb4310 # v5.0.1
- uses: helm/kind-action@ef37e7f390d99f746eb8b610417061a60e82a6cc # v1.14.0
with:
install_only: true
- name: Prepare pytest temp directory
run: mkdir -p .tmp
- name: Run Helm schema live validation
run: python -m pytest tests/integration/test_helm_values_live_validation.py -v -m integration --tb=short
perf-check:
runs-on: ubuntu-latest
needs:
- test-unit
- test-integration
timeout-minutes: 20
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.11"
- name: Install dependencies
run: pip install -e ".[dev,load,cloud]"
- name: Run benchmark
# DuckDB-profile benchmark: the pipeline runs on the runner host with no
# ClickHouse service, and the perf history compares against DuckDB-era
# numbers. The shipped ClickHouse path is exercised by the E2E lane.
env:
SERVING_BACKEND: duckdb
run: python scripts/run_benchmark.py
- name: Convert benchmark report to JSON
run: |
python - <<'PY'
import json
import re
from pathlib import Path
report_path = Path("docs/benchmark.md")
report = report_path.read_text(encoding="utf-8")
lines = [line.strip() for line in report.splitlines() if line.startswith("|")]
if len(lines) < 3:
raise SystemExit("Benchmark results table not found in docs/benchmark.md")
generated_at_match = re.search(r"Generated: `([^`]+)`", report)
endpoints = {}
for line in lines[2:]:
columns = [column.strip() for column in line.strip("|").split("|")]
if len(columns) != 8:
continue
endpoint, requests, failures, failure_rate, rps, p50, p95, p99 = columns
endpoints[endpoint] = {
"request_count": int(requests),
"failure_count": int(failures),
"failure_rate_percent": float(failure_rate.removesuffix("%")),
"requests_per_second": float(rps),
"p50_latency_ms": float(p50.removesuffix(" ms")),
"p95_latency_ms": float(p95.removesuffix(" ms")),
"p99_latency_ms": float(p99.removesuffix(" ms")),
}
aggregate = endpoints.pop("ALL", None)
if aggregate is None:
raise SystemExit("Missing ALL aggregate row in benchmark report.")
current_report = {
"generated_at": generated_at_match.group(1) if generated_at_match else None,
"source": str(report_path),
"aggregate": aggregate,
"endpoints": endpoints,
}
Path("/tmp/current.json").write_text(
json.dumps(current_report, indent=2) + "\n",
encoding="utf-8",
)
PY
- name: Compare to baseline
run: python scripts/check_performance.py docs/benchmark-baseline.json /tmp/current.json
terraform-validate:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1
with:
terraform_version: "1.8.0"
- name: Terraform fmt check
run: terraform fmt -check -recursive infrastructure/terraform/
- name: Terraform init
run: |
cd infrastructure/terraform
terraform init -backend=false
- name: Terraform validate
run: |
cd infrastructure/terraform
terraform validate