Skip to content

test(code-index): assert observed refusal and accepted values #6977

test(code-index): assert observed refusal and accepted values

test(code-index): assert observed refusal and accepted values #6977

Workflow file for this run

name: CI
on:
push:
branches: [master, feature/holographic-memory, codex/tracedecay-total-redesign-plan-reopened]
pull_request:
branches: ['**']
permissions:
contents: read
# One run per pull request (or per pushed ref). A newer push to a pull
# request supersedes the run in flight: its verdict is for a head nobody can
# merge any more, and letting it finish only holds runners the current head is
# waiting for. Keeping master-bound runs to completion did not buy a verdict
# per tip either: GitHub holds one pending run per group, so while a 100-min
# run finished, every intermediate push was queued and then cancelled by the
# next (runs 34311300344, 34312118984, 34313237349 never scheduled a job) and
# only the head at the moment the old run ended got a verdict, ~100 min late.
# `push` runs are never cancelled.
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
env:
CARGO_TERM_COLOR: always
CARGO_INCREMENTAL: "0"
CARGO_PROFILE_DEV_DEBUG: "0"
CARGO_PROFILE_TEST_DEBUG: "0"
AST_GREP_VERSION: "0.44.0"
jobs:
# Single source of truth for the fork/base-branch guard: heavy jobs run on
# pushes, on PRs from this repository, and on fork PRs that target an
# integration branch. Guarded jobs `needs:`-check this job's output instead
# of each repeating the expression, so the branch list lives in one place.
scope-gate:
name: Scope gate
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
run-heavy: ${{ steps.decide.outputs.run-heavy }}
linux-partitions: ${{ steps.linux-partitions.outputs.matrix }}
macos-groups: ${{ steps.macos-groups.outputs.matrix }}
steps:
- name: Decide whether guarded jobs run
id: decide
run: echo "run-heavy=${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository || contains(fromJSON('["master","feature/holographic-memory"]'), github.event.pull_request.base.ref) }}" >> "$GITHUB_OUTPUT"
# The Linux and macOS test matrices come from the same manifest the
# partition jobs select their targets from, so a partition cannot exist
# without a job or a job without a partition. This needs the checkout
# and nothing else.
- uses: actions/checkout@v7
with:
sparse-checkout: |
.github/linux-test-partitions.json
scripts/linux-test-partitions.py
sparse-checkout-cone-mode: false
- name: Derive the Linux test matrix
id: linux-partitions
run: echo "matrix=$(python3 scripts/linux-test-partitions.py matrix)" >> "$GITHUB_OUTPUT"
- name: Derive the macOS test matrix
id: macos-groups
run: echo "matrix=$(python3 scripts/linux-test-partitions.py macos-matrix)" >> "$GITHUB_OUTPUT"
# Exercises the checked-out PR head through the CLI against an isolated
# daemon (scripts/ci-pr-dogfood-smoke.sh). The binary is the one `debug-cli`
# builds from the merge ref: the smoke only needs an executable `tracedecay`,
# and building a second one from the head cost a cold
# `cargo build -p tracedecay-cli` that never fit this job's budget (runs
# 34217290139 and 34231734416: the build was cancelled after 29 minutes
# while `debug-cli` finished the same build in 30-31 minutes and published
# it).
pr-dogfood:
name: PR checkout dogfood
needs: [scope-gate, debug-cli]
if: ${{ github.event_name == 'pull_request' && needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
steps:
- uses: actions/checkout@v7
with:
fetch-depth: 0
persist-credentials: false
ref: ${{ github.event.pull_request.head.sha }}
- name: Fetch the debug CLI built once for this run
uses: actions/download-artifact@v4
with:
name: debug-tracedecay-cli
path: target/debug
- name: Restore the executable bit the artifact store drops
run: chmod +x target/debug/tracedecay
- name: Test dogfood harnesses
run: |
python3 scripts/test-check-pr-dogfood-output.py
python3 scripts/test-efficiency-scorecard.py
python3 scripts/test-pr-dogfood-portability.py
- name: Initialize checkout and exercise CLI tools
run: scripts/ci-pr-dogfood-smoke.sh
env:
TRACEDECAY_BIN: target/debug/tracedecay
TRACEDECAY_DOGFOOD_BASE_REF: ${{ github.event.pull_request.base.sha }}
TRACEDECAY_DOGFOOD_HEAD_REF: ${{ github.event.pull_request.head.sha }}
TRACEDECAY_DOGFOOD_BASE_BRANCH: ${{ github.event.pull_request.base.ref }}
TRACEDECAY_DOGFOOD_HEAD_BRANCH: ${{ github.event.pull_request.head.ref }}
# One dev-profile `tracedecay` binary for every job that only exercises the
# built CLI (`pr-dogfood`, Hermes, Claude Code, OpenCode). Each of those
# jobs used to compile the whole workspace itself behind its own cache key,
# so a cold cache cost the run several 30-minute builds instead of one and
# tripped their timeouts.
debug-cli:
name: Build debug CLI
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
timeout-minutes: 45
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@stable
- uses: ./.github/actions/setup-linux-mold
# This is the workflow's only dev-profile build, so `ci-dev-*` has one
# writer. rust-cache stores dependency artifacts only. The repository
# cache is capped at 10 GB and was full of per-job copies of the same
# dependencies (`pr-dogfood` and `dashboard` each built this graph too),
# which evicted the test lanes' caches between runs and made every test
# run a cold build. Only the lockfile-keyed dependency cache is kept:
# the workspace crates are exactly what a push changes, so caching
# their outputs per run re-saved ~1 GB per push for a restore that hit
# nothing.
- name: Cache Rust build outputs
uses: Swatinem/rust-cache@v2
with:
shared-key: ci-dev-${{ runner.os }}-${{ runner.arch }}
cache-on-failure: true
- name: Build tracedecay binary
run: cargo build -p tracedecay-cli --bin tracedecay --locked
- name: Publish the binary for the host integration jobs
uses: actions/upload-artifact@v4
with:
name: debug-tracedecay-cli
path: target/debug/tracedecay
if-no-files-found: error
retention-days: 1
commit-messages:
name: Commit Messages
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
with:
fetch-depth: 0
- uses: actions/setup-node@v4
with:
node-version: 22
cache: npm
- name: Install commit message lint dependencies
run: npm ci
- name: Reject tracked ignored files
run: scripts/check-release-pr-integrity.sh HEAD HEAD
- name: Test bounded commit range linting
run: python3 scripts/test-lint-commit-range.py
- name: Validate PR commit messages
if: github.event_name == 'pull_request'
env:
BASE_SHA: ${{ github.event.pull_request.base.sha }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: node scripts/lint-commit-range.mjs --repository "$PWD" "$BASE_SHA" "$HEAD_SHA"
- name: Validate pushed commit messages
if: github.event_name == 'push'
env:
BEFORE_SHA: ${{ github.event.before }}
HEAD_SHA: ${{ github.sha }}
run: |
set -euo pipefail
if [ "$BEFORE_SHA" = "0000000000000000000000000000000000000000" ]; then
if git rev-parse "${HEAD_SHA}^" >/dev/null 2>&1; then
base_sha="${HEAD_SHA}^"
else
git show --no-patch --format=%B "$HEAD_SHA" | npm run lint:commit --
exit 0
fi
else
base_sha="$BEFORE_SHA"
fi
node scripts/lint-commit-range.mjs --repository "$PWD" "$base_sha" "$HEAD_SHA"
release-version-drift:
name: Release Version Drift
if: github.event_name == 'pull_request' && !startsWith(github.head_ref, 'release-please--')
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-python@v6
with:
python-version: "3.12"
- name: Test release version drift guard
run: bash tests/release_drift_check_test.sh
# The workflow-contract pin scripts were replaced by the safety-guard
# assertions (f3b065545) and the dogfood surface was retired (b73a8a396);
# this job keeps the guards that protect the shipped release flow.
- name: Check release safety guards
run: bash tests/release_safety_test.sh
- name: Test GitHub release installer
run: bash tests/install_script_test.sh
# These two sat unwired: they are the tests for
# `scripts/check-release-pr-integrity.sh` and
# `scripts/check-dashboard-bundle.py`, both of which CI does run. So the
# guards were enforced while the tests proving the guards work were not,
# which is the shape that let a `test-transport` target rot until it
# stopped compiling. Three of the five `tests/*_test.sh` gates were wired;
# these are the other two.
- name: Test release PR integrity guard
run: bash tests/release_pr_integrity_test.sh
- name: Test dashboard bundle check
run: bash tests/dashboard_bundle_check_test.sh
- name: Check release version drift
run: scripts/check-release-drift.sh
env:
GITHUB_TOKEN: ${{ github.token }}
# macOS compiles and runs the suite as groups of the Linux partitions
# (`macos_groups` in .github/linux-test-partitions.json), one hosted 3-vCPU
# `macos-14` job per group, so the lane's wall time is the slowest group
# instead of the whole workspace. The single job this replaces built
# `--workspace --bins --tests` in 80.2 min and ran the suite in 19.9 min
# (run 34303181887; run 34309279328 built for 74.9 min and was cancelled in
# its tests at 85 min), about 80 min after every other lane had reported,
# so it was the critical path of every pull request run. rust-cache carries
# the dependency graph only; the workspace crates compile fresh every run,
# and on three cores with ld64 that build is the whole cost.
#
# A group runs each of its partitions in turn against one target directory
# with the selection `linux-test-partition` resolves for it, so
# `root-sessions` and `root-journeys`, which select the same packages with
# the same feature, compile the chain beneath the root crate once and the
# second adds only its own suites. The grouping balances the warm Linux
# measurements (run 34309279328, compile + tests in minutes): root-lib
# 14.7 + 7.5 alone, since its root chain, library test target and 1675
# tests are the floor no split lowers; root-sessions 15.3 + 3.1 with
# root-journeys, which adds 2.7 + 2.7 on the shared resolution; runtime
# 15.6 + 2.4 with core-storage 10.4 + 1.9; core-contracts 13.9 + 0.6 with
# root-dashboard-api 12.7 + 0.8. Every pairing that puts a second partition
# beside a root group is heavier than the heaviest of these, and no two of
# the other four share a feature resolution, so grouping them differently
# saves no compile. At the 3-vCPU floor (compile × 4/3, tests as measured)
# the groups come to 28, 31, 40 and 38 min; at the 2× the one macOS
# measurement showed (the chain beneath the root crate compiled in 19.0
# min against 9.3 on Arm in the runs above) to 38, 43, 57 and 56. The
# budgets in the manifest take the 2×, plus the cold-dependency allowance
# and headroom the Linux budgets carry, until a hosted run measures the
# groups. The account admits five concurrent macOS jobs; a run takes at
# most four (`MACOS_GROUP_CAP`) so another run's macOS jobs can start.
macos-test-partition:
name: Test macOS ${{ matrix.group }}
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: macos-14
# Each group's budget is set in the manifest beside the group, with the
# measurement it rests on.
timeout-minutes: ${{ matrix.timeout }}
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.scope-gate.outputs.macos-groups) }}
steps:
# Full history: the search-quality workload fixture pins a
# `source_repository_commit`, and `validate_source_bindings` resolves that
# object out of this checkout to prove the checked-in corpus really is the
# product source at that commit. A depth-1 checkout carries only the tip,
# so every candidate_output/report test fails with "resolve fixture source
# commit: An object with id ... could not be found".
- uses: actions/checkout@v7
with:
fetch-depth: 0
# `rust-analyzer` is a test dependency of this lane, not a convenience.
# `runtime_surface_acceptance::production_lsp_negotiates_and_projects_\
# canonical_context` asserts that a routed analyzer negotiates the
# standard methods the client declared, and the upstream `initialize`
# response is the only authority for those. Without the component the
# runner still has rustup's `rust-analyzer` proxy on PATH, so the daemon
# routes the language and only the spawn fails — the test would then
# measure a missing component instead of the negotiation contract. This
# action exports `RUSTUP_TOOLCHAIN`, so the component has to be installed
# here rather than through `rust-toolchain.toml`, which it overrides.
- uses: dtolnay/rust-toolchain@stable
with:
components: rust-analyzer
- uses: actions/setup-node@v4
with:
node-version: 22
- uses: actions/setup-python@v6
with:
python-version: "3.12"
- name: Install ast-grep
uses: ./.github/actions/install-ast-grep
with:
version: ${{ env.AST_GREP_VERSION }}
- name: Install cargo-nextest
uses: taiki-e/install-action@nextest
# rust-cache carries the dependency graph, keyed by toolchain and
# lockfile; the workspace crates compile fresh every run. No per-run
# compiler store: the perf-profile test graph wrote 15.7 GB of outputs
# per build on macOS, which no store inside the repository's 10 GB
# cache budget can hold. One writer keeps the lane at one entry:
# `root-suites` resolves the widest graph (the root crate with its
# dev-dependencies, the CLI and the evaluator), so its save covers
# `root-lib` exactly; the other groups unify dependency features
# differently and recompile a few small dependency units against it
# every run, as the lower Linux partitions do. Restored before the first
# `cargo metadata` below so the registry index comes from the cache too.
- name: Cache Rust build outputs
uses: Swatinem/rust-cache@v2
with:
shared-key: ci-test-full-${{ runner.os }}
save-if: ${{ matrix.group == 'root-suites' }}
cache-on-failure: true
# Every group job re-proves the cover and the grouping before it
# compiles anything: a gap found here fails the lane rather than passing
# it with a target nobody ran.
- name: Check the partitions cover every test target
run: |
python3 scripts/test-linux-test-partitions.py
python3 scripts/linux-test-partitions.py check
# Each partition in turn, as `linux-test-partition` runs it: the
# executables its suites spawn are built into the same resolution first
# (`build-args`; nothing for a partition whose tests spawn nothing),
# then nextest under the `ci` policy and the perf cargo profile with
# `--no-tests=fail`. One target directory for the group, so a later
# partition's shared units are fingerprint hits. Every partition runs
# even after an earlier one fails — as nextest's `ci` policy runs every
# test — and the step fails if any did. Every nextest run rewrites the
# `ci` profile's junit.xml, so each partition's report is moved under
# its own name before the next runs. The per-partition wall times this
# prints are the measurements the manifest's macOS budgets wait for.
- name: Run the group's partitions
env:
PARTITIONS: ${{ matrix.partitions }}
run: |
set -uo pipefail
run_partition() {
local partition="$1" build selection
build="$(python3 scripts/linux-test-partitions.py build-args "$partition")" || return 1
selection="$(python3 scripts/linux-test-partitions.py cargo-args "$partition")" || return 1
if [ -n "$build" ]; then
eval cargo build --locked --profile perf "$build" || return 1
fi
eval cargo nextest run --profile ci --cargo-profile perf --locked "$selection" --no-tests=fail
}
read -ra partitions <<< "$PARTITIONS"
mkdir -p target/nextest/macos
failed=""
for partition in "${partitions[@]}"; do
echo "::group::Test partition ${partition}"
started=$SECONDS
if run_partition "$partition"; then
echo "${partition}: passed in $(( SECONDS - started )) s"
else
failed="${failed} ${partition}"
echo "${partition}: failed after $(( SECONDS - started )) s"
fi
[ ! -f target/nextest/ci/junit.xml ] || mv target/nextest/ci/junit.xml "target/nextest/macos/${partition}.xml"
echo "::endgroup::"
done
if [ -n "$failed" ]; then
echo "::error::Failed partitions:${failed}"
exit 1
fi
# `macos-test` folds the group reports back into one report.
- name: Upload macOS group nextest report
if: always()
uses: actions/upload-artifact@v4
with:
name: nextest-junit-macOS-${{ matrix.group }}
path: target/nextest/macos/
if-no-files-found: ignore
retention-days: 1
# The one `Test macOS` verdict, as when a single job carried the suite:
# every group must pass. It also folds the group reports back into the
# `nextest-junit-macOS` artifact the single job published, one directory
# per group holding a junit.xml per partition.
macos-test:
name: Test macOS
if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }}
needs: [macos-test-partition, scope-gate]
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Merge the group reports into the macOS nextest report
if: ${{ needs.macos-test-partition.result != 'skipped' }}
uses: actions/upload-artifact/merge@v4
with:
name: nextest-junit-macOS
pattern: nextest-junit-macOS-*
separate-directories: true
retention-days: 7
- name: Check macOS groups
if: ${{ !cancelled() }}
run: |
if [ "${{ needs.macos-test-partition.result }}" != "success" ]; then
echo "macOS group result: ${{ needs.macos-test-partition.result }}"
exit 1
fi
echo "All macOS groups passed."
# Linux compiles and runs the suite as parallel partitions of the test
# targets (.github/linux-test-partitions.json), one hosted 4-vCPU job each,
# so the lane's wall time is the slowest partition instead of the whole
# workspace. The single job that preceded it (run 34278967966, warm
# dependency cache, four compile slots) spent 55.5 min compiling: 32 min of
# workspace libraries beneath the root crate, then the root library
# (~8 min) and its library test target (~15.5 min), a serial chain that no
# job count shortens; the other 300-odd test units fit in the slots beside
# it. Tests then took 22.5 min, the dashboard's `test-transport` rebuild
# and run 11 min, and the hotpath parity rebuild ~16 min. Partitioned, the
# root chain is one job (`root-lib`: 19.7 min compile + 5.2 min tests, 25.9
# min in run 34296614024, the first partitioned run) while every other
# partition finishes in 17–27 min, so the lane's critical path drops from
# ~105 min to the slowest partition; the parity rebuild runs beside the
# partitions as `hotpath-parity` rather than after the root chain. Building
# the graph once and sharding only the test execution (the Windows lane's
# shape) keeps the whole 55 min compile plus an archive round trip on the
# path and then the dashboard rebuild and parity on the same job: ~88 min,
# or ~70 with both moved elsewhere. The partitions pay for the shorter path
# with duplicated compile — each root partition rebuilds the library chain
# beneath the root crate, about 200 core-minutes per run across the seven
# jobs — which is free on hosted runners where only wall time and the
# 20-job concurrency cap count. A cold dependency cache adds ~6 min to every
# job alike (the dependencies compile at opt-level 0 in ~17 core-minutes).
# The floor no partitioning reaches is the root chain itself: library
# chain + root lib + root lib-test + its tests, the ~26 min above.
#
# The partitions are cargo selections, not nextest filters: `-p` plus
# `--lib` / `--test <name>` / `--bins` decide what compiles, and a filterset
# would compile everything and skip at run time. The three root partitions
# share one package selection (`tracedecay`, `tracedecay-cli`,
# `tracedecay-search-eval`) with the root fixture feature, so they resolve
# one identical dependency graph and every cargo invocation inside a job
# (the test build, the executables the suites spawn) is a cache hit against
# it. `scripts/linux-test-partitions.py check` proves, from `cargo
# metadata`, that every test target in the workspace is selected by exactly
# one partition or listed under `not_run` with a reason, so a new crate or
# suite cannot fall out of the lane silently; `scope-gate` derives the
# matrix from the same manifest.
linux-test-partition:
name: Test Linux ${{ matrix.partition }}
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
# Each partition's budget is its measured wall time plus headroom, set in
# the manifest beside the selection it covers.
timeout-minutes: ${{ matrix.timeout }}
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.scope-gate.outputs.linux-partitions) }}
steps:
# Full history: the search-quality workload fixture pins a
# `source_repository_commit`, and `validate_source_bindings` resolves that
# object out of this checkout to prove the checked-in corpus really is the
# product source at that commit. A depth-1 checkout carries only the tip,
# so every candidate_output/report test fails with "resolve fixture source
# commit: An object with id ... could not be found".
- uses: actions/checkout@v7
with:
fetch-depth: 0
# `rust-analyzer` is a test dependency of this lane, not a convenience.
# `runtime_surface_acceptance::production_lsp_negotiates_and_projects_\
# canonical_context` asserts that a routed analyzer negotiates the
# standard methods the client declared, and the upstream `initialize`
# response is the only authority for those. Without the component the
# runner still has rustup's `rust-analyzer` proxy on PATH, so the daemon
# routes the language and only the spawn fails — the test would then
# measure a missing component instead of the negotiation contract. This
# action exports `RUSTUP_TOOLCHAIN`, so the component has to be installed
# here rather than through `rust-toolchain.toml`, which it overrides.
- uses: dtolnay/rust-toolchain@stable
with:
components: rust-analyzer
- uses: ./.github/actions/setup-linux-mold
- uses: actions/setup-node@v4
with:
node-version: 22
- uses: actions/setup-python@v6
with:
python-version: "3.12"
- name: Install ast-grep
uses: ./.github/actions/install-ast-grep
with:
version: ${{ env.AST_GREP_VERSION }}
- name: Install cargo-nextest
uses: taiki-e/install-action@nextest
# One dependency cache for the lane, as before the split. rust-cache
# stores the dependency graph only, keyed by toolchain and lockfile; the
# workspace crates compile fresh every run. A single writer keeps the
# lane at one entry in the repository's 10 GB cache: `root-journeys`
# resolves the widest graph (the root crate with its dev-dependencies,
# the CLI and the evaluator), so its save covers the other root
# partitions exactly. The three lower partitions unify dependency
# features differently and each recompile ~30 small dependency units
# (syn, serde_json, futures, chrono, ...) against it every run — a few
# minutes inside jobs that are far from the critical path, cheaper than
# three more cache entries. Restored before the first `cargo metadata`
# below so the registry index comes from the cache too.
- name: Cache Rust build outputs
uses: Swatinem/rust-cache@v2
with:
shared-key: ci-test-full-Linux-${{ runner.arch }}
save-if: ${{ matrix.partition == 'root-journeys' }}
cache-on-failure: true
# Every partition job re-proves the cover before it compiles anything:
# the manifest is consumed here, and a gap found here fails the lane
# rather than passing it with a target nobody ran.
- name: Check the partitions cover every test target
run: |
python3 scripts/test-linux-test-partitions.py
python3 scripts/linux-test-partitions.py check
- name: Resolve this partition's cargo selection
id: selection
run: |
{
echo "test=$(python3 scripts/linux-test-partitions.py cargo-args '${{ matrix.partition }}')"
echo "build=$(python3 scripts/linux-test-partitions.py build-args '${{ matrix.partition }}')"
} >> "$GITHUB_OUTPUT"
# Suites that spawn the CLI (`tests/common::tracedecay_bin`) or the
# search-eval evaluators (`search_eval_bin`) resolve them out of the
# profile directory, and cargo links a package's bins for a test build
# only when that package's own integration tests are selected. This
# builds the partition's selection plus those executables in one
# resolution: the test targets keep dev-dependencies in the graph, so
# the nextest build below is a cache hit, and the selection includes
# `tracedecay-search-eval` so the evaluator resolves
# `tracedecay-code-index` with its language tiers (a lone
# `-p tracedecay-search-eval` build yields one that cannot parse Rust).
# The `tracedecay-host-cli-fixture` example is the host fixture the CLI
# suites drive. Partitions whose tests spawn nothing skip this.
- name: Build the executables the suites spawn
if: ${{ steps.selection.outputs.build != '' }}
run: cargo build --locked --profile perf ${{ steps.selection.outputs.build }}
# nextest `ci` policy (.config/nextest.toml: fail-fast off, one retry
# that still fails flaky results, 8 threads, slow-timeout termination)
# and the perf cargo profile, as `cargo test-ci`; the selection replaces
# the alias's `--workspace`. `--no-tests=fail` so a selection that
# resolves to nothing cannot report green.
- name: Run tests
run: cargo nextest run --profile ci --cargo-profile perf --locked ${{ steps.selection.outputs.test }} --no-tests=fail
# `linux-test` folds the partition reports back into one report.
- name: Upload Linux partition nextest report
if: always()
uses: actions/upload-artifact@v4
with:
name: nextest-junit-Linux-${{ matrix.partition }}
path: target/nextest/ci/junit.xml
if-no-files-found: ignore
retention-days: 1
# The controlled-workload hotpath parity gate: `tracedecay-search-eval`'s
# `emit_controlled_workload_reports` example, built once with the hotpath
# profiler compiled out and once with `controlled-workload-hotpath`
# (`hotpath/hotpath`) on, must produce byte-identical durable results for
# the framed-log (private-fs) and cursor-parse (capture) workloads, from
# provably distinct executables. `scripts/build-controlled-workload-
# hotpath-helpers.py` builds the pair into target/controlled-workload-
# hotpath/, and the `#[ignore]`d test in the crate's library test target
# (`controlled_workloads::tests::hotpath_off_vs_on_durable_results_are_
# identical`) spawns them from there; `--run-ignored only` with
# `--no-tests=fail` is the only way it runs, and nothing else — no
# partition, artifact upload or junit merge — consumes the executables.
#
# Its own job rather than a tail on the `root-lib` partition, where it ran
# after the root library's tests: the feature-on build recompiles every
# workspace crate beneath the evaluator (all 21 depend on `hotpath`, so
# the feature flip changes each one's metadata hash) plus the example,
# ~10 min on top of the ~26 min root chain, which made that job the lane's
# critical path. Here the graph beneath the evaluator compiles from the
# checkout in both modes — a serial chain domain → contracts →
# rusqlite-runtime → runtime-core → code-index → sessions → query →
# search-eval → example of ~9 min per mode with four compile slots (cargo
# `--timings`, -j4 on EPYC 7742 cores; the same pair of builds took 19.5
# min on a hosted runner in run 34138693824, when the single-job lane
# still built them as their own resolution) — then the library test target
# (24 s against the feature-off graph) and the test itself (under 1 s).
# With setup, ~24 min hosted: beside the partitions and no longer than the
# slowest of them (25.9 and 27.4 min in run 34296614024), so the `Test
# Linux` verdict no longer waits for it. The property does not depend on
# the platform (the Windows lane never provisioned it for that reason), so
# a Linux runner carries it.
#
# Not a job in hotpath-coverage.yml: that workflow is path-filtered to
# Rust inputs and never runs on `push`, while this is a gate on every pull
# request head, and its feature-on slices (storage, sessions) do not cover
# the query/code-index/runtime-core graph the evaluator needs; rust-cache
# keeps dependency artifacts only, so no other job's compiled workspace
# crates could be reused anyway. The evaluator alone (`-p
# tracedecay-search-eval`) is the selection: the root partitions' wider
# one would turn every language tier on in `tracedecay-code-extraction`
# and `-code-index` for two builds that parse no source, and carry the root
# fixture feature for a crate this job never compiles. No cache: under
# this selection ~160 of the 420 registry crates (`syn`, `serde_core`,
# `tokio` and their dependents) unify differently from the lane's
# `ci-test-full-Linux` graph and would miss it, and the rest compile in
# the slots the serial chain leaves idle (the cold build measured 8.8 min
# against 9.1 for the warm feature-on one), so a restore
# would buy about what it costs, while another lineage would compete for
# the 10 GB budget the test lanes' dependency caches are already evicted
# from.
hotpath-parity:
name: Hotpath parity
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
# 19.5 min for the two builds (hosted, above) + ~1.5 cold registry the
# chain does not hide + ~1.5 setup + 0.5 test target + 0.5 run ≈ 24 min,
# plus 25 % headroom, as for the partition budgets.
timeout-minutes: 30
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@stable
- uses: ./.github/actions/setup-linux-mold
- name: Install cargo-nextest
uses: taiki-e/install-action@nextest
# Both modes under the evaluator's own selection; the helper copies
# each result out of target/perf/examples immediately so the second
# resolution cannot replace the first.
- name: Build controlled-workload hotpath parity executables
run: python3 scripts/build-controlled-workload-hotpath-helpers.py --profile perf
# The library test target resolves the same feature-off graph as the
# first build above, so only the test target itself compiles here.
# nextest `ci` policy and the perf cargo profile, as the partitions.
- name: Verify controlled-workload hotpath parity
run: |
cargo nextest run --profile ci --cargo-profile perf --locked -p tracedecay-search-eval --lib \
--run-ignored only \
-E 'test(=controlled_workloads::tests::hotpath_off_vs_on_durable_results_are_identical)' \
--no-tests=fail
# The one `Test Linux` verdict, as when a single job carried the suite:
# every partition and the hotpath parity gate must pass. It also folds the
# partition reports back into the `nextest-junit-Linux` artifact the single
# job published; the parity verdict is its step's exit status, as it was
# inside the partition, and is not part of that report.
linux-test:
name: Test Linux
if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }}
needs: [linux-test-partition, hotpath-parity, scope-gate]
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
# Each partition's junit.xml keeps its name under its partition's
# directory; the partition artifacts themselves expire after a day.
- name: Merge the partition reports into the Linux nextest report
if: ${{ needs.linux-test-partition.result != 'skipped' }}
uses: actions/upload-artifact/merge@v4
with:
name: nextest-junit-Linux
pattern: nextest-junit-Linux-*
separate-directories: true
retention-days: 7
- name: Check Linux partitions and hotpath parity
if: ${{ !cancelled() }}
run: |
if [ "${{ needs.linux-test-partition.result }}" != "success" ]; then
echo "Linux partition result: ${{ needs.linux-test-partition.result }}"
exit 1
fi
if [ "${{ needs.hotpath-parity.result }}" != "success" ]; then
echo "Hotpath parity result: ${{ needs.hotpath-parity.result }}"
exit 1
fi
echo "All Linux partitions and the hotpath parity gate passed."
windows-build:
name: Build Windows tests
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: windows-latest
# Build every Windows test once, then shard execution from the archive.
# Keep a bounded guard around the hosted build; Cargo timings are retained
# below so a slow or failed archive can be attributed instead of treating
# the timeout itself as the defect. Measured 2026-09-05 (run 33954670866,
# no parity rebuild): portability compile 31 min, archive still compiling
# the root crate's test targets at the 75-minute bound, so the archive
# never completed and the compiler caches never seeded. The bound is
# widened so one archive can finish and seed them; shrink it back once a
# warm run's timings show the real cost.
timeout-minutes: 120
steps:
- uses: actions/checkout@v7
- name: Tune Windows runner for build I/O
uses: ./.github/actions/tune-windows-runner
# Install only the toolchain pinned by rust-toolchain.toml. Installing
# `stable` as well makes Swatinem/rust-cache include an unused compiler
# in its environment key, fragmenting the cache whenever stable moves.
- name: Install pinned toolchain
shell: pwsh
run: |
rustup toolchain install
rustup show active-toolchain
- name: Use lld-link linker
shell: pwsh
run: |
where.exe lld-link.exe
lld-link.exe --version
"CARGO_TARGET_X86_64_PC_WINDOWS_MSVC_LINKER=lld-link.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append
# Node is only needed so build.rs can produce the embedded dashboard
# dist assets (npm ci + npm run build) once, instead of once per shard.
- uses: actions/setup-node@v4
with:
node-version: 22
cache: npm
cache-dependency-path: dashboard/package-lock.json
- name: Install cargo-nextest
uses: taiki-e/install-action@nextest
- name: Cache Windows Rust build outputs
uses: Swatinem/rust-cache@v2
with:
shared-key: ci-test-full-windows-msvc-lld
cache-on-failure: true
# Acceptance tests execute the ordinary evaluator binaries. Match the
# archive's features and perf profile to reuse dependency artifacts;
# nextest's archive.include carries the executables to every shard.
# `--tests` keeps the binaries in the dev-dependency graph the archive
# is built from; a `--bins`-only build resolves 60 dependencies with
# different features and compiles them a second time (see the Linux
# lane).
- name: Build workspace binaries and tests for the Windows test lane
shell: pwsh
run: cargo build --workspace --bins --tests --locked --profile perf --features tracedecay/test-helpers
# `--workspace`, not `-p tracedecay-cli`: the package selection decides
# feature unification, and the narrower one recompiles the code-index
# and extraction crates in a second configuration (see the Linux lane).
- name: Build Windows host-CLI test fixture
shell: pwsh
run: cargo build --workspace --example tracedecay-host-cli-fixture --locked --profile perf --features tracedecay/test-helpers
# The controlled-workload Hotpath parity helpers are provisioned and
# verified by the `hotpath-parity` job. Building them here would put a
# second, feature-on resolution of the search-eval dependency graph
# inside this job's bound (18 of its 62 minutes before it was cancelled),
# and the parity property does not depend on the platform.
# Same selection as `cargo test-ci` (.cargo/config.toml); nextest has no
# archive alias, so the arguments are spelled out here once.
- name: Build nextest archive
shell: pwsh
run: cargo nextest archive --workspace --profile ci --locked --features tracedecay/test-helpers --cargo-profile perf --timings --archive-file "$env:RUNNER_TEMP/nextest-archive.tar.zst"
- name: Upload Windows archive build timings
if: always()
uses: actions/upload-artifact@v4
with:
name: windows-nextest-build-timings
path: target/cargo-timings/cargo-timing.html
if-no-files-found: warn
retention-days: 7
- name: Upload nextest archive
uses: actions/upload-artifact@v4
with:
name: windows-nextest-archive
path: ${{ runner.temp }}/nextest-archive.tar.zst
retention-days: 1
compression-level: 0
# The archive already carries these (`archive.include` in
# .config/nextest.toml), but nothing lands on a shard until the run
# step extracts, so a missing evaluator can only surface as product
# test failures. Publishing them separately lets each shard restore and
# prove them *before* the suite starts. Same executables, from the
# workspace/test-helpers build above: an evaluator built under a
# narrower `-p tracedecay-search-eval` selection resolves
# tracedecay-code-index without its default language tiers and cannot
# parse Rust.
- name: Upload Windows evaluator executables
uses: actions/upload-artifact@v4
with:
name: windows-search-eval-bins
path: |
target/perf/tracedecay-search-eval.exe
target/perf/tracedecay-search-eval-direct.exe
if-no-files-found: error
retention-days: 1
compression-level: 0
windows-test-shard:
name: Test Windows shard ${{ matrix.partition }}/5
needs: windows-build
runs-on: windows-latest
timeout-minutes: 20
strategy:
fail-fast: false
matrix:
partition: [1, 2, 3, 4, 5]
steps:
# Full history for the same reason as the Linux and macOS partition
# jobs: the shards run the workspace suite remapped onto this checkout,
# so the pinned search-quality fixture commit must be resolvable here too.
- uses: actions/checkout@v7
with:
fetch-depth: 0
- name: Tune Windows runner for test I/O
uses: ./.github/actions/tune-windows-runner
with:
redirect-temp: "true"
- uses: actions/setup-node@v4
with:
node-version: 22
- uses: actions/setup-python@v6
with:
python-version: "3.12"
- name: Install ast-grep
uses: ./.github/actions/install-ast-grep
with:
version: ${{ env.AST_GREP_VERSION }}
# The architecture tests exec `cargo metadata` at runtime, and they run
# concurrently under nextest: without the pinned toolchain installed,
# each test process's rustup shim races to self-install it and the
# concurrent downloads corrupt each other's partial files. Install the
# rust-toolchain.toml pin serially up front.
- name: Install pinned toolchain
shell: pwsh
run: |
rustup toolchain install
rustup show active-toolchain
- name: Install cargo-nextest
uses: taiki-e/install-action@nextest
- name: Download nextest archive
uses: actions/download-artifact@v4
with:
name: windows-nextest-archive
path: ${{ runner.temp }}
- name: Download Windows evaluator executables
uses: actions/download-artifact@v4
with:
name: windows-search-eval-bins
path: ${{ runner.temp }}/search-eval-bin
# Preflight: the evaluator subprocesses the acceptance and packaged
# suites launch must exist, execute, and report their bundled workload
# before any test runs, so a provisioning gap fails here instead of as
# five product-test failures elsewhere in the shard matrix. `validate`
# against an empty directory is the packaged-identity probe: it reads
# only the workload compiled into the binary, so a build missing it
# cannot pass. The exports are the overrides both resolvers declare
# (`search_eval_bin` in crates/tracedecay/tests/common/mod.rs and
# `search_eval_direct_bin` in the default-feature packaged suite), which
# bind every test to the executables proven here rather than to whatever
# the archive happens to extract beside the test binaries.
- name: Preflight the restored evaluator executables
shell: pwsh
run: |
$ErrorActionPreference = "Stop"
$binDir = Join-Path $env:RUNNER_TEMP "search-eval-bin"
$probe = Join-Path $env:RUNNER_TEMP "search-eval-preflight"
New-Item -ItemType Directory -Force -Path $probe | Out-Null
foreach ($name in @("tracedecay-search-eval", "tracedecay-search-eval-direct")) {
$path = Join-Path $binDir "$name.exe"
if (-not (Test-Path -LiteralPath $path -PathType Leaf)) {
throw "restored evaluator '$name.exe' is missing at $path"
}
# stdout only: the typed report is the whole of it, and merging
# stderr in would make any log line unparseable JSON.
$stdout = & $path validate --repo-root $probe | Out-String
if ($LASTEXITCODE -ne 0) {
throw "$name validate exited with $LASTEXITCODE`n$stdout"
}
$report = $stdout | ConvertFrom-Json
if ($report.command -ne "validate" -or $report.status -ne "pass") {
throw "$name reported command=$($report.command) status=$($report.status)`n$stdout"
}
if ([int]$report.query_count -le 0 -or [int]$report.profile_count -le 0) {
throw "$name has no bundled workload: query_count=$($report.query_count) profile_count=$($report.profile_count)"
}
Write-Host "$name.exe ok: $($report.query_count) queries, $($report.profile_count) profiles"
}
"TRACEDECAY_SEARCH_EVAL_TEST_BIN=$binDir\tracedecay-search-eval.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append
"TRACEDECAY_SEARCH_EVAL_DIRECT_TEST_BIN=$binDir\tracedecay-search-eval-direct.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append
- name: Run Windows tests
shell: pwsh
# --extract-to must be the workspace so target/ lands at the same
# absolute path as on the build job: integration tests bake
# env!("CARGO_BIN_EXE_tracedecay") into the binaries at compile time.
run: >-
cargo-nextest nextest run --profile ci
--archive-file "$env:RUNNER_TEMP/nextest-archive.tar.zst"
--extract-to "$env:GITHUB_WORKSPACE"
--workspace-remap "$env:GITHUB_WORKSPACE"
--partition slice:${{ matrix.partition }}/5
--test-threads num-cpus --status-level slow
- name: Clean abandoned Windows test children
if: always()
shell: pwsh
run: |
$workspace = (Resolve-Path $env:GITHUB_WORKSPACE).Path
$all = Get-CimInstance Win32_Process
$liveProcessIds = [System.Collections.Generic.HashSet[uint32]]::new()
foreach ($process in $all) {
[void]$liveProcessIds.Add([uint32]$process.ProcessId)
}
$stale = foreach ($process in $all) {
if ($process.Name -ne "tracedecay.exe") {
continue
}
if (-not $process.CommandLine) {
continue
}
$inCurrentWorkspace = $process.CommandLine.IndexOf($workspace, [StringComparison]::OrdinalIgnoreCase) -ge 0
if (-not $inCurrentWorkspace) {
continue
}
if (-not $liveProcessIds.Contains([uint32]$process.ParentProcessId)) {
$process
}
}
foreach ($process in $stale) {
Write-Host "Stopping abandoned tracedecay child pid=$($process.ProcessId)"
Stop-Process -Id $process.ProcessId -Force -ErrorAction SilentlyContinue
}
- name: Upload Windows nextest report
if: always()
uses: actions/upload-artifact@v4
with:
name: windows-nextest-junit-${{ matrix.partition }}
path: target/nextest/ci/junit.xml
if-no-files-found: ignore
retention-days: 7
windows-test:
name: Test Windows
if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }}
needs: [windows-test-shard, scope-gate]
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Check Windows shards
run: |
if [ "${{ needs.windows-test-shard.result }}" != "success" ]; then
echo "Windows shard result: ${{ needs.windows-test-shard.result }}"
exit 1
fi
echo "All Windows shards passed."
clippy:
name: Clippy
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
# The free 4-vCPU Arm runner (Cobalt 100), as a measured comparison
# against the x64 `ubuntu-latest` this lane took 19 minutes on: same
# workload, same cache shape, different core. Public repositories get it
# at no cost; the result decides whether the partitioned test jobs move.
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@stable
with:
components: clippy
- uses: ./.github/actions/setup-linux-mold
- name: Cache Rust build outputs
uses: Swatinem/rust-cache@v2
with:
shared-key: ci-clippy-full-${{ runner.os }}-${{ runner.arch }}
cache-on-failure: true
- name: Run blocking Clippy policy
run: cargo clippy --workspace --all-targets --locked -- -D warnings
- name: Check lean build
run: cargo check -p tracedecay --no-default-features --locked
# Compile-only gate for feature graphs beyond the default test-lane
# selections. `test-transport` acceptance suites (`mcp_suite`,
# `transport_acceptance_suite`, `host_journeys_suite`, `work_loop_journey`)
# now run as the `root-transport` Linux partition; this job still
# `cargo check`s the full `--features test-transport` / hotpath graphs so
# non-selected targets cannot rot until release `--all-features`.
#
# Its own job, not a tail on the Linux build: these are dev-profile check
# builds, a graph the perf-profile test lanes share nothing with, so there
# is no compiled tree to reuse and nothing to gain from waiting for one.
# As the last steps of the Linux test job they ran only after a green
# suite — never once on this branch — and would have held the Linux shards
# back by their own duration. Here the verdict lands independently at
# about clippy's cost: the cold `cargo clippy --workspace --all-targets`
# took 13.7 min (run 34231734416), the second check re-resolves only the
# crates the hotpath features touch (clippy's lean re-check: 2.7 min).
# Feature wiring does not vary by host, so one OS carries the gate. No
# cache: it is off every lane's critical path, and another lineage would
# compete for the 10 GB budget the test lanes' dependency caches are
# already evicted from.
feature-gates:
name: Feature gates
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@stable
- uses: ./.github/actions/setup-linux-mold
- name: Check feature-gated surfaces compile
run: |
cargo check --workspace --all-targets --features test-transport --locked
cargo check --workspace --all-targets --features hotpath,hotpath-mcp --locked
fmt:
name: Format
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
# Judge formatting with the toolchain `rust-toolchain.toml` pins, not
# whatever `stable` is on the runner. rustfmt's output changes between
# releases — 1.98.1 breaks a `.unwrap_or_else(|err| { ... })` chain that
# 1.97.1 keeps inline — so a `stable` rustfmt rejects a tree the pinned
# rustfmt developers run formats exactly, and every contributor sees a
# red lane they cannot reproduce. `dtolnay/rust-toolchain@stable` also
# exports `RUSTUP_TOOLCHAIN`, which overrides the toolchain file rather
# than deferring to it. Installing through rustup with no override keeps
# the pin in one place: the toolchain file, components included.
- name: Install the repository's pinned toolchain
run: rustup show active-toolchain || rustup toolchain install
- run: cargo fmt --all -- --check
dashboard:
name: Dashboard
needs: scope-gate
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-latest
timeout-minutes: 25
steps:
- uses: actions/checkout@v7
- uses: actions/setup-node@v6
with:
node-version: "22"
cache: npm
cache-dependency-path: dashboard/package-lock.json
- name: Install dashboard dependencies
working-directory: dashboard
run: npm ci
- name: Build dashboard assets
working-directory: dashboard
run: npm run build
- name: Run dashboard unit tests
working-directory: dashboard
run: npm test
# The dashboard is one rsbuild app emitting `app-dist/` (the legacy
# shell/holographic/lcm/graph/savings bundle split no longer exists), so
# the bundle validator is the artifact authority and the determinism
# check hashes whatever the build actually produced.
- name: Verify embedded dist artifacts
working-directory: dashboard
run: |
set -euo pipefail
python3 ../scripts/check-dashboard-bundle.py app-dist
(cd app-dist && find . -type f | sort | xargs sha256sum) > /tmp/dashboard-dist.sha
npm run build
(cd app-dist && find . -type f | sort | xargs sha256sum) | diff /tmp/dashboard-dist.sha -
hermes-integration:
name: Hermes integration (stock)
needs: [scope-gate, debug-cli]
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
env:
# Stock (upstream) Hermes pin — the generated plugin must keep working
# against this exact upstream commit. Bump deliberately after rerunning
# scripts/hermes_stock_integration.sh against the new ref locally.
HERMES_UPSTREAM_REPO: https://github.com/NousResearch/hermes-agent.git
HERMES_UPSTREAM_REF: 9dd9ef0ec99a87f078f7272b4323df5440b4b3f9
steps:
- uses: actions/checkout@v7
- name: Install ast-grep
uses: ./.github/actions/install-ast-grep
with:
version: ${{ env.AST_GREP_VERSION }}
- name: Fetch the debug CLI built once for this run
uses: actions/download-artifact@v4
with:
name: debug-tracedecay-cli
path: target/debug
- name: Restore the executable bit the artifact store drops
run: chmod +x target/debug/tracedecay
- name: Run generated-plugin unit checks (no Hermes required)
run: python3 scripts/hermes_plugin_unit_check.py
- name: Clone stock Hermes at pinned ref
run: |
set -euo pipefail
git init -q "$RUNNER_TEMP/hermes-upstream"
git -C "$RUNNER_TEMP/hermes-upstream" remote add origin "$HERMES_UPSTREAM_REPO"
git -C "$RUNNER_TEMP/hermes-upstream" fetch --depth 1 origin "$HERMES_UPSTREAM_REF"
git -C "$RUNNER_TEMP/hermes-upstream" checkout --detach FETCH_HEAD
- uses: astral-sh/setup-uv@v8.2.0
with:
enable-cache: true
cache-dependency-glob: ${{ runner.temp }}/hermes-upstream/uv.lock
cache-suffix: hermes-${{ env.HERMES_UPSTREAM_REF }}
- name: Set up stock Hermes environment
working-directory: ${{ runner.temp }}/hermes-upstream
run: uv sync --frozen --no-dev
- name: Run stock Hermes integration checks
env:
TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay
HERMES_UPSTREAM_DIR: ${{ runner.temp }}/hermes-upstream
run: scripts/hermes_stock_integration.sh
host-stock-integration:
name: Claude Code + OpenCode integration (stock)
needs: [scope-gate, debug-cli]
if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }}
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
env:
# Stock host pins — the installed bundle must keep working against these
# exact host releases. Bump deliberately after rerunning
# scripts/claude_stock_integration.sh and
# scripts/opencode_stock_integration.sh against the new versions locally.
CLAUDE_CODE_VERSION: 2.1.224
OPENCODE_VERSION: 1.18.4
steps:
- uses: actions/checkout@v7
- name: Fetch the debug CLI built once for this run
uses: actions/download-artifact@v4
with:
name: debug-tracedecay-cli
path: target/debug
- name: Restore the executable bit the artifact store drops
run: chmod +x target/debug/tracedecay
- name: Install stock Claude Code and OpenCode at pinned versions
run: |
npm install --global "@anthropic-ai/claude-code@${CLAUDE_CODE_VERSION}" "opencode-ai@${OPENCODE_VERSION}"
claude --version
opencode --version
- name: Run stock Claude Code integration checks
env:
TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay
run: scripts/claude_stock_integration.sh
- name: Run stock OpenCode integration checks
env:
TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay
run: scripts/opencode_stock_integration.sh