Repository navigation
test(code-index): assert observed refusal and accepted values #6977
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: CI | |
| on: | |
| push: | |
| branches: [master, feature/holographic-memory, codex/tracedecay-total-redesign-plan-reopened] | |
| pull_request: | |
| branches: ['**'] | |
| permissions: | |
| contents: read | |
| # One run per pull request (or per pushed ref). A newer push to a pull | |
| # request supersedes the run in flight: its verdict is for a head nobody can | |
| # merge any more, and letting it finish only holds runners the current head is | |
| # waiting for. Keeping master-bound runs to completion did not buy a verdict | |
| # per tip either: GitHub holds one pending run per group, so while a 100-min | |
| # run finished, every intermediate push was queued and then cancelled by the | |
| # next (runs 34311300344, 34312118984, 34313237349 never scheduled a job) and | |
| # only the head at the moment the old run ended got a verdict, ~100 min late. | |
| # `push` runs are never cancelled. | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} | |
| cancel-in-progress: ${{ github.event_name == 'pull_request' }} | |
| env: | |
| CARGO_TERM_COLOR: always | |
| CARGO_INCREMENTAL: "0" | |
| CARGO_PROFILE_DEV_DEBUG: "0" | |
| CARGO_PROFILE_TEST_DEBUG: "0" | |
| AST_GREP_VERSION: "0.44.0" | |
| jobs: | |
| # Single source of truth for the fork/base-branch guard: heavy jobs run on | |
| # pushes, on PRs from this repository, and on fork PRs that target an | |
| # integration branch. Guarded jobs `needs:`-check this job's output instead | |
| # of each repeating the expression, so the branch list lives in one place. | |
| scope-gate: | |
| name: Scope gate | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 5 | |
| outputs: | |
| run-heavy: ${{ steps.decide.outputs.run-heavy }} | |
| linux-partitions: ${{ steps.linux-partitions.outputs.matrix }} | |
| macos-groups: ${{ steps.macos-groups.outputs.matrix }} | |
| steps: | |
| - name: Decide whether guarded jobs run | |
| id: decide | |
| run: echo "run-heavy=${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository || contains(fromJSON('["master","feature/holographic-memory"]'), github.event.pull_request.base.ref) }}" >> "$GITHUB_OUTPUT" | |
| # The Linux and macOS test matrices come from the same manifest the | |
| # partition jobs select their targets from, so a partition cannot exist | |
| # without a job or a job without a partition. This needs the checkout | |
| # and nothing else. | |
| - uses: actions/checkout@v7 | |
| with: | |
| sparse-checkout: | | |
| .github/linux-test-partitions.json | |
| scripts/linux-test-partitions.py | |
| sparse-checkout-cone-mode: false | |
| - name: Derive the Linux test matrix | |
| id: linux-partitions | |
| run: echo "matrix=$(python3 scripts/linux-test-partitions.py matrix)" >> "$GITHUB_OUTPUT" | |
| - name: Derive the macOS test matrix | |
| id: macos-groups | |
| run: echo "matrix=$(python3 scripts/linux-test-partitions.py macos-matrix)" >> "$GITHUB_OUTPUT" | |
| # Exercises the checked-out PR head through the CLI against an isolated | |
| # daemon (scripts/ci-pr-dogfood-smoke.sh). The binary is the one `debug-cli` | |
| # builds from the merge ref: the smoke only needs an executable `tracedecay`, | |
| # and building a second one from the head cost a cold | |
| # `cargo build -p tracedecay-cli` that never fit this job's budget (runs | |
| # 34217290139 and 34231734416: the build was cancelled after 29 minutes | |
| # while `debug-cli` finished the same build in 30-31 minutes and published | |
| # it). | |
| pr-dogfood: | |
| name: PR checkout dogfood | |
| needs: [scope-gate, debug-cli] | |
| if: ${{ github.event_name == 'pull_request' && needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 30 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| fetch-depth: 0 | |
| persist-credentials: false | |
| ref: ${{ github.event.pull_request.head.sha }} | |
| - name: Fetch the debug CLI built once for this run | |
| uses: actions/download-artifact@v4 | |
| with: | |
| name: debug-tracedecay-cli | |
| path: target/debug | |
| - name: Restore the executable bit the artifact store drops | |
| run: chmod +x target/debug/tracedecay | |
| - name: Test dogfood harnesses | |
| run: | | |
| python3 scripts/test-check-pr-dogfood-output.py | |
| python3 scripts/test-efficiency-scorecard.py | |
| python3 scripts/test-pr-dogfood-portability.py | |
| - name: Initialize checkout and exercise CLI tools | |
| run: scripts/ci-pr-dogfood-smoke.sh | |
| env: | |
| TRACEDECAY_BIN: target/debug/tracedecay | |
| TRACEDECAY_DOGFOOD_BASE_REF: ${{ github.event.pull_request.base.sha }} | |
| TRACEDECAY_DOGFOOD_HEAD_REF: ${{ github.event.pull_request.head.sha }} | |
| TRACEDECAY_DOGFOOD_BASE_BRANCH: ${{ github.event.pull_request.base.ref }} | |
| TRACEDECAY_DOGFOOD_HEAD_BRANCH: ${{ github.event.pull_request.head.ref }} | |
| # One dev-profile `tracedecay` binary for every job that only exercises the | |
| # built CLI (`pr-dogfood`, Hermes, Claude Code, OpenCode). Each of those | |
| # jobs used to compile the whole workspace itself behind its own cache key, | |
| # so a cold cache cost the run several 30-minute builds instead of one and | |
| # tripped their timeouts. | |
| debug-cli: | |
| name: Build debug CLI | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 45 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: dtolnay/rust-toolchain@stable | |
| - uses: ./.github/actions/setup-linux-mold | |
| # This is the workflow's only dev-profile build, so `ci-dev-*` has one | |
| # writer. rust-cache stores dependency artifacts only. The repository | |
| # cache is capped at 10 GB and was full of per-job copies of the same | |
| # dependencies (`pr-dogfood` and `dashboard` each built this graph too), | |
| # which evicted the test lanes' caches between runs and made every test | |
| # run a cold build. Only the lockfile-keyed dependency cache is kept: | |
| # the workspace crates are exactly what a push changes, so caching | |
| # their outputs per run re-saved ~1 GB per push for a restore that hit | |
| # nothing. | |
| - name: Cache Rust build outputs | |
| uses: Swatinem/rust-cache@v2 | |
| with: | |
| shared-key: ci-dev-${{ runner.os }}-${{ runner.arch }} | |
| cache-on-failure: true | |
| - name: Build tracedecay binary | |
| run: cargo build -p tracedecay-cli --bin tracedecay --locked | |
| - name: Publish the binary for the host integration jobs | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: debug-tracedecay-cli | |
| path: target/debug/tracedecay | |
| if-no-files-found: error | |
| retention-days: 1 | |
| commit-messages: | |
| name: Commit Messages | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| fetch-depth: 0 | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| cache: npm | |
| - name: Install commit message lint dependencies | |
| run: npm ci | |
| - name: Reject tracked ignored files | |
| run: scripts/check-release-pr-integrity.sh HEAD HEAD | |
| - name: Test bounded commit range linting | |
| run: python3 scripts/test-lint-commit-range.py | |
| - name: Validate PR commit messages | |
| if: github.event_name == 'pull_request' | |
| env: | |
| BASE_SHA: ${{ github.event.pull_request.base.sha }} | |
| HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| run: node scripts/lint-commit-range.mjs --repository "$PWD" "$BASE_SHA" "$HEAD_SHA" | |
| - name: Validate pushed commit messages | |
| if: github.event_name == 'push' | |
| env: | |
| BEFORE_SHA: ${{ github.event.before }} | |
| HEAD_SHA: ${{ github.sha }} | |
| run: | | |
| set -euo pipefail | |
| if [ "$BEFORE_SHA" = "0000000000000000000000000000000000000000" ]; then | |
| if git rev-parse "${HEAD_SHA}^" >/dev/null 2>&1; then | |
| base_sha="${HEAD_SHA}^" | |
| else | |
| git show --no-patch --format=%B "$HEAD_SHA" | npm run lint:commit -- | |
| exit 0 | |
| fi | |
| else | |
| base_sha="$BEFORE_SHA" | |
| fi | |
| node scripts/lint-commit-range.mjs --repository "$PWD" "$base_sha" "$HEAD_SHA" | |
| release-version-drift: | |
| name: Release Version Drift | |
| if: github.event_name == 'pull_request' && !startsWith(github.head_ref, 'release-please--') | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-python@v6 | |
| with: | |
| python-version: "3.12" | |
| - name: Test release version drift guard | |
| run: bash tests/release_drift_check_test.sh | |
| # The workflow-contract pin scripts were replaced by the safety-guard | |
| # assertions (f3b065545) and the dogfood surface was retired (b73a8a396); | |
| # this job keeps the guards that protect the shipped release flow. | |
| - name: Check release safety guards | |
| run: bash tests/release_safety_test.sh | |
| - name: Test GitHub release installer | |
| run: bash tests/install_script_test.sh | |
| # These two sat unwired: they are the tests for | |
| # `scripts/check-release-pr-integrity.sh` and | |
| # `scripts/check-dashboard-bundle.py`, both of which CI does run. So the | |
| # guards were enforced while the tests proving the guards work were not, | |
| # which is the shape that let a `test-transport` target rot until it | |
| # stopped compiling. Three of the five `tests/*_test.sh` gates were wired; | |
| # these are the other two. | |
| - name: Test release PR integrity guard | |
| run: bash tests/release_pr_integrity_test.sh | |
| - name: Test dashboard bundle check | |
| run: bash tests/dashboard_bundle_check_test.sh | |
| - name: Check release version drift | |
| run: scripts/check-release-drift.sh | |
| env: | |
| GITHUB_TOKEN: ${{ github.token }} | |
| # macOS compiles and runs the suite as groups of the Linux partitions | |
| # (`macos_groups` in .github/linux-test-partitions.json), one hosted 3-vCPU | |
| # `macos-14` job per group, so the lane's wall time is the slowest group | |
| # instead of the whole workspace. The single job this replaces built | |
| # `--workspace --bins --tests` in 80.2 min and ran the suite in 19.9 min | |
| # (run 34303181887; run 34309279328 built for 74.9 min and was cancelled in | |
| # its tests at 85 min), about 80 min after every other lane had reported, | |
| # so it was the critical path of every pull request run. rust-cache carries | |
| # the dependency graph only; the workspace crates compile fresh every run, | |
| # and on three cores with ld64 that build is the whole cost. | |
| # | |
| # A group runs each of its partitions in turn against one target directory | |
| # with the selection `linux-test-partition` resolves for it, so | |
| # `root-sessions` and `root-journeys`, which select the same packages with | |
| # the same feature, compile the chain beneath the root crate once and the | |
| # second adds only its own suites. The grouping balances the warm Linux | |
| # measurements (run 34309279328, compile + tests in minutes): root-lib | |
| # 14.7 + 7.5 alone, since its root chain, library test target and 1675 | |
| # tests are the floor no split lowers; root-sessions 15.3 + 3.1 with | |
| # root-journeys, which adds 2.7 + 2.7 on the shared resolution; runtime | |
| # 15.6 + 2.4 with core-storage 10.4 + 1.9; core-contracts 13.9 + 0.6 with | |
| # root-dashboard-api 12.7 + 0.8. Every pairing that puts a second partition | |
| # beside a root group is heavier than the heaviest of these, and no two of | |
| # the other four share a feature resolution, so grouping them differently | |
| # saves no compile. At the 3-vCPU floor (compile × 4/3, tests as measured) | |
| # the groups come to 28, 31, 40 and 38 min; at the 2× the one macOS | |
| # measurement showed (the chain beneath the root crate compiled in 19.0 | |
| # min against 9.3 on Arm in the runs above) to 38, 43, 57 and 56. The | |
| # budgets in the manifest take the 2×, plus the cold-dependency allowance | |
| # and headroom the Linux budgets carry, until a hosted run measures the | |
| # groups. The account admits five concurrent macOS jobs; a run takes at | |
| # most four (`MACOS_GROUP_CAP`) so another run's macOS jobs can start. | |
| macos-test-partition: | |
| name: Test macOS ${{ matrix.group }} | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: macos-14 | |
| # Each group's budget is set in the manifest beside the group, with the | |
| # measurement it rests on. | |
| timeout-minutes: ${{ matrix.timeout }} | |
| strategy: | |
| fail-fast: false | |
| matrix: ${{ fromJSON(needs.scope-gate.outputs.macos-groups) }} | |
| steps: | |
| # Full history: the search-quality workload fixture pins a | |
| # `source_repository_commit`, and `validate_source_bindings` resolves that | |
| # object out of this checkout to prove the checked-in corpus really is the | |
| # product source at that commit. A depth-1 checkout carries only the tip, | |
| # so every candidate_output/report test fails with "resolve fixture source | |
| # commit: An object with id ... could not be found". | |
| - uses: actions/checkout@v7 | |
| with: | |
| fetch-depth: 0 | |
| # `rust-analyzer` is a test dependency of this lane, not a convenience. | |
| # `runtime_surface_acceptance::production_lsp_negotiates_and_projects_\ | |
| # canonical_context` asserts that a routed analyzer negotiates the | |
| # standard methods the client declared, and the upstream `initialize` | |
| # response is the only authority for those. Without the component the | |
| # runner still has rustup's `rust-analyzer` proxy on PATH, so the daemon | |
| # routes the language and only the spawn fails — the test would then | |
| # measure a missing component instead of the negotiation contract. This | |
| # action exports `RUSTUP_TOOLCHAIN`, so the component has to be installed | |
| # here rather than through `rust-toolchain.toml`, which it overrides. | |
| - uses: dtolnay/rust-toolchain@stable | |
| with: | |
| components: rust-analyzer | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| - uses: actions/setup-python@v6 | |
| with: | |
| python-version: "3.12" | |
| - name: Install ast-grep | |
| uses: ./.github/actions/install-ast-grep | |
| with: | |
| version: ${{ env.AST_GREP_VERSION }} | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@nextest | |
| # rust-cache carries the dependency graph, keyed by toolchain and | |
| # lockfile; the workspace crates compile fresh every run. No per-run | |
| # compiler store: the perf-profile test graph wrote 15.7 GB of outputs | |
| # per build on macOS, which no store inside the repository's 10 GB | |
| # cache budget can hold. One writer keeps the lane at one entry: | |
| # `root-suites` resolves the widest graph (the root crate with its | |
| # dev-dependencies, the CLI and the evaluator), so its save covers | |
| # `root-lib` exactly; the other groups unify dependency features | |
| # differently and recompile a few small dependency units against it | |
| # every run, as the lower Linux partitions do. Restored before the first | |
| # `cargo metadata` below so the registry index comes from the cache too. | |
| - name: Cache Rust build outputs | |
| uses: Swatinem/rust-cache@v2 | |
| with: | |
| shared-key: ci-test-full-${{ runner.os }} | |
| save-if: ${{ matrix.group == 'root-suites' }} | |
| cache-on-failure: true | |
| # Every group job re-proves the cover and the grouping before it | |
| # compiles anything: a gap found here fails the lane rather than passing | |
| # it with a target nobody ran. | |
| - name: Check the partitions cover every test target | |
| run: | | |
| python3 scripts/test-linux-test-partitions.py | |
| python3 scripts/linux-test-partitions.py check | |
| # Each partition in turn, as `linux-test-partition` runs it: the | |
| # executables its suites spawn are built into the same resolution first | |
| # (`build-args`; nothing for a partition whose tests spawn nothing), | |
| # then nextest under the `ci` policy and the perf cargo profile with | |
| # `--no-tests=fail`. One target directory for the group, so a later | |
| # partition's shared units are fingerprint hits. Every partition runs | |
| # even after an earlier one fails — as nextest's `ci` policy runs every | |
| # test — and the step fails if any did. Every nextest run rewrites the | |
| # `ci` profile's junit.xml, so each partition's report is moved under | |
| # its own name before the next runs. The per-partition wall times this | |
| # prints are the measurements the manifest's macOS budgets wait for. | |
| - name: Run the group's partitions | |
| env: | |
| PARTITIONS: ${{ matrix.partitions }} | |
| run: | | |
| set -uo pipefail | |
| run_partition() { | |
| local partition="$1" build selection | |
| build="$(python3 scripts/linux-test-partitions.py build-args "$partition")" || return 1 | |
| selection="$(python3 scripts/linux-test-partitions.py cargo-args "$partition")" || return 1 | |
| if [ -n "$build" ]; then | |
| eval cargo build --locked --profile perf "$build" || return 1 | |
| fi | |
| eval cargo nextest run --profile ci --cargo-profile perf --locked "$selection" --no-tests=fail | |
| } | |
| read -ra partitions <<< "$PARTITIONS" | |
| mkdir -p target/nextest/macos | |
| failed="" | |
| for partition in "${partitions[@]}"; do | |
| echo "::group::Test partition ${partition}" | |
| started=$SECONDS | |
| if run_partition "$partition"; then | |
| echo "${partition}: passed in $(( SECONDS - started )) s" | |
| else | |
| failed="${failed} ${partition}" | |
| echo "${partition}: failed after $(( SECONDS - started )) s" | |
| fi | |
| [ ! -f target/nextest/ci/junit.xml ] || mv target/nextest/ci/junit.xml "target/nextest/macos/${partition}.xml" | |
| echo "::endgroup::" | |
| done | |
| if [ -n "$failed" ]; then | |
| echo "::error::Failed partitions:${failed}" | |
| exit 1 | |
| fi | |
| # `macos-test` folds the group reports back into one report. | |
| - name: Upload macOS group nextest report | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: nextest-junit-macOS-${{ matrix.group }} | |
| path: target/nextest/macos/ | |
| if-no-files-found: ignore | |
| retention-days: 1 | |
| # The one `Test macOS` verdict, as when a single job carried the suite: | |
| # every group must pass. It also folds the group reports back into the | |
| # `nextest-junit-macOS` artifact the single job published, one directory | |
| # per group holding a junit.xml per partition. | |
| macos-test: | |
| name: Test macOS | |
| if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }} | |
| needs: [macos-test-partition, scope-gate] | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| - name: Merge the group reports into the macOS nextest report | |
| if: ${{ needs.macos-test-partition.result != 'skipped' }} | |
| uses: actions/upload-artifact/merge@v4 | |
| with: | |
| name: nextest-junit-macOS | |
| pattern: nextest-junit-macOS-* | |
| separate-directories: true | |
| retention-days: 7 | |
| - name: Check macOS groups | |
| if: ${{ !cancelled() }} | |
| run: | | |
| if [ "${{ needs.macos-test-partition.result }}" != "success" ]; then | |
| echo "macOS group result: ${{ needs.macos-test-partition.result }}" | |
| exit 1 | |
| fi | |
| echo "All macOS groups passed." | |
| # Linux compiles and runs the suite as parallel partitions of the test | |
| # targets (.github/linux-test-partitions.json), one hosted 4-vCPU job each, | |
| # so the lane's wall time is the slowest partition instead of the whole | |
| # workspace. The single job that preceded it (run 34278967966, warm | |
| # dependency cache, four compile slots) spent 55.5 min compiling: 32 min of | |
| # workspace libraries beneath the root crate, then the root library | |
| # (~8 min) and its library test target (~15.5 min), a serial chain that no | |
| # job count shortens; the other 300-odd test units fit in the slots beside | |
| # it. Tests then took 22.5 min, the dashboard's `test-transport` rebuild | |
| # and run 11 min, and the hotpath parity rebuild ~16 min. Partitioned, the | |
| # root chain is one job (`root-lib`: 19.7 min compile + 5.2 min tests, 25.9 | |
| # min in run 34296614024, the first partitioned run) while every other | |
| # partition finishes in 17–27 min, so the lane's critical path drops from | |
| # ~105 min to the slowest partition; the parity rebuild runs beside the | |
| # partitions as `hotpath-parity` rather than after the root chain. Building | |
| # the graph once and sharding only the test execution (the Windows lane's | |
| # shape) keeps the whole 55 min compile plus an archive round trip on the | |
| # path and then the dashboard rebuild and parity on the same job: ~88 min, | |
| # or ~70 with both moved elsewhere. The partitions pay for the shorter path | |
| # with duplicated compile — each root partition rebuilds the library chain | |
| # beneath the root crate, about 200 core-minutes per run across the seven | |
| # jobs — which is free on hosted runners where only wall time and the | |
| # 20-job concurrency cap count. A cold dependency cache adds ~6 min to every | |
| # job alike (the dependencies compile at opt-level 0 in ~17 core-minutes). | |
| # The floor no partitioning reaches is the root chain itself: library | |
| # chain + root lib + root lib-test + its tests, the ~26 min above. | |
| # | |
| # The partitions are cargo selections, not nextest filters: `-p` plus | |
| # `--lib` / `--test <name>` / `--bins` decide what compiles, and a filterset | |
| # would compile everything and skip at run time. The three root partitions | |
| # share one package selection (`tracedecay`, `tracedecay-cli`, | |
| # `tracedecay-search-eval`) with the root fixture feature, so they resolve | |
| # one identical dependency graph and every cargo invocation inside a job | |
| # (the test build, the executables the suites spawn) is a cache hit against | |
| # it. `scripts/linux-test-partitions.py check` proves, from `cargo | |
| # metadata`, that every test target in the workspace is selected by exactly | |
| # one partition or listed under `not_run` with a reason, so a new crate or | |
| # suite cannot fall out of the lane silently; `scope-gate` derives the | |
| # matrix from the same manifest. | |
| linux-test-partition: | |
| name: Test Linux ${{ matrix.partition }} | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| # Each partition's budget is its measured wall time plus headroom, set in | |
| # the manifest beside the selection it covers. | |
| timeout-minutes: ${{ matrix.timeout }} | |
| strategy: | |
| fail-fast: false | |
| matrix: ${{ fromJSON(needs.scope-gate.outputs.linux-partitions) }} | |
| steps: | |
| # Full history: the search-quality workload fixture pins a | |
| # `source_repository_commit`, and `validate_source_bindings` resolves that | |
| # object out of this checkout to prove the checked-in corpus really is the | |
| # product source at that commit. A depth-1 checkout carries only the tip, | |
| # so every candidate_output/report test fails with "resolve fixture source | |
| # commit: An object with id ... could not be found". | |
| - uses: actions/checkout@v7 | |
| with: | |
| fetch-depth: 0 | |
| # `rust-analyzer` is a test dependency of this lane, not a convenience. | |
| # `runtime_surface_acceptance::production_lsp_negotiates_and_projects_\ | |
| # canonical_context` asserts that a routed analyzer negotiates the | |
| # standard methods the client declared, and the upstream `initialize` | |
| # response is the only authority for those. Without the component the | |
| # runner still has rustup's `rust-analyzer` proxy on PATH, so the daemon | |
| # routes the language and only the spawn fails — the test would then | |
| # measure a missing component instead of the negotiation contract. This | |
| # action exports `RUSTUP_TOOLCHAIN`, so the component has to be installed | |
| # here rather than through `rust-toolchain.toml`, which it overrides. | |
| - uses: dtolnay/rust-toolchain@stable | |
| with: | |
| components: rust-analyzer | |
| - uses: ./.github/actions/setup-linux-mold | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| - uses: actions/setup-python@v6 | |
| with: | |
| python-version: "3.12" | |
| - name: Install ast-grep | |
| uses: ./.github/actions/install-ast-grep | |
| with: | |
| version: ${{ env.AST_GREP_VERSION }} | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@nextest | |
| # One dependency cache for the lane, as before the split. rust-cache | |
| # stores the dependency graph only, keyed by toolchain and lockfile; the | |
| # workspace crates compile fresh every run. A single writer keeps the | |
| # lane at one entry in the repository's 10 GB cache: `root-journeys` | |
| # resolves the widest graph (the root crate with its dev-dependencies, | |
| # the CLI and the evaluator), so its save covers the other root | |
| # partitions exactly. The three lower partitions unify dependency | |
| # features differently and each recompile ~30 small dependency units | |
| # (syn, serde_json, futures, chrono, ...) against it every run — a few | |
| # minutes inside jobs that are far from the critical path, cheaper than | |
| # three more cache entries. Restored before the first `cargo metadata` | |
| # below so the registry index comes from the cache too. | |
| - name: Cache Rust build outputs | |
| uses: Swatinem/rust-cache@v2 | |
| with: | |
| shared-key: ci-test-full-Linux-${{ runner.arch }} | |
| save-if: ${{ matrix.partition == 'root-journeys' }} | |
| cache-on-failure: true | |
| # Every partition job re-proves the cover before it compiles anything: | |
| # the manifest is consumed here, and a gap found here fails the lane | |
| # rather than passing it with a target nobody ran. | |
| - name: Check the partitions cover every test target | |
| run: | | |
| python3 scripts/test-linux-test-partitions.py | |
| python3 scripts/linux-test-partitions.py check | |
| - name: Resolve this partition's cargo selection | |
| id: selection | |
| run: | | |
| { | |
| echo "test=$(python3 scripts/linux-test-partitions.py cargo-args '${{ matrix.partition }}')" | |
| echo "build=$(python3 scripts/linux-test-partitions.py build-args '${{ matrix.partition }}')" | |
| } >> "$GITHUB_OUTPUT" | |
| # Suites that spawn the CLI (`tests/common::tracedecay_bin`) or the | |
| # search-eval evaluators (`search_eval_bin`) resolve them out of the | |
| # profile directory, and cargo links a package's bins for a test build | |
| # only when that package's own integration tests are selected. This | |
| # builds the partition's selection plus those executables in one | |
| # resolution: the test targets keep dev-dependencies in the graph, so | |
| # the nextest build below is a cache hit, and the selection includes | |
| # `tracedecay-search-eval` so the evaluator resolves | |
| # `tracedecay-code-index` with its language tiers (a lone | |
| # `-p tracedecay-search-eval` build yields one that cannot parse Rust). | |
| # The `tracedecay-host-cli-fixture` example is the host fixture the CLI | |
| # suites drive. Partitions whose tests spawn nothing skip this. | |
| - name: Build the executables the suites spawn | |
| if: ${{ steps.selection.outputs.build != '' }} | |
| run: cargo build --locked --profile perf ${{ steps.selection.outputs.build }} | |
| # nextest `ci` policy (.config/nextest.toml: fail-fast off, one retry | |
| # that still fails flaky results, 8 threads, slow-timeout termination) | |
| # and the perf cargo profile, as `cargo test-ci`; the selection replaces | |
| # the alias's `--workspace`. `--no-tests=fail` so a selection that | |
| # resolves to nothing cannot report green. | |
| - name: Run tests | |
| run: cargo nextest run --profile ci --cargo-profile perf --locked ${{ steps.selection.outputs.test }} --no-tests=fail | |
| # `linux-test` folds the partition reports back into one report. | |
| - name: Upload Linux partition nextest report | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: nextest-junit-Linux-${{ matrix.partition }} | |
| path: target/nextest/ci/junit.xml | |
| if-no-files-found: ignore | |
| retention-days: 1 | |
| # The controlled-workload hotpath parity gate: `tracedecay-search-eval`'s | |
| # `emit_controlled_workload_reports` example, built once with the hotpath | |
| # profiler compiled out and once with `controlled-workload-hotpath` | |
| # (`hotpath/hotpath`) on, must produce byte-identical durable results for | |
| # the framed-log (private-fs) and cursor-parse (capture) workloads, from | |
| # provably distinct executables. `scripts/build-controlled-workload- | |
| # hotpath-helpers.py` builds the pair into target/controlled-workload- | |
| # hotpath/, and the `#[ignore]`d test in the crate's library test target | |
| # (`controlled_workloads::tests::hotpath_off_vs_on_durable_results_are_ | |
| # identical`) spawns them from there; `--run-ignored only` with | |
| # `--no-tests=fail` is the only way it runs, and nothing else — no | |
| # partition, artifact upload or junit merge — consumes the executables. | |
| # | |
| # Its own job rather than a tail on the `root-lib` partition, where it ran | |
| # after the root library's tests: the feature-on build recompiles every | |
| # workspace crate beneath the evaluator (all 21 depend on `hotpath`, so | |
| # the feature flip changes each one's metadata hash) plus the example, | |
| # ~10 min on top of the ~26 min root chain, which made that job the lane's | |
| # critical path. Here the graph beneath the evaluator compiles from the | |
| # checkout in both modes — a serial chain domain → contracts → | |
| # rusqlite-runtime → runtime-core → code-index → sessions → query → | |
| # search-eval → example of ~9 min per mode with four compile slots (cargo | |
| # `--timings`, -j4 on EPYC 7742 cores; the same pair of builds took 19.5 | |
| # min on a hosted runner in run 34138693824, when the single-job lane | |
| # still built them as their own resolution) — then the library test target | |
| # (24 s against the feature-off graph) and the test itself (under 1 s). | |
| # With setup, ~24 min hosted: beside the partitions and no longer than the | |
| # slowest of them (25.9 and 27.4 min in run 34296614024), so the `Test | |
| # Linux` verdict no longer waits for it. The property does not depend on | |
| # the platform (the Windows lane never provisioned it for that reason), so | |
| # a Linux runner carries it. | |
| # | |
| # Not a job in hotpath-coverage.yml: that workflow is path-filtered to | |
| # Rust inputs and never runs on `push`, while this is a gate on every pull | |
| # request head, and its feature-on slices (storage, sessions) do not cover | |
| # the query/code-index/runtime-core graph the evaluator needs; rust-cache | |
| # keeps dependency artifacts only, so no other job's compiled workspace | |
| # crates could be reused anyway. The evaluator alone (`-p | |
| # tracedecay-search-eval`) is the selection: the root partitions' wider | |
| # one would turn every language tier on in `tracedecay-code-extraction` | |
| # and `-code-index` for two builds that parse no source, and carry the root | |
| # fixture feature for a crate this job never compiles. No cache: under | |
| # this selection ~160 of the 420 registry crates (`syn`, `serde_core`, | |
| # `tokio` and their dependents) unify differently from the lane's | |
| # `ci-test-full-Linux` graph and would miss it, and the rest compile in | |
| # the slots the serial chain leaves idle (the cold build measured 8.8 min | |
| # against 9.1 for the warm feature-on one), so a restore | |
| # would buy about what it costs, while another lineage would compete for | |
| # the 10 GB budget the test lanes' dependency caches are already evicted | |
| # from. | |
| hotpath-parity: | |
| name: Hotpath parity | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| # 19.5 min for the two builds (hosted, above) + ~1.5 cold registry the | |
| # chain does not hide + ~1.5 setup + 0.5 test target + 0.5 run ≈ 24 min, | |
| # plus 25 % headroom, as for the partition budgets. | |
| timeout-minutes: 30 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: dtolnay/rust-toolchain@stable | |
| - uses: ./.github/actions/setup-linux-mold | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@nextest | |
| # Both modes under the evaluator's own selection; the helper copies | |
| # each result out of target/perf/examples immediately so the second | |
| # resolution cannot replace the first. | |
| - name: Build controlled-workload hotpath parity executables | |
| run: python3 scripts/build-controlled-workload-hotpath-helpers.py --profile perf | |
| # The library test target resolves the same feature-off graph as the | |
| # first build above, so only the test target itself compiles here. | |
| # nextest `ci` policy and the perf cargo profile, as the partitions. | |
| - name: Verify controlled-workload hotpath parity | |
| run: | | |
| cargo nextest run --profile ci --cargo-profile perf --locked -p tracedecay-search-eval --lib \ | |
| --run-ignored only \ | |
| -E 'test(=controlled_workloads::tests::hotpath_off_vs_on_durable_results_are_identical)' \ | |
| --no-tests=fail | |
| # The one `Test Linux` verdict, as when a single job carried the suite: | |
| # every partition and the hotpath parity gate must pass. It also folds the | |
| # partition reports back into the `nextest-junit-Linux` artifact the single | |
| # job published; the parity verdict is its step's exit status, as it was | |
| # inside the partition, and is not part of that report. | |
| linux-test: | |
| name: Test Linux | |
| if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }} | |
| needs: [linux-test-partition, hotpath-parity, scope-gate] | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| # Each partition's junit.xml keeps its name under its partition's | |
| # directory; the partition artifacts themselves expire after a day. | |
| - name: Merge the partition reports into the Linux nextest report | |
| if: ${{ needs.linux-test-partition.result != 'skipped' }} | |
| uses: actions/upload-artifact/merge@v4 | |
| with: | |
| name: nextest-junit-Linux | |
| pattern: nextest-junit-Linux-* | |
| separate-directories: true | |
| retention-days: 7 | |
| - name: Check Linux partitions and hotpath parity | |
| if: ${{ !cancelled() }} | |
| run: | | |
| if [ "${{ needs.linux-test-partition.result }}" != "success" ]; then | |
| echo "Linux partition result: ${{ needs.linux-test-partition.result }}" | |
| exit 1 | |
| fi | |
| if [ "${{ needs.hotpath-parity.result }}" != "success" ]; then | |
| echo "Hotpath parity result: ${{ needs.hotpath-parity.result }}" | |
| exit 1 | |
| fi | |
| echo "All Linux partitions and the hotpath parity gate passed." | |
| windows-build: | |
| name: Build Windows tests | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: windows-latest | |
| # Build every Windows test once, then shard execution from the archive. | |
| # Keep a bounded guard around the hosted build; Cargo timings are retained | |
| # below so a slow or failed archive can be attributed instead of treating | |
| # the timeout itself as the defect. Measured 2026-09-05 (run 33954670866, | |
| # no parity rebuild): portability compile 31 min, archive still compiling | |
| # the root crate's test targets at the 75-minute bound, so the archive | |
| # never completed and the compiler caches never seeded. The bound is | |
| # widened so one archive can finish and seed them; shrink it back once a | |
| # warm run's timings show the real cost. | |
| timeout-minutes: 120 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Tune Windows runner for build I/O | |
| uses: ./.github/actions/tune-windows-runner | |
| # Install only the toolchain pinned by rust-toolchain.toml. Installing | |
| # `stable` as well makes Swatinem/rust-cache include an unused compiler | |
| # in its environment key, fragmenting the cache whenever stable moves. | |
| - name: Install pinned toolchain | |
| shell: pwsh | |
| run: | | |
| rustup toolchain install | |
| rustup show active-toolchain | |
| - name: Use lld-link linker | |
| shell: pwsh | |
| run: | | |
| where.exe lld-link.exe | |
| lld-link.exe --version | |
| "CARGO_TARGET_X86_64_PC_WINDOWS_MSVC_LINKER=lld-link.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append | |
| # Node is only needed so build.rs can produce the embedded dashboard | |
| # dist assets (npm ci + npm run build) once, instead of once per shard. | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| cache: npm | |
| cache-dependency-path: dashboard/package-lock.json | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@nextest | |
| - name: Cache Windows Rust build outputs | |
| uses: Swatinem/rust-cache@v2 | |
| with: | |
| shared-key: ci-test-full-windows-msvc-lld | |
| cache-on-failure: true | |
| # Acceptance tests execute the ordinary evaluator binaries. Match the | |
| # archive's features and perf profile to reuse dependency artifacts; | |
| # nextest's archive.include carries the executables to every shard. | |
| # `--tests` keeps the binaries in the dev-dependency graph the archive | |
| # is built from; a `--bins`-only build resolves 60 dependencies with | |
| # different features and compiles them a second time (see the Linux | |
| # lane). | |
| - name: Build workspace binaries and tests for the Windows test lane | |
| shell: pwsh | |
| run: cargo build --workspace --bins --tests --locked --profile perf --features tracedecay/test-helpers | |
| # `--workspace`, not `-p tracedecay-cli`: the package selection decides | |
| # feature unification, and the narrower one recompiles the code-index | |
| # and extraction crates in a second configuration (see the Linux lane). | |
| - name: Build Windows host-CLI test fixture | |
| shell: pwsh | |
| run: cargo build --workspace --example tracedecay-host-cli-fixture --locked --profile perf --features tracedecay/test-helpers | |
| # The controlled-workload Hotpath parity helpers are provisioned and | |
| # verified by the `hotpath-parity` job. Building them here would put a | |
| # second, feature-on resolution of the search-eval dependency graph | |
| # inside this job's bound (18 of its 62 minutes before it was cancelled), | |
| # and the parity property does not depend on the platform. | |
| # Same selection as `cargo test-ci` (.cargo/config.toml); nextest has no | |
| # archive alias, so the arguments are spelled out here once. | |
| - name: Build nextest archive | |
| shell: pwsh | |
| run: cargo nextest archive --workspace --profile ci --locked --features tracedecay/test-helpers --cargo-profile perf --timings --archive-file "$env:RUNNER_TEMP/nextest-archive.tar.zst" | |
| - name: Upload Windows archive build timings | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: windows-nextest-build-timings | |
| path: target/cargo-timings/cargo-timing.html | |
| if-no-files-found: warn | |
| retention-days: 7 | |
| - name: Upload nextest archive | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: windows-nextest-archive | |
| path: ${{ runner.temp }}/nextest-archive.tar.zst | |
| retention-days: 1 | |
| compression-level: 0 | |
| # The archive already carries these (`archive.include` in | |
| # .config/nextest.toml), but nothing lands on a shard until the run | |
| # step extracts, so a missing evaluator can only surface as product | |
| # test failures. Publishing them separately lets each shard restore and | |
| # prove them *before* the suite starts. Same executables, from the | |
| # workspace/test-helpers build above: an evaluator built under a | |
| # narrower `-p tracedecay-search-eval` selection resolves | |
| # tracedecay-code-index without its default language tiers and cannot | |
| # parse Rust. | |
| - name: Upload Windows evaluator executables | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: windows-search-eval-bins | |
| path: | | |
| target/perf/tracedecay-search-eval.exe | |
| target/perf/tracedecay-search-eval-direct.exe | |
| if-no-files-found: error | |
| retention-days: 1 | |
| compression-level: 0 | |
| windows-test-shard: | |
| name: Test Windows shard ${{ matrix.partition }}/5 | |
| needs: windows-build | |
| runs-on: windows-latest | |
| timeout-minutes: 20 | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| partition: [1, 2, 3, 4, 5] | |
| steps: | |
| # Full history for the same reason as the Linux and macOS partition | |
| # jobs: the shards run the workspace suite remapped onto this checkout, | |
| # so the pinned search-quality fixture commit must be resolvable here too. | |
| - uses: actions/checkout@v7 | |
| with: | |
| fetch-depth: 0 | |
| - name: Tune Windows runner for test I/O | |
| uses: ./.github/actions/tune-windows-runner | |
| with: | |
| redirect-temp: "true" | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| - uses: actions/setup-python@v6 | |
| with: | |
| python-version: "3.12" | |
| - name: Install ast-grep | |
| uses: ./.github/actions/install-ast-grep | |
| with: | |
| version: ${{ env.AST_GREP_VERSION }} | |
| # The architecture tests exec `cargo metadata` at runtime, and they run | |
| # concurrently under nextest: without the pinned toolchain installed, | |
| # each test process's rustup shim races to self-install it and the | |
| # concurrent downloads corrupt each other's partial files. Install the | |
| # rust-toolchain.toml pin serially up front. | |
| - name: Install pinned toolchain | |
| shell: pwsh | |
| run: | | |
| rustup toolchain install | |
| rustup show active-toolchain | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@nextest | |
| - name: Download nextest archive | |
| uses: actions/download-artifact@v4 | |
| with: | |
| name: windows-nextest-archive | |
| path: ${{ runner.temp }} | |
| - name: Download Windows evaluator executables | |
| uses: actions/download-artifact@v4 | |
| with: | |
| name: windows-search-eval-bins | |
| path: ${{ runner.temp }}/search-eval-bin | |
| # Preflight: the evaluator subprocesses the acceptance and packaged | |
| # suites launch must exist, execute, and report their bundled workload | |
| # before any test runs, so a provisioning gap fails here instead of as | |
| # five product-test failures elsewhere in the shard matrix. `validate` | |
| # against an empty directory is the packaged-identity probe: it reads | |
| # only the workload compiled into the binary, so a build missing it | |
| # cannot pass. The exports are the overrides both resolvers declare | |
| # (`search_eval_bin` in crates/tracedecay/tests/common/mod.rs and | |
| # `search_eval_direct_bin` in the default-feature packaged suite), which | |
| # bind every test to the executables proven here rather than to whatever | |
| # the archive happens to extract beside the test binaries. | |
| - name: Preflight the restored evaluator executables | |
| shell: pwsh | |
| run: | | |
| $ErrorActionPreference = "Stop" | |
| $binDir = Join-Path $env:RUNNER_TEMP "search-eval-bin" | |
| $probe = Join-Path $env:RUNNER_TEMP "search-eval-preflight" | |
| New-Item -ItemType Directory -Force -Path $probe | Out-Null | |
| foreach ($name in @("tracedecay-search-eval", "tracedecay-search-eval-direct")) { | |
| $path = Join-Path $binDir "$name.exe" | |
| if (-not (Test-Path -LiteralPath $path -PathType Leaf)) { | |
| throw "restored evaluator '$name.exe' is missing at $path" | |
| } | |
| # stdout only: the typed report is the whole of it, and merging | |
| # stderr in would make any log line unparseable JSON. | |
| $stdout = & $path validate --repo-root $probe | Out-String | |
| if ($LASTEXITCODE -ne 0) { | |
| throw "$name validate exited with $LASTEXITCODE`n$stdout" | |
| } | |
| $report = $stdout | ConvertFrom-Json | |
| if ($report.command -ne "validate" -or $report.status -ne "pass") { | |
| throw "$name reported command=$($report.command) status=$($report.status)`n$stdout" | |
| } | |
| if ([int]$report.query_count -le 0 -or [int]$report.profile_count -le 0) { | |
| throw "$name has no bundled workload: query_count=$($report.query_count) profile_count=$($report.profile_count)" | |
| } | |
| Write-Host "$name.exe ok: $($report.query_count) queries, $($report.profile_count) profiles" | |
| } | |
| "TRACEDECAY_SEARCH_EVAL_TEST_BIN=$binDir\tracedecay-search-eval.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append | |
| "TRACEDECAY_SEARCH_EVAL_DIRECT_TEST_BIN=$binDir\tracedecay-search-eval-direct.exe" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append | |
| - name: Run Windows tests | |
| shell: pwsh | |
| # --extract-to must be the workspace so target/ lands at the same | |
| # absolute path as on the build job: integration tests bake | |
| # env!("CARGO_BIN_EXE_tracedecay") into the binaries at compile time. | |
| run: >- | |
| cargo-nextest nextest run --profile ci | |
| --archive-file "$env:RUNNER_TEMP/nextest-archive.tar.zst" | |
| --extract-to "$env:GITHUB_WORKSPACE" | |
| --workspace-remap "$env:GITHUB_WORKSPACE" | |
| --partition slice:${{ matrix.partition }}/5 | |
| --test-threads num-cpus --status-level slow | |
| - name: Clean abandoned Windows test children | |
| if: always() | |
| shell: pwsh | |
| run: | | |
| $workspace = (Resolve-Path $env:GITHUB_WORKSPACE).Path | |
| $all = Get-CimInstance Win32_Process | |
| $liveProcessIds = [System.Collections.Generic.HashSet[uint32]]::new() | |
| foreach ($process in $all) { | |
| [void]$liveProcessIds.Add([uint32]$process.ProcessId) | |
| } | |
| $stale = foreach ($process in $all) { | |
| if ($process.Name -ne "tracedecay.exe") { | |
| continue | |
| } | |
| if (-not $process.CommandLine) { | |
| continue | |
| } | |
| $inCurrentWorkspace = $process.CommandLine.IndexOf($workspace, [StringComparison]::OrdinalIgnoreCase) -ge 0 | |
| if (-not $inCurrentWorkspace) { | |
| continue | |
| } | |
| if (-not $liveProcessIds.Contains([uint32]$process.ParentProcessId)) { | |
| $process | |
| } | |
| } | |
| foreach ($process in $stale) { | |
| Write-Host "Stopping abandoned tracedecay child pid=$($process.ProcessId)" | |
| Stop-Process -Id $process.ProcessId -Force -ErrorAction SilentlyContinue | |
| } | |
| - name: Upload Windows nextest report | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: windows-nextest-junit-${{ matrix.partition }} | |
| path: target/nextest/ci/junit.xml | |
| if-no-files-found: ignore | |
| retention-days: 7 | |
| windows-test: | |
| name: Test Windows | |
| if: ${{ !cancelled() && needs.scope-gate.outputs.run-heavy == 'true' }} | |
| needs: [windows-test-shard, scope-gate] | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| - name: Check Windows shards | |
| run: | | |
| if [ "${{ needs.windows-test-shard.result }}" != "success" ]; then | |
| echo "Windows shard result: ${{ needs.windows-test-shard.result }}" | |
| exit 1 | |
| fi | |
| echo "All Windows shards passed." | |
| clippy: | |
| name: Clippy | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| # The free 4-vCPU Arm runner (Cobalt 100), as a measured comparison | |
| # against the x64 `ubuntu-latest` this lane took 19 minutes on: same | |
| # workload, same cache shape, different core. Public repositories get it | |
| # at no cost; the result decides whether the partitioned test jobs move. | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 30 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: dtolnay/rust-toolchain@stable | |
| with: | |
| components: clippy | |
| - uses: ./.github/actions/setup-linux-mold | |
| - name: Cache Rust build outputs | |
| uses: Swatinem/rust-cache@v2 | |
| with: | |
| shared-key: ci-clippy-full-${{ runner.os }}-${{ runner.arch }} | |
| cache-on-failure: true | |
| - name: Run blocking Clippy policy | |
| run: cargo clippy --workspace --all-targets --locked -- -D warnings | |
| - name: Check lean build | |
| run: cargo check -p tracedecay --no-default-features --locked | |
| # Compile-only gate for feature graphs beyond the default test-lane | |
| # selections. `test-transport` acceptance suites (`mcp_suite`, | |
| # `transport_acceptance_suite`, `host_journeys_suite`, `work_loop_journey`) | |
| # now run as the `root-transport` Linux partition; this job still | |
| # `cargo check`s the full `--features test-transport` / hotpath graphs so | |
| # non-selected targets cannot rot until release `--all-features`. | |
| # | |
| # Its own job, not a tail on the Linux build: these are dev-profile check | |
| # builds, a graph the perf-profile test lanes share nothing with, so there | |
| # is no compiled tree to reuse and nothing to gain from waiting for one. | |
| # As the last steps of the Linux test job they ran only after a green | |
| # suite — never once on this branch — and would have held the Linux shards | |
| # back by their own duration. Here the verdict lands independently at | |
| # about clippy's cost: the cold `cargo clippy --workspace --all-targets` | |
| # took 13.7 min (run 34231734416), the second check re-resolves only the | |
| # crates the hotpath features touch (clippy's lean re-check: 2.7 min). | |
| # Feature wiring does not vary by host, so one OS carries the gate. No | |
| # cache: it is off every lane's critical path, and another lineage would | |
| # compete for the 10 GB budget the test lanes' dependency caches are | |
| # already evicted from. | |
| feature-gates: | |
| name: Feature gates | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 30 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: dtolnay/rust-toolchain@stable | |
| - uses: ./.github/actions/setup-linux-mold | |
| - name: Check feature-gated surfaces compile | |
| run: | | |
| cargo check --workspace --all-targets --features test-transport --locked | |
| cargo check --workspace --all-targets --features hotpath,hotpath-mcp --locked | |
| fmt: | |
| name: Format | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| # Judge formatting with the toolchain `rust-toolchain.toml` pins, not | |
| # whatever `stable` is on the runner. rustfmt's output changes between | |
| # releases — 1.98.1 breaks a `.unwrap_or_else(|err| { ... })` chain that | |
| # 1.97.1 keeps inline — so a `stable` rustfmt rejects a tree the pinned | |
| # rustfmt developers run formats exactly, and every contributor sees a | |
| # red lane they cannot reproduce. `dtolnay/rust-toolchain@stable` also | |
| # exports `RUSTUP_TOOLCHAIN`, which overrides the toolchain file rather | |
| # than deferring to it. Installing through rustup with no override keeps | |
| # the pin in one place: the toolchain file, components included. | |
| - name: Install the repository's pinned toolchain | |
| run: rustup show active-toolchain || rustup toolchain install | |
| - run: cargo fmt --all -- --check | |
| dashboard: | |
| name: Dashboard | |
| needs: scope-gate | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 25 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-node@v6 | |
| with: | |
| node-version: "22" | |
| cache: npm | |
| cache-dependency-path: dashboard/package-lock.json | |
| - name: Install dashboard dependencies | |
| working-directory: dashboard | |
| run: npm ci | |
| - name: Build dashboard assets | |
| working-directory: dashboard | |
| run: npm run build | |
| - name: Run dashboard unit tests | |
| working-directory: dashboard | |
| run: npm test | |
| # The dashboard is one rsbuild app emitting `app-dist/` (the legacy | |
| # shell/holographic/lcm/graph/savings bundle split no longer exists), so | |
| # the bundle validator is the artifact authority and the determinism | |
| # check hashes whatever the build actually produced. | |
| - name: Verify embedded dist artifacts | |
| working-directory: dashboard | |
| run: | | |
| set -euo pipefail | |
| python3 ../scripts/check-dashboard-bundle.py app-dist | |
| (cd app-dist && find . -type f | sort | xargs sha256sum) > /tmp/dashboard-dist.sha | |
| npm run build | |
| (cd app-dist && find . -type f | sort | xargs sha256sum) | diff /tmp/dashboard-dist.sha - | |
| hermes-integration: | |
| name: Hermes integration (stock) | |
| needs: [scope-gate, debug-cli] | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 30 | |
| env: | |
| # Stock (upstream) Hermes pin — the generated plugin must keep working | |
| # against this exact upstream commit. Bump deliberately after rerunning | |
| # scripts/hermes_stock_integration.sh against the new ref locally. | |
| HERMES_UPSTREAM_REPO: https://github.com/NousResearch/hermes-agent.git | |
| HERMES_UPSTREAM_REF: 9dd9ef0ec99a87f078f7272b4323df5440b4b3f9 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Install ast-grep | |
| uses: ./.github/actions/install-ast-grep | |
| with: | |
| version: ${{ env.AST_GREP_VERSION }} | |
| - name: Fetch the debug CLI built once for this run | |
| uses: actions/download-artifact@v4 | |
| with: | |
| name: debug-tracedecay-cli | |
| path: target/debug | |
| - name: Restore the executable bit the artifact store drops | |
| run: chmod +x target/debug/tracedecay | |
| - name: Run generated-plugin unit checks (no Hermes required) | |
| run: python3 scripts/hermes_plugin_unit_check.py | |
| - name: Clone stock Hermes at pinned ref | |
| run: | | |
| set -euo pipefail | |
| git init -q "$RUNNER_TEMP/hermes-upstream" | |
| git -C "$RUNNER_TEMP/hermes-upstream" remote add origin "$HERMES_UPSTREAM_REPO" | |
| git -C "$RUNNER_TEMP/hermes-upstream" fetch --depth 1 origin "$HERMES_UPSTREAM_REF" | |
| git -C "$RUNNER_TEMP/hermes-upstream" checkout --detach FETCH_HEAD | |
| - uses: astral-sh/setup-uv@v8.2.0 | |
| with: | |
| enable-cache: true | |
| cache-dependency-glob: ${{ runner.temp }}/hermes-upstream/uv.lock | |
| cache-suffix: hermes-${{ env.HERMES_UPSTREAM_REF }} | |
| - name: Set up stock Hermes environment | |
| working-directory: ${{ runner.temp }}/hermes-upstream | |
| run: uv sync --frozen --no-dev | |
| - name: Run stock Hermes integration checks | |
| env: | |
| TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay | |
| HERMES_UPSTREAM_DIR: ${{ runner.temp }}/hermes-upstream | |
| run: scripts/hermes_stock_integration.sh | |
| host-stock-integration: | |
| name: Claude Code + OpenCode integration (stock) | |
| needs: [scope-gate, debug-cli] | |
| if: ${{ needs.scope-gate.outputs.run-heavy == 'true' }} | |
| runs-on: ubuntu-24.04-arm | |
| timeout-minutes: 30 | |
| env: | |
| # Stock host pins — the installed bundle must keep working against these | |
| # exact host releases. Bump deliberately after rerunning | |
| # scripts/claude_stock_integration.sh and | |
| # scripts/opencode_stock_integration.sh against the new versions locally. | |
| CLAUDE_CODE_VERSION: 2.1.224 | |
| OPENCODE_VERSION: 1.18.4 | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Fetch the debug CLI built once for this run | |
| uses: actions/download-artifact@v4 | |
| with: | |
| name: debug-tracedecay-cli | |
| path: target/debug | |
| - name: Restore the executable bit the artifact store drops | |
| run: chmod +x target/debug/tracedecay | |
| - name: Install stock Claude Code and OpenCode at pinned versions | |
| run: | | |
| npm install --global "@anthropic-ai/claude-code@${CLAUDE_CODE_VERSION}" "opencode-ai@${OPENCODE_VERSION}" | |
| claude --version | |
| opencode --version | |
| - name: Run stock Claude Code integration checks | |
| env: | |
| TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay | |
| run: scripts/claude_stock_integration.sh | |
| - name: Run stock OpenCode integration checks | |
| env: | |
| TRACEDECAY_BIN: ${{ github.workspace }}/target/debug/tracedecay | |
| run: scripts/opencode_stock_integration.sh |