v0.50.4.0 ci: accelerate required checks with isolated test workers (#5131)

* ci: refresh timing weights and share balanced scheduling

* ci: freeze selected E2E files across four isolated workers

* ci: shard isolated serial pools and verify coverage identity

* test: reuse isolated snapshot profiles for BrainBench CLI evals

* test: preserve snapshot ownership across normal lock release

* test: cover exclusive cancellation and interrupted snapshot publication

* v0.50.4.0 ci: shorten required-check feedback and record measurements

Co-Authored-By: Codex <noreply@openai.com>

* docs: update project documentation for v0.50.4.0

---------

Co-authored-by: Codex <noreply@openai.com>
This commit is contained in:
Garry Tan
2026-09-15 21:23:22 -07:00
committed by GitHub
parent 7fb6617dbf
commit 039f2cd52a
73 changed files with 4532 additions and 2533 deletions

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Personal knowledge brain for your coding agent — hybrid search, synthesis, graph traversal, and durable cross-session memory over Postgres/PGLite with pgvector, plus a curated brain-first skill set.",
"author": {
"name": "Garry Tan",

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Personal knowledge brain for your coding agent — hybrid search, synthesis, graph traversal, and durable cross-session memory over Postgres/PGLite with pgvector, plus a curated brain-first skill set.",
"author": {
"name": "Garry Tan",

View File

@@ -101,24 +101,47 @@ jobs:
# (the unit wrappers strip the URL per #3485).
run: bun test --timeout=60000 test/e2e/op-checkpoint-jsonb-parity.test.ts test/e2e/jsonb-roundtrip.test.ts test/phantom-redirect-engine-parity.test.ts
# ──────────────────────────────────────────────────────────────────────
# selected-e2e (test-gap plan G1): PR-time signal for the e2e files that
# otherwise run only in the scheduled nightly full glob.
# scripts/select-e2e.ts picks the diff-relevant files — fail-closed:
# uncertain diffs emit ALL, git failure exits 2 and FAILS this job (the
# run step must never be wrapped in `|| true`). Files already carried by
# the named jobs and the live-key token spenders are excluded with a loud
# per-file echo (no silent caps); the nightly full glob remains the
# backstop for the excluded set. Fork PRs: this job needs ONLY the
# Postgres service container (no repository secrets), so outside
# contributors run it natively — no skip path needed. select-e2e's
# --classify-only mode is load-bearing in coverage-diff-gate.ts on every
# PR; this job consumes the default file-selection mode and must not
# change the script's exit semantics. On scheduled runs the step
# early-exits green: coverage-full-e2e already runs the full glob that
# night. The aggregate requires that full-corpus execution on schedules.
# Freeze selection before setup; each worker owns a separate Postgres service.
prepare-e2e:
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
matrix: ${{ steps.select.outputs.matrix }}
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
fetch-depth: 0
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.3.13
- name: Select and freeze E2E files
id: select
shell: bash
run: |
if [ "${{ github.event_name }}" = "schedule" ]; then
# Nightly coverage carries the full E2E corpus separately.
: > "$RUNNER_TEMP/selected.txt"
else
git fetch origin master --quiet
bun scripts/select-e2e.ts > "$RUNNER_TEMP/selected.txt"
fi
matrix=$(bun scripts/e2e-matrix.ts prepare < "$RUNNER_TEMP/selected.txt")
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
printf '%s\n' "$matrix" > "$RUNNER_TEMP/e2e-selection.json"
- name: Upload frozen selection
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: e2e-selection
path: ${{ runner.temp }}/e2e-selection.json
retention-days: 14
overwrite: true
selected-e2e:
name: Selected E2E (diff-relevant)
name: Selected E2E (diff-relevant) ${{ matrix.shard }}
needs: prepare-e2e
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.prepare-e2e.outputs.matrix) }}
runs-on: ubuntu-latest
timeout-minutes: 60
services:
@@ -137,9 +160,6 @@ jobs:
--health-retries 5
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
# select-e2e diffs origin/master...HEAD — needs real history.
fetch-depth: 0
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.3.13
@@ -158,39 +178,21 @@ jobs:
test/fixtures/pglite-snapshot.version
key: pglite-snapshot-${{ runner.os }}-${{ hashFiles('src/core/migrate.ts', 'src/core/pglite-schema.ts', 'src/core/pglite-engine.ts', 'src/core/fts-language.ts', 'src/core/vector-index.ts', 'src/core/ai/defaults.ts', 'src/core/timeline-dedup-repair.ts', 'src/core/pages-upsert-arbiter.ts', 'src/core/link-extraction.ts', 'src/core/grants/*.ts', 'src/core/scope.ts', 'src/core/sql-query.ts', 'src/core/minions/tools/brain-allowlist.ts', 'src/core/facts/withdrawal-schema.ts', 'test/helpers/legacy-embedding-config.ts', 'scripts/build-pglite-snapshot.ts') }}
- run: bun install --frozen-lockfile
- name: Select and run diff-relevant E2E files
- name: Run frozen E2E partition
shell: bash
env:
DATABASE_URL: postgresql://postgres:postgres@localhost:5432/gbrain_test
# #3485 preload guard: this job intentionally tests against a DB.
GBRAIN_TEST_ALLOW_DATABASE_URL: '1'
run: |
if [ "${{ github.event_name }}" = "schedule" ]; then
echo "scheduled run — coverage-full-e2e carries the full glob tonight; nothing to select"
exit 0
fi
git fetch origin master --quiet
# exit 2 from the selector fails this step — fail-loud by design.
bun scripts/select-e2e.ts > /tmp/selected.txt
# Exclusions (echoed per file, never silent): files the named jobs
# above already run this same workflow, plus live-key token
# spenders (this job carries no provider keys, so they would
# self-skip anyway — the list makes the spend guarantee explicit).
EXCLUDE='test/e2e/op-checkpoint-jsonb-parity.test.ts test/e2e/jsonb-roundtrip.test.ts test/e2e/mechanical.test.ts test/e2e/mcp.test.ts test/e2e/job-isolation.test.ts test/e2e/sync-reconcile-postgres.test.ts test/e2e/engine-parity.test.ts test/e2e/serve-http-multi-agent.test.ts test/e2e/postgres-bootstrap.test.ts test/e2e/sync-delegation-under-serve.serial.test.ts test/e2e/dream-synthesize-pglite.test.ts test/e2e/skills.test.ts test/e2e/zeroentropy-live.test.ts test/e2e/voyage-rerank-live.test.ts test/e2e/voyage-multimodal.test.ts'
: > /tmp/run.txt
while IFS= read -r f; do
[ -z "$f" ] && continue
case " $EXCLUDE " in
*" $f "*) echo "excluded (named-job / live-key lane): $f" ;;
*) echo "$f" >> /tmp/run.txt ;;
esac
done < /tmp/selected.txt
COUNT=$(wc -l < /tmp/run.txt | tr -d ' ')
echo "selector emitted $(wc -l < /tmp/selected.txt | tr -d ' ') file(s); running $COUNT after exclusions"
if [ "$COUNT" -eq 0 ]; then
echo "doc-only or fully-excluded diff — nothing to run here"
exit 0
fi
xargs -a /tmp/run.txt bash scripts/run-e2e.sh
E2E_MATRIX_ROW: ${{ toJSON(matrix) }}
run: bun scripts/capture-test-log.ts --job 'Selected E2E (diff-relevant) ${{ matrix.shard }}' --out "$RUNNER_TEMP/e2e.log" -- bun scripts/e2e-matrix.ts run
- name: Upload E2E timing log
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: timings-e2e-${{ matrix.shard }}
path: ${{ runner.temp }}/e2e.log
retention-days: 14
overwrite: true
tier1:
name: Tier 1 (Mechanical)
@@ -252,7 +254,7 @@ jobs:
# binding path could reach master with every PR lane green. It
# already uses the same setupDB/teardownDB TRUNCATE-reset pattern
# (test/e2e/helpers.ts) as the other DB-backed files on this line.
# Also added to the selected-e2e EXCLUDE list above so a PR
# Also added to the E2E_EXCLUSIONS in scripts/e2e-matrix.ts so a PR
# touching sync.ts doesn't run it a second time there.
run: bun test --timeout=60000 test/e2e/mechanical.test.ts test/e2e/mcp.test.ts test/e2e/job-isolation.test.ts test/e2e/sync-reconcile-postgres.test.ts
env:
@@ -611,7 +613,7 @@ jobs:
# The stable aggregate includes nightly execution lanes on schedules.
# Coverage reporting itself remains advisory; failed tests never do.
e2e-status:
needs: [jsonb-parity, tier1, tier2, selected-e2e, coverage-full-unit, coverage-full-serial, coverage-full-slow, coverage-full-e2e]
needs: [jsonb-parity, tier1, tier2, prepare-e2e, selected-e2e, coverage-full-unit, coverage-full-serial, coverage-full-slow, coverage-full-e2e]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 5
@@ -621,11 +623,12 @@ jobs:
JSONB="${{ needs.jsonb-parity.result }}"
TIER1="${{ needs.tier1.result }}"
TIER2="${{ needs.tier2.result }}"
PREPARE="${{ needs.prepare-e2e.result }}"
SELECTED="${{ needs.selected-e2e.result }}"
EVENT="${{ github.event_name }}"
echo "event=$EVENT"
echo "jsonb-parity=$JSONB tier1=$TIER1 tier2=$TIER2 selected-e2e=$SELECTED"
for r in "$JSONB" "$TIER1" "$TIER2" "$SELECTED"; do
echo "jsonb-parity=$JSONB tier1=$TIER1 tier2=$TIER2 prepare-e2e=$PREPARE selected-e2e=$SELECTED"
for r in "$JSONB" "$TIER1" "$TIER2" "$PREPARE" "$SELECTED"; do
if [ "$r" != "success" ]; then
echo "✗ gated e2e job did not succeed (got $r) — E2E fail"
exit 1

View File

@@ -153,13 +153,13 @@ jobs:
# sequential: an 8.5-minute job whose serialization the quarantine
# contract never required). Lives in its own runner so the matrix shards
# aren't carrying the serial tail.
strategy:
fail-fast: false
matrix:
shard: [1, 2, 3, 4]
runs-on: ubuntu-latest
# 25, not 15: on smaller runners the memory clamp in
# scripts/run-serial-tests.sh sizes the pool to 1, and the fully
# sequential suite runs 12-13 minutes — a slow runner then hits a 15m
# cap with every test passing (observed: canceled at exactly 15:00
# with zero failures). On 4-core runners pool=4 finishes in ~3m and
# the extra headroom is never used.
# Keep the existing timeout headroom for memory-constrained runners,
# bounded rescue attempts, and the machine-exclusive tail on shard 1.
timeout-minutes: 25
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -181,7 +181,8 @@ jobs:
# slow-eval, slow-perf, slow-brainbench; + 5 restores in e2e.yml:
# jsonb-parity, selected-e2e, tier1, tier2, coverage-full-e2e) — edit
# all together, or drift shows up only as silent rebuild cost.
- uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3
- uses: actions/cache/restore@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3
id: snapshot-cache
with:
path: |
test/fixtures/pglite-snapshot.tar
@@ -190,16 +191,36 @@ jobs:
- run: bun install --frozen-lockfile
- run: bun run test:serial
env:
SHARD: ${{ matrix.shard }}/4
COVERAGE_DIR: ${{ runner.temp }}/coverage
- name: Upload coverage (serial lane)
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: coverage-serial
name: coverage-serial-${{ matrix.shard }}
path: ${{ runner.temp }}/coverage
retention-days: 14
if-no-files-found: ignore
overwrite: true
- name: Save legacy snapshot (single writer)
if: success() && matrix.shard == 1
uses: actions/cache/save@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3
with:
path: |
test/fixtures/pglite-snapshot.tar
test/fixtures/pglite-snapshot.version
key: ${{ steps.snapshot-cache.outputs.cache-primary-key }}
- name: Upload serial timings
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: timings-serial-${{ matrix.shard }}
path: |
.context/serial-timings.json
.context/serial-durations.txt
include-hidden-files: true
retention-days: 14
overwrite: true
slow-eval-longmemeval:
# Dedicated runner for the LongMemEval end-to-end test file. The file
@@ -272,6 +293,12 @@ jobs:
test/fixtures/pglite-snapshot.tar
test/fixtures/pglite-snapshot.version
key: pglite-snapshot-${{ runner.os }}-${{ hashFiles('src/core/migrate.ts', 'src/core/pglite-schema.ts', 'src/core/pglite-engine.ts', 'src/core/fts-language.ts', 'src/core/vector-index.ts', 'src/core/ai/defaults.ts', 'src/core/timeline-dedup-repair.ts', 'src/core/pages-upsert-arbiter.ts', 'src/core/link-extraction.ts', 'src/core/grants/*.ts', 'src/core/scope.ts', 'src/core/sql-query.ts', 'src/core/minions/tools/brain-allowlist.ts', 'src/core/facts/withdrawal-schema.ts', 'test/helpers/legacy-embedding-config.ts', 'scripts/build-pglite-snapshot.ts') }}
- uses: actions/cache/restore@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3
with:
path: |
test/fixtures/pglite-snapshot-default.tar
test/fixtures/pglite-snapshot-default.version
key: pglite-snapshot-default-${{ runner.os }}-${{ hashFiles('src/core/migrate.ts', 'src/core/pglite-schema.ts', 'src/core/pglite-engine.ts', 'src/core/fts-language.ts', 'src/core/vector-index.ts', 'src/core/ai/defaults.ts', 'src/core/timeline-dedup-repair.ts', 'src/core/pages-upsert-arbiter.ts', 'src/core/link-extraction.ts', 'src/core/grants/*.ts', 'src/core/scope.ts', 'src/core/sql-query.ts', 'src/core/minions/tools/brain-allowlist.ts', 'src/core/facts/withdrawal-schema.ts', 'test/helpers/legacy-embedding-config.ts', 'scripts/build-pglite-snapshot.ts') }}
- run: bun install --frozen-lockfile
- name: Ensure PGLite snapshot (build-or-validate, non-fatal)
# Absolutized: this file's batch jobs spawn CLI children with varying
@@ -279,6 +306,8 @@ jobs:
run: |
. scripts/lib/test-env.sh
ensure_pglite_snapshot slow-brainbench
ensure_default_pglite_snapshot slow-brainbench
echo "GBRAIN_TEST_DEFAULT_SNAPSHOT=${GBRAIN_TEST_DEFAULT_SNAPSHOT:-}" >> "$GITHUB_ENV"
# Export absolute (CLI children spawn with varying cwd); keep an
# already-absolute inherited path as-is; export nothing on build failure.
if [ -n "${GBRAIN_PGLITE_SNAPSHOT:-}" ]; then
@@ -327,6 +356,12 @@ jobs:
path: ~/.bun/install/cache
key: bun-cache-${{ runner.os }}-${{ hashFiles('bun.lock') }}
restore-keys: bun-cache-${{ runner.os }}-
- uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3
with:
path: |
test/fixtures/pglite-snapshot-default.tar
test/fixtures/pglite-snapshot-default.version
key: pglite-snapshot-default-${{ runner.os }}-${{ hashFiles('src/core/migrate.ts', 'src/core/pglite-schema.ts', 'src/core/pglite-engine.ts', 'src/core/fts-language.ts', 'src/core/vector-index.ts', 'src/core/ai/defaults.ts', 'src/core/timeline-dedup-repair.ts', 'src/core/pages-upsert-arbiter.ts', 'src/core/link-extraction.ts', 'src/core/grants/*.ts', 'src/core/scope.ts', 'src/core/sql-query.ts', 'src/core/minions/tools/brain-allowlist.ts', 'src/core/facts/withdrawal-schema.ts', 'test/helpers/legacy-embedding-config.ts', 'scripts/build-pglite-snapshot.ts') }}
- run: bun install --frozen-lockfile
- run: bash scripts/ci-brainbench-gate.sh
env:
@@ -447,7 +482,8 @@ jobs:
key: pglite-snapshot-${{ runner.os }}-${{ hashFiles('src/core/migrate.ts', 'src/core/pglite-schema.ts', 'src/core/pglite-engine.ts', 'src/core/fts-language.ts', 'src/core/vector-index.ts', 'src/core/ai/defaults.ts', 'src/core/timeline-dedup-repair.ts', 'src/core/pages-upsert-arbiter.ts', 'src/core/link-extraction.ts', 'src/core/grants/*.ts', 'src/core/scope.ts', 'src/core/sql-query.ts', 'src/core/minions/tools/brain-allowlist.ts', 'src/core/facts/withdrawal-schema.ts', 'test/helpers/legacy-embedding-config.ts', 'scripts/build-pglite-snapshot.ts') }}
- run: bun install --frozen-lockfile
- name: Run test shard ${{ matrix.shard }}/10
run: scripts/test-shard.sh ${{ matrix.shard }} 10
shell: bash
run: bun scripts/capture-test-log.ts --job 'test (${{ matrix.shard }})' --out "$RUNNER_TEMP/unit.log" -- bash scripts/test-shard.sh ${{ matrix.shard }} 10
env:
COVERAGE_DIR: ${{ runner.temp }}/coverage
- name: Upload coverage (shard lane)
@@ -459,9 +495,18 @@ jobs:
retention-days: 14
if-no-files-found: ignore
overwrite: true
- name: Upload unit timing log
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: timings-unit-${{ matrix.shard }}
path: ${{ runner.temp }}/unit.log
retention-days: 14
overwrite: true
# ──────────────────────────────────────────────────────────────────────
# coverage-report: merges the PR corpus (10 shards + serial + 2 slow
# coverage-report: merges the PR corpus (10 unit + 4 serial + 3 slow
# lanes) into one honest number, renders it to the step summary, and runs
# the diff + baseline gates in report-only mode (COVERAGE_GATE_ENFORCE=0).
# ADVISORY during the report-only window: deliberately NOT in test-status's
@@ -499,7 +544,7 @@ jobs:
bun scripts/merge-lcov.ts \
--out-lcov "$RUNNER_TEMP/coverage-merged/lcov.info" \
--out-json "$RUNNER_TEMP/coverage-merged/summary.json" \
--manifest-expect shard-1,shard-2,shard-3,shard-4,shard-5,shard-6,shard-7,shard-8,shard-9,shard-10,serial,sloweval,slowperf,slowbrainbench \
--manifest-expect shard-1,shard-2,shard-3,shard-4,shard-5,shard-6,shard-7,shard-8,shard-9,shard-10,serial-1,serial-2,serial-3,serial-4,sloweval,slowperf,slowbrainbench \
"$RUNNER_TEMP/coverage-artifacts"
- name: Coverage summary → step summary
run: bun scripts/render-coverage-summary.ts --summary "$RUNNER_TEMP/coverage-merged/summary.json" --structural scripts/structural-suites.tsv >> "$GITHUB_STEP_SUMMARY"

4
.gitignore vendored
View File

@@ -46,6 +46,10 @@ AGENTS.local.md
# Tier 3 PGLite snapshot fixture (built on demand by build:pglite-snapshot)
test/fixtures/pglite-snapshot.tar
test/fixtures/pglite-snapshot.version
test/fixtures/pglite-snapshot-default.tar
test/fixtures/pglite-snapshot-default.version
test/fixtures/.pglite-snapshot*.lock*
test/fixtures/pglite-snapshot*.tmp
# Private brain reports — never check these in (per CLAUDE.md privacy rule)
reports/network-intelligence/

View File

@@ -1,4 +1,4 @@
<!-- gbrain-runbook-stamp: 0.50.2.0 -->
<!-- gbrain-runbook-stamp: 0.50.4.0 -->
<!-- This stamp must equal the VERSION file at every release; CI enforces it
(scripts/check-bootstrap-tag.sh). `gbrain bootstrap status` compares it to
the installed binary and warns on skew. -->

View File

@@ -2,6 +2,31 @@
All notable changes to GBrain will be documented in this file.
## [0.50.4.0] - 2026-09-16
### For contributors
**Required CI checks finish about two-thirds sooner in matched warm-cache runs.** Selected E2E tests and isolated serial tests each use four workers, while ten unit workers share a duration-weighted scheduler. BrainBench CLI children reuse a snapshot built for their default embedding profile.
Three matched warm-cache pairs and one cold-cache pair on Bun 1.3.13 measured the time from workflow dispatch through both required aggregate checks, including job queues. Dispatch timing is a proxy for push-to-green and excludes webhook delivery.
| Measurement | Before | After | Change |
|---|---:|---:|---:|
| Warm required checks, median | 16m23s | 5m21s | 67.3% shorter |
| Cold required checks, one pair | 12m57s | 5m42s | 56.0% shorter |
| Warm runner minutes, median | 66.57 | 70.17 | 5.4% more |
| Cold runner minutes, one pair | 63.77 | 71.65 | 12.4% more |
The measured improvement exceeds the 50% acceptance criterion. The projected 4–5 minute target remains unmet by 21 seconds at the warm median; the slowest unit worker itself did not get faster.
### Itemized changes
- Selected E2E files are selected and filtered once, then frozen into validated partitions with separate Postgres services. An explicit empty selection runs no tests; missing input fails.
- Serial workers retain one process per file and memory-aware pools. Machine-exclusive files run once, after shard 1's pool drains. Seventeen coverage lanes verify identity, commit, completion, and actual LCOV counts; incomplete evidence is visibly degraded.
- Unit, serial, and E2E timing refreshes retain source metadata, reject failed or truncated inputs, and keep timing artifacts for 14 days. Testing documentation explains when and how to refresh weights.
- Legacy unit and default CLI snapshots have separate artifacts and locks. Normal lock release and crash recovery preserve owner records so delayed cleanup cannot steal another builder's lock. Explicit cold-mode and schema/embedding validation remain supported.
- Matched runs retain all 262 serial and 222 selected E2E files. Cold and snapshot BrainBench runs produce identical scores across 804 turns. Required check names, eval thresholds, and nightly coverage remain unchanged; successful test results are never cached.
## [0.50.2.0] - 2026-09-15
**Saved pages stay saved, rejected writes stay rejected, and maintenance stops reporting unfinished work as complete.**

View File

@@ -2173,9 +2173,8 @@ Each was explicitly deferred in the pass's CEO/eng/outside-voice reviews.
Postgres and no TRUNCATE-race protection — run them in a parallel lane; default
the existing SHARD support (only ci-local uses it). Fold into the Postgres
template-database entry below in this file (CREATE DATABASE … TEMPLATE, ~50ms).
**Why deferred:** e2e is off the CI critical path after the workflow restructure;
ci-local + nightly benefit only. **Effort:** M. **Priority:** P2.
- [ ] **Second PGLite snapshot keyed by dims/model.** **What:** ~34 test files
**Current status:** selected E2E is on the measured PR critical path. Four isolated weighted CI workers now address it without moving tests between lanes; PGLite-only lane moves remain deferred. **Effort:** M. **Priority:** P2.
- [ ] **Second PGLite snapshot keyed by dims/model.** Implemented for BrainBench default-profile CLI children in the CI optimization pass; extending reuse to other deliberately reconfigured tests remains deferred. **What:** ~34 test files
configure zembed/1280 and always cold-init (the snapshot's shape gate correctly
refuses the 1536 fixture). Bake a second snapshot per shape; the version-file
format already carries dims/model. **Why deferred:** moderate effort, small win,
@@ -2374,8 +2373,21 @@ review-deferred, not fix-now). Grouped by component.
### Test infra (master-owned)
- [ ] **P1 — Test-infra pass Ships 2+3: serial burn-down, e2e lane moves + CI
sharding, weights re-mine.** **What:** the approved test-infra plan
- [x] **CI speed pass: weighted selected E2E and serial matrices.** Selected E2E
now freezes one selection across up to four isolated workers; serial uses four
weighted partitions with exclusive ownership and unique coverage artifacts.
Unit/serial weights were re-mined, E2E weights added, and the miner now captures
complete lane timings plus provenance. Local unit/E2E use the same scheduler.
The duplicate entity-card performance invocation was removed from unit shards.
The ten-way unit matrix remains: the 12-way increase and test reclassification
below are still deferred. Three matched warm-cache pairs measured median
required checks of 16m23s → 5m21s (67.3% shorter); one cold pair measured
12m57s → 5m42s (56.0% shorter). The 4–5 minute projection remains unmet.
**Completed:** v0.50.4.0 (2026-09-16).
- [ ] **P1 — Test-infra pass Ships 2+3: remaining serial burn-down and E2E lane
moves.** **What:** the approved test-infra plan
(`~/.claude/plans/system-instruction-you-are-working-sprightly-bee.md`, Ship 1
landed as the v0.47.7.0 wave) deliberately split into 3 ships for regression
attribution. Remaining: Phase 4 serial-lane burn-down (38 rename-safe
@@ -2386,11 +2398,11 @@ review-deferred, not fix-now). Grouped by component.
moving the ~52 PGLite-only `test/e2e/` files into the unit matrix (behavioral
move criterion: direct PGLite ctor + no e2e/helpers import + no
hasDatabase/DATABASE_URL gate + header read; lockstep: e2e-test-map rows,
e2e-unmapped-baseline shrink, classify-tests, seeded weights) + 4-way
`SHARD=N/M` matrix for `selected-e2e`/`coverage-full-e2e` with one postgres
service per matrix job, and Phase 6 `mine-shard-weights` re-mine (381 files
unweighted; add a `weights:mine` package script + documented cadence) then
matrix 10→12. Graduated batch gates: 5×-green first batch per class, 2×+CI
e2e-unmapped-baseline shrink, classify-tests, seeded weights), a possible
four-way `coverage-full-e2e` nightly matrix, and unit matrix 10→12.
The selected-E2E matrix, timing refresh, `weights:mine` command, and refresh
cadence are completed by the CI speed pass above. Graduated batch gates:
5×-green first batch per class, 2×+CI
after. **Why:** the remaining ~half of the measured win: serial lane 220→~130
files, e2e 60-min worst-case lane → ~15-25 min, honest weights. **Effort:** L
(spread over 2 ships).

View File

@@ -1 +1 @@
0.50.2.0
0.50.4.0

View File

@@ -16,7 +16,7 @@ Two equivalent paths:
guards + typecheck, then 4-shard parallel unit + E2E against four pgvector
containers plus a transaction-mode PgBouncer service (unit phase keeps
`DATABASE_URL` unset; `--no-shard` for the legacy sequential flow). Stronger
than PR CI's 2-file Tier 1 set; closer to what nightly Tier 1 catches. Spins
than PR CI's four-file Tier 1 job; closer to what nightly Tier 1 catches. Spins
up + tears down postgres automatically via `docker-compose.ci.yml`. Override
the host port with `GBRAIN_CI_PG_PORT=5435 bun run ci:local` if 5434 collides.
- `bun run ci:local:diff` runs only the E2E files matched by the diff selector

View File

@@ -28,12 +28,12 @@ Eight test command tiers, each with a clear scope:
| Command | What it runs | Wallclock | When to use |
|---|---|---|---|
| `bun run test` | Parallel unit-test fast loop. Sharded fan-out via `scripts/run-unit-parallel.sh` (default 4 shards — CPU-detected, clamped to a max of 8; 4 matches CI's fan-out and avoids PGLite WASM-init contention), then a serial pass over `*.serial.test.ts`. Excludes `*.slow.test.ts` and `test/e2e/*`. No pre-checks, no typecheck. Builds/refreshes the PGLite schema snapshot BEFORE the shard fan-out and exports `GBRAIN_PGLITE_SNAPSHOT` so PGLite-booting files restore a baked schema instead of replaying every migration (~3.5x per booting file; see "PGLite schema snapshot" below). Opt out: `GBRAIN_NO_SNAPSHOT=1`. Memory-safe by default: total concurrency (shards × intra-shard width) is capped to available memory at `GBRAIN_TEST_MEM_PER_FILE_MB` (default 1536 — a PGLite WASM instance) per concurrent slot, shedding INTRA-SHARD width first and shards only after it (bun's `--max-concurrency` bounds only `test.concurrent` tests — 1 file in the corpus — so intra width is nearly free to shed, while every dropped shard removes a whole bun process of real fan-out; shedding shards first would collapse a 16GB box to a serial 1×4 run, measured 3.25× slower than 4×1 on the same machine). Two phantom-failure classes are automatically re-run serially (the rescue pass): failures carrying the WASM out-of-memory signature, and shards killed externally (SIGTERM/SIGKILL well before the shard timeout — sibling workspaces' process cleanup, memory jetsam). On machines without coreutils `timeout`, the fallback watchdog drops a `.watchdog` sentinel before TERMing a shard at the cap so the WEDGED/EXIT-HANG classifier stays reachable there (a bare rc=143 would otherwise read as a plain failure). Phantoms pass serially and the run goes green with an `oom_rescued` note; real failures fail again serially and stay red. Knobs: `GBRAIN_TEST_NO_MEM_ADAPT=1`, `GBRAIN_TEST_NO_OOM_FALLBACK=1`, `GBRAIN_TEST_MAX_CONCURRENCY` (intra-shard, default 4), `GBRAIN_TEST_SHARD_TIMEOUT` / `GBRAIN_TEST_SHARD_KILL_AFTER`, plus `--shards N` / `--max-concurrency N` / `--dry-run` script args. | a few minutes on a Mac dev box | Inner edit loop. Default. |
| `bun run test` | Parallel unit-test fast loop. Sharded fan-out via `scripts/run-unit-parallel.sh` (default 4 shards — CPU-detected, clamped to a max of 8; 4 limits local PGLite WASM-init contention; GitHub CI uses 10 unit shards), then a serial pass over `*.serial.test.ts`. Excludes `*.slow.test.ts` and `test/e2e/*`. No pre-checks, no typecheck. Builds/refreshes the PGLite schema snapshot BEFORE the shard fan-out and exports `GBRAIN_PGLITE_SNAPSHOT` so PGLite-booting files restore a baked schema instead of replaying every migration (~3.5x per booting file; see "PGLite schema snapshot" below). Opt out: `GBRAIN_NO_SNAPSHOT=1`. Memory-safe by default: total concurrency (shards × intra-shard width) is capped to available memory at `GBRAIN_TEST_MEM_PER_FILE_MB` (default 1536 — a PGLite WASM instance) per concurrent slot, shedding INTRA-SHARD width first and shards only after it (bun's `--max-concurrency` bounds only `test.concurrent` tests — 1 file in the corpus — so intra width is nearly free to shed, while every dropped shard removes a whole bun process of real fan-out; shedding shards first would collapse a 16GB box to a serial 1×4 run, measured 3.25× slower than 4×1 on the same machine). Two phantom-failure classes are automatically re-run serially (the rescue pass): failures carrying the WASM out-of-memory signature, and shards killed externally (SIGTERM/SIGKILL well before the shard timeout — sibling workspaces' process cleanup, memory jetsam). On machines without coreutils `timeout`, the fallback watchdog drops a `.watchdog` sentinel before TERMing a shard at the cap so the WEDGED/EXIT-HANG classifier stays reachable there (a bare rc=143 would otherwise read as a plain failure). Phantoms pass serially and the run goes green with an `oom_rescued` note; real failures fail again serially and stay red. Knobs: `GBRAIN_TEST_NO_MEM_ADAPT=1`, `GBRAIN_TEST_NO_OOM_FALLBACK=1`, `GBRAIN_TEST_MAX_CONCURRENCY` (intra-shard, default 4), `GBRAIN_TEST_SHARD_TIMEOUT` / `GBRAIN_TEST_SHARD_KILL_AFTER`, plus `--shards N` / `--max-concurrency N` / `--dry-run` script args. | a few minutes on a Mac dev box | Inner edit loop. Default. |
| `bun run verify` | CI's authoritative pre-test gate set, fanned out by `scripts/run-verify-parallel.sh` through a bounded worker pool (default `detect_cpus`; override `GBRAIN_VERIFY_MAX_PARALLEL`) with the heavy checks ordered first (typecheck, the two compile-embed checks, admin build, fuzz bundles, guard self-tests, the PGLite-booting chronicle eval check, whole-tree greps). The battery includes the deterministic `check:eval-chronicle` eval gate; `check:eval-canary` is deliberately NOT in the battery (its test-file twin `test/eval-canary.test.ts` spawns the identical runner in the unit matrix, and CI's verify job and matrix always run together — the package script stays for on-demand runs, so `verify`-only local callers should know the canary rides the unit lane instead). The `CHECKS` array in that script is the single source of truth — CI literally calls `bun run verify` in a dedicated job. | ~50s (pool-bounded; longest check dominates) | Before pushing; before `/ship`. |
| `bun run test:full` | `verify && bun run test && bun run test:slow && [smart e2e]`. Smart e2e runs only when `DATABASE_URL` is set and propagates its failure; otherwise it prints a skip notice to stderr. Use `ci:local` to provision the databases and require PgBouncer execution. | ~3-5min depending on slow + e2e | Pre-merge sanity, before opening a PR. |
| `bun run ci:local` | Independent host gitleaks scans, then frozen dependencies, guards/typecheck, the complete serial and slow lanes, and four unit/E2E shards inside Docker. Each E2E shard has its own pgvector database; selected PgBouncer tests must execute against the transaction-mode pooler. Unit, serial, and slow lanes have database URL overrides unset. Any failed stage fails the command. Complete shard logs survive container teardown under `.context/ci-local-shards/`. `ci:local:diff` narrows E2E selection; `--no-shard` runs unit/E2E sequentially. Doc-only diffs still require successful gitleaks scans. | Depends on the full corpus | Full local gate before shipping. |
| `bun run test:slow` | Just the `*.slow.test.ts` set (intentional cold-path correctness checks). | seconds-to-minutes | When touching slow-path code. |
| `bun run test:serial` | Just the `*.serial.test.ts` set (cross-file-contention quarantine; one bun process per file for true module-registry isolation), run through a POOL of concurrent per-file processes — the isolation is per-process, not per-machine. Dispatch is heaviest-first (LPT) from the advisory `scripts/serial-weights.json` (seconds; mined from the `.context/serial-durations.txt` table each run banks; absent/corrupt weights fall back to discovery order, absent keys to the corpus p75 — scheduling only, never correctness; LPT order + the corrupt-weights fail-soft are pinned by `test/scripts/run-serial-pool.test.ts`). Pool defaults to `min(detect_cpus, 4)` then memory-adapts (same doctrine as the parallel runner); a small growth-guarded set of files (machine-global state or contention-critical timing — see the justified `EXCLUSIVE_FILES` list in `scripts/run-serial-tests.sh`, capped at 3 by `test/scripts/serial-files.test.ts`) runs on a sequential EXCLUSIVE lane after the pool. Per-test timeout 120s (pooled contention headroom); each pooled file is wall-clock-killed at 300s (`timeout -k`, exit-hang containment). Externally-killed files (exit 143/137 or a missing exit sentinel — sibling-workspace cleanup, memory jetsam) get ONE sequential rescue re-run, mirroring the parallel runner's doctrine: phantoms stay green with a rescue note, real failures stay red. Prints per-file PASS lines plus a top-10 slowest-files list. Knobs: `GBRAIN_SERIAL_POOL=N` (explicit pool width — bypasses the memory clamp; `1` restores fully-sequential), `GBRAIN_SERIAL_FILE_TIMEOUT`. | a few minutes for all ~220 files at pool=4 | Debugging quarantined files; CI's serial-tests job. |
| `bun run test:serial` | Just the `*.serial.test.ts` set (cross-file-contention quarantine; one bun process per file for true module-registry isolation), run through a POOL of concurrent per-file processes — the isolation is per-process, not per-machine. Dispatch is heaviest-first (LPT) from the advisory `scripts/serial-weights.json` (seconds; mined from the `.context/serial-durations.txt` table each run banks; absent/corrupt weights fall back to discovery order, absent keys to the corpus p75 — scheduling only, never correctness; LPT order + the corrupt-weights fail-soft are pinned by `test/scripts/run-serial-pool.test.ts`). Pool defaults to `min(detect_cpus, 4)` then memory-adapts (same doctrine as the parallel runner); a small growth-guarded set of files (machine-global state or contention-critical timing — see the justified `EXCLUSIVE_FILES` list in `scripts/run-serial-tests.sh`, capped at 3 by `test/scripts/serial-files.test.ts`) runs on a sequential EXCLUSIVE lane after the pool. Per-test timeout 120s (pooled contention headroom); each pooled file is wall-clock-killed at 300s (`timeout -k`, exit-hang containment). `SHARD=N/M` partitions pooled files by duration; the three exclusive files run only on shard 1. Unset runs the complete corpus. Routing variables are cleared before tests start, so nested runners remain independent. Externally-killed files (exit 143/137 or a missing exit sentinel — sibling-workspace cleanup, memory jetsam) get ONE sequential rescue re-run, mirroring the parallel runner's doctrine: phantoms stay green with a rescue note, real failures stay red. Prints per-file PASS lines plus a top-10 slowest-files list. Knobs: `GBRAIN_SERIAL_POOL=N` (explicit pool width — bypasses the memory clamp; `1` restores fully-sequential), `GBRAIN_SERIAL_FILE_TIMEOUT`. | a few minutes for all ~220 files at pool=4 | Debugging quarantined files; CI's serial-tests job. |
| `bun run test:e2e` | Real Postgres E2E. Requires Docker + `DATABASE_URL`. Sequential within a shard; `SHARD=N/M` fans out against separate databases (ci-local runs 4 containers). Activates the PGLite snapshot like every other runner (per-file cold-path opt-outs where the test asserts the path TO post-initSchema state), exporting it as an ABSOLUTE path so CLI children spawned with varying cwd still find it. | ~5-10min | Pre-ship; nightly. |
| `bun run test:compile-smoke` | Self-update integrity verify under a REAL `bun build --compile` binary, offline (sets `GBRAIN_SELFUPDATE_COMPILE_SMOKE=1`). The unit suite mocks the network seams; this proves the dependency-free crypto/base64/JSON verify path survives compilation — the failure mode `sigstore-js` would have hit. | ~5s (one compile) | When touching `src/core/binary-self-update.ts`; pre-ship on self-update changes. |
@@ -70,14 +70,15 @@ migration, ~3.1s each on a CI shard). Properties:
handler changes invalidate the fixture; coverage instrumentation does not
change the hash. Keep the dependency list in `computeSnapshotSchemaHash`
and the CI cache keys aligned when adding another schema helper.
- **Concurrency-safe.** Parallel shard runners / sibling workspaces serialize
on an atomic `mkdir` lock (`test/fixtures/.pglite-snapshot.lock`) with
staleness-verified takeover of a crashed builder; the tar is written first
and the version file last, so a crash can never leave a fresh-looking torn
fixture. `GBRAIN_SNAPSHOT_LOCK_TIMEOUT_MS` (default 120000) bounds the
waiter; an exhausted waiter facing a still-live lock proceeds unlocked as a
last resort (the loader gate below validates the version file, not the tar
bytes).
- **Concurrency-safe.** Each profile has its own lock with a PID/token owner
and host/process-namespace identity. Only a confirmed dead local owner using
the current retirement protocol can be reclaimed. Both normal release and
crash recovery retain a nonempty owner tombstone so a delayed observer cannot
remove the next builder's lock. Keep those records while builders may run.
Live, foreign, ownerless, or older-protocol locks time out without building;
callers visibly fall back to cold initialization.
Temporary tar/version files are atomically renamed, with the version last.
- **Never authoritative.** The loader (`tryLoadSnapshot` in
`src/core/pglite-engine.ts`) verifies the schema hash AND the embedding
shape the snapshot was baked with (`dims=` / `model=` lines in the version
@@ -90,6 +91,55 @@ migration, ~3.1s each on a CI shard). Properties:
Pinned by `test/snapshot-shape-guard.test.ts` (hash + shape refusal matrix,
imported SQL/handler dependency hash sensitivity).
The builder accepts `--profile legacy|default` (legacy remains the default).
Legacy uses the unit preload's embedding shape. Default uses the CLI's canonical
embedding shape and writes `pglite-snapshot-default.tar` plus its `.version`.
The artifacts, locks, and CI caches are separate. `ensure_default_pglite_snapshot`
exports an absolute `GBRAIN_TEST_DEFAULT_SNAPSHOT`; BrainBench applies it only
to CLI children, including `run-all`. The parent unit process retains its legacy
snapshot. The slow runner and direct BrainBench test invocation prepare the
default profile automatically. `GBRAIN_NO_SNAPSHOT=1` clears both paths and
survives test preloads.
### Keeping CI partitions balanced
Required CI runs ten weighted unit workers, four serial workers with bounded
per-file pools, and up to four selected E2E workers. E2E selection and exclusions
run once before setup; the resulting file lists are frozen and executed against
separate Postgres services. An explicit empty selection launches no tests;
selection errors, failed workers, cancellations, and unexpected skips fail the
existing aggregate checks. Nightly full-corpus lanes remain separate.
Refresh after a large test wave or when the longest shard repeatedly exceeds
the mean shard execution time by 25%:
```bash
bun run weights:mine --lane unit --run <successful-test-run>
bun run weights:mine --lane serial --run <successful-test-run>
bun run weights:mine --lane e2e --run <successful-e2e-run>
```
The miner accepts `--from-file` or stdin for timestamped GitHub-format logs and
`--out` for inspection before replacing a checked-in map. Unit timing uses only
unit matrix jobs, includes `evals/`, and closes the final file at the Bun summary.
Serial timing uses runner durations, never timestamps of buffered output. E2E
uses each file's Bun summary and merges partial selections into known weights.
File/stdin imports also merge unobserved entries; only a complete GitHub unit or
serial run replaces that lane's entire map. Captured artifacts need their final
successful completion marker, and GitHub imports verify every expected job.
Incomplete or failed inputs leave the existing map intact. Sidecar metadata
records the source run/commit, units, and counts. Unit/E2E weights are milliseconds;
serial weights remain seconds. New files receive the corpus p75 estimate. Empty
maps and zero-cost ties distribute files deterministically; corrupt serial
weights warn and retain safe fallback scheduling.
CI retains timestamped unit/E2E logs, frozen E2E selection, and serial attempt
records for 14 days. Compare push-to-required-green time including queueing,
first failure, rescues/reruns, runner minutes, unique file counts, and coverage
completeness. Compare cold and warm caches separately. Snapshot timings and
partition estimates are projections until matched workflow runs confirm them;
successful test results are never cached.
### Guard registry and self-test
`scripts/guards-manifest.tsv` is THE single registry of `scripts/check-*`
@@ -169,8 +219,8 @@ there even though they pass on Linux and macOS.
### CI vs local: intentionally divergent file sets
- **CI matrix** (`.github/workflows/test.yml`) runs `scripts/test-shard.sh` across 10 matrix shards partitioned by weight-aware LPT bin-packing (`scripts/sharding.ts`; files with no mined weight fall back to the p75 file weight so a new unweighted file can't silently unbalance a shard) and INCLUDES `*.slow.test.ts` (the three outlier slow files — longmemeval, entity-resolve-perf, brainbench-e2e — run as dedicated jobs alongside the matrix) plus `evals/**/*.test.ts` (keyless-allowlist-gated — `test/scripts/evals-collection.test.ts`). Each shard's bun process is bounded by `--max-concurrency` (`GBRAIN_TEST_MAX_CONCURRENCY`, default 4). Every bun-test job — matrix shards, serial-tests, verify, the slow/eval jobs — activates the PGLite schema snapshot (built in-runner via `scripts/lib/test-env.sh`; the brainbench gate brings its own in-memory PGLite and skips it; the ~42MB tar is also cached across jobs via actions/cache, with the runner's own hash check staying authoritative). CI EXCLUDES `*.serial.test.ts` from the shards and runs them in the pooled `serial-tests` job via `bun run test:serial` — one bun process per file preserves the `mock.module` quarantine; the pool runs those processes concurrently. `bun run verify` gets its own job too, as does the BrainBench memory-conformance gate (`brainbench` job → `scripts/ci-brainbench-gate.sh`, hermetic in-memory PGLite, ~15s), which compares HEAD's fresh run against master's committed baseline (`evals/brainbench/baselines/main.json`) — the `test-status` aggregate checks its result explicitly. E2E (`.github/workflows/e2e.yml`) always runs its applicable execution lanes, with the jsonb-parity job in front of tier2 as the token-spend gate, and aggregates through `e2e-status`. Scheduled runs also require the full-corpus lanes, including each slow suite excluded from the coverage shards (longmemeval, entity-resolve-perf, and brainbench-e2e). Both aggregates reject failures, cancellations, and unexpected skips. Dependency caches and validated PGLite snapshots remain; successful test results are never reused. CI is the ground truth for "did everything pass."
- **Local fast loop** (`scripts/run-unit-shard.sh` via the parallel wrapper) uses round-robin-by-index sharding and EXCLUDES `*.slow.test.ts` AND `*.serial.test.ts`. Local trades coverage for inner-loop speed; CI catches what local skips.
- **CI matrix** (`.github/workflows/test.yml`) runs `scripts/test-shard.sh` across 10 matrix shards partitioned by weight-aware LPT bin-packing (`scripts/sharding.ts`; files with no mined weight fall back to the p75 file weight so a new unweighted file can't silently unbalance a shard) and INCLUDES `*.slow.test.ts` (the four dedicated slow files — longmemeval, entity-resolve-perf, entity-card-perf, brainbench-e2e — run as dedicated jobs alongside the matrix) plus `evals/**/*.test.ts` (keyless-allowlist-gated — `test/scripts/evals-collection.test.ts`). Each shard's bun process is bounded by `--max-concurrency` (`GBRAIN_TEST_MAX_CONCURRENCY`, default 4). Every bun-test job — matrix shards, serial-tests, verify, the slow/eval jobs — activates the PGLite schema snapshot (built in-runner via `scripts/lib/test-env.sh`; the BrainBench gate uses the separate default-profile snapshot for its in-memory PGLite; the ~42MB tar is also cached across jobs via actions/cache, with the runner's own hash check staying authoritative). CI EXCLUDES `*.serial.test.ts` from the shards and runs them across four `serial-tests` workers via `bun run test:serial` — one bun process per file preserves the `mock.module` quarantine; the pool runs those processes concurrently. `bun run verify` gets its own job too, as does the BrainBench memory-conformance gate (`brainbench` job → `scripts/ci-brainbench-gate.sh`, hermetic in-memory PGLite, ~15s), which compares HEAD's fresh run against master's committed baseline (`evals/brainbench/baselines/main.json`) — the `test-status` aggregate checks its result explicitly. E2E (`.github/workflows/e2e.yml`) always runs its applicable execution lanes, with the jsonb-parity job in front of tier2 as the token-spend gate, and aggregates through `e2e-status`. Scheduled runs also require the full-corpus lanes, including each slow suite excluded from the coverage shards (longmemeval, entity-resolve-perf, and brainbench-e2e). Both aggregates reject failures, cancellations, and unexpected skips. Dependency caches and validated PGLite snapshots remain; successful test results are never reused. CI is the ground truth for "did everything pass."
- **Local fast loop** (`scripts/run-unit-shard.sh` via the parallel wrapper) uses the same weighted partitioner as CI and EXCLUDES `*.slow.test.ts` AND `*.serial.test.ts`. Local trades coverage for inner-loop speed; CI catches what local skips.
This divergence is intentional. Don't try to make them equal — the two scripts deliberately solve different problems. The regression test at `test/scripts/run-unit-shard.test.ts` pins what the local fast loop should and shouldn't include, and that no unit-lane file spawning the CLI through `test/helpers/cli-spawn.ts` hand-pins a per-test timeout below the bunfig default (an explicit `test(name, fn, N)` ceiling overrides bun's `--timeout`, so `GBRAIN_TEST_TIMEOUT_MULTIPLIER` never reaches it — inherit the default instead; cli-spawn's own kill timer still reaps a hung child); `test/scripts/run-unit-parallel.test.ts` pins the wrapper's memory-adaptive concurrency, and the OOM/external-kill serial rescue pass, and operator-interrupt teardown (a Ctrl-C / SIGTERM to the wrapper while shards are live TERMs then KILLs every shard descendant, so a cancelled run cannot leave gtimeout/bun alive until the shard cap).
@@ -196,8 +246,8 @@ non-`GBRAIN_`-prefixed so the hermetic env scrub keeps them.
**Two corpora.**
- **PR corpus** (`prCorpus`) — the 14 coverage-collecting lanes in
`.github/workflows/test.yml`: the 10 matrix shards, `serial-tests`, and the
- **PR corpus** (`prCorpus`) — the 17 coverage-collecting lanes in
`.github/workflows/test.yml`: the 10 matrix shards, four `serial-tests` partitions, and the
three dedicated slow jobs (`slow-eval-longmemeval`,
`slow-entity-resolve-perf`, `slow-brainbench-e2e`). Deterministic (runs identically on every PR); this
is the corpus the gates run against.
@@ -213,7 +263,9 @@ non-`GBRAIN_`-prefixed so the hermetic env scrub keeps them.
repo-relative, and emits a merged lcov plus a summary JSON: src-only
totals/per-dir/per-file percentages, a `lineHits` map (the diff gate's input),
and the never-loaded src file list. `--manifest-expect lane,lane,...` pins the
expected lane set; a missing or `complete: false` manifest, an unparseable
expected lane set (`serial-1` through `serial-4` for PRs, `serial` nightly).
The merger checks commit SHA (`--sha` overrides checkout HEAD for offline
artifacts), duplicate identities, and actual per-lane LCOV counts; a missing or `complete: false` manifest, an unparseable
lcov, or a `shard` lane with `lcovCount != 1` marks the summary
`degraded: true`. Degraded is data, not failure: the merge never aborts (exit
0), and both gates print `WOULD PASS`/`WOULD FAIL` and exit 0 on a degraded
@@ -252,7 +304,7 @@ with both corpus sections unseeded. `scripts/update-coverage-baseline.ts
(per-file detail limited to the baseline's `watchlist`); `--promote` flips
`provisional: false` at graduation.
**CI wiring.** The 14 PR lanes upload `coverage-*` artifacts; the advisory
**CI wiring.** The 17 PR lanes upload `coverage-*` artifacts; the advisory
`coverage-report` job downloads + merges (`COVERAGE_CORPUS=prCorpus`), renders
`scripts/render-coverage-summary.ts` to the step summary (including the
behavioral-vs-structural counts from `scripts/structural-suites.tsv`), and
@@ -783,6 +835,14 @@ When asked to "run all E2E tests" or "run tests", that means ALL tiers:
### E2E test DB lifecycle (ALWAYS follow this)
`setupDB()` clears rows while preserving physical schema. Fixtures that seed
fixed legacy-width text vectors use `setupLegacyEmbeddingDB()` instead: it
establishes the canonical test shape after clearing the database, including
facts and takes, so a preceding CLI-init test cannot change their assumptions.
Custom-dimension and migration tests continue using ordinary `setupDB()`.
For fixtures testing schema/index creation or source-scoped cleanup, preserve
that lifecycle and derive incidental text-vector widths from the database.
You are responsible for spinning up and tearing down the test Postgres container.
Do not leave containers running after tests. Do not skip E2E tests, do not ask
permission to run them — see the "run without asking" rule above.

View File

@@ -475,13 +475,18 @@ per-release `**vX.Y.Z:**` narration — CI enforces this
- `scripts/check-jsonb-params.mjs` — AST-lite CI guard for the POSITIONAL jsonb double-encode form the template grep above misses: an `executeRaw`/`executeRawDirect`/`.unsafe()` call whose balanced arg span binds `JSON.stringify(x)` into a bare `$N::jsonb` cast. Walks each call's balanced span respecting strings/templates/comments, handles generic-typed calls (`executeRaw<T>(`), and allows the sanctioned forms (`$N::text::jsonb`, `$N::text[]`, `executeRawJsonb`, `sql.json`, an inline `jsonb-guard-ok` comment). PGLite's native `db.query` is deliberately not scanned (it parses text→jsonb, so the bug can't occur there). Heuristic by design (whole-span correlation; can't see a `JSON.stringify` assigned to a variable before the call) — the real backstop is the DATABASE_URL-gated e2e parity tests. Scan roots overridable via argv for its self-test (`test/check-jsonb-params.test.ts`).
- `scripts/check-source-id-projection.sh` — CI grep guard for the multi-source bug class. Greps `src/core/postgres-engine.ts` + `src/core/pglite-engine.ts` for `SELECT.*FROM pages` projections matching the `rowToPage` feeder shape (id + slug + type + title) and fails if `source_id` is missing. `Page.source_id` is required at the type level; a projection dropping the column produces `Page` rows with `source_id: undefined` while TypeScript's `: string` lies about it. Wired into `bun run verify`.
- `scripts/guards-manifest.tsv` + `scripts/guard-self-test.sh` — THE single registry of `scripts/check-*` CI guards (52 guards) and its self-test harness. Every guard is classified `scanner` (greps/parses repo sources — must eventually carry fixtures), `buildfresh`, or `repostate` (exempt-with-reason, not fixture-tested). `guard-self-test.sh` (`bun run check:guard-self-test`, wired into `bun run verify`) runs each `selftest=yes` scanner against known-bad (must exit non-zero) and known-good (must pass) fixture trees under `test/fixtures/guards/<guard>/{bad,good}/` via the `GBRAIN_GUARD_ROOT` env seam, and fails the build when a new `scripts/check-*` script is missing from the manifest — so a guard whose pattern rots into a permanently-green no-op fails CI instead of masquerading as coverage. The manifest registers and classifies guards but does not itself schedule them — `run-verify-parallel.sh`'s `CHECKS` array remains the execution list, and a registered guard is not automatically wired into verify. New guard = new manifest row (+ fixtures if scanner) + a `CHECKS` entry if it should gate pushes.
- `scripts/merge-lcov.ts` + `scripts/coverage-diff-gate.ts` + `scripts/coverage-baseline-gate.ts` + `scripts/update-coverage-baseline.ts` + `scripts/render-coverage-summary.ts` + `scripts/coverage-gate-exemptions.txt` + `scripts/coverage-baseline.json` — the coverage measurement + gating cluster; the operating guide is docs/TESTING.md "Coverage lanes and gates". `merge-lcov.ts` walks artifact dirs for `lcov.info` + `lane-manifest.json`, sums DA hits per file:line, normalizes paths repo-relative, and emits a merged lcov + summary JSON (src-only totals/per-dir/per-file, the `lineHits` extension the diff gate consumes, and never-loaded src files as count + sorted list — deliberately never a percentage, since physical lines ≠ executable lines); `--manifest-expect` pins the lane set, and a missing/incomplete lane or a `shard` lane with `lcovCount != 1` (the xargs-batching tripwire) marks the summary `degraded: true` — still exit 0 (degraded is data, and both gates go report-only on it). `coverage-diff-gate.ts` gates added/changed gate-scoped lines (non-test, non-generated `src/**.ts`) at ≥80% covered plus zero changed-but-never-loaded files; report-only unless `COVERAGE_GATE_ENFORCE=1`; a `[coverage-exempt: reason]` commit trailer passes with a loud warning; `coverage-gate-exemptions.txt` rows (exact path or trailing-`/` prefix; SHRINK-ONLY — additions need a graduation review) are excluded from the gate but still reported (`[e2e-exempt]` / `[subprocess-undercount]`); exit contract: 0 = pass or report-only, 1 = fail while enforcing, 2 = infrastructure error (never conflated with a coverage verdict). `coverage-baseline-gate.ts` reads the baseline via `git show origin/master:scripts/coverage-baseline.json` (never the working tree, so a PR can't weaken its own bar) and compares corpus-matched sections only (`--corpus prCorpus|fullCorpus`), failing on >0.5pp global or >1.0pp per-dir drops; `provisional: true` in the baseline (the current state — both corpus sections unseeded) keeps it report-only regardless of enforcement; `update-coverage-baseline.ts` writes the working-tree baseline (per-file detail limited to the committed `watchlist`) and `--promote` flips `provisional: false`. `render-coverage-summary.ts` renders the summary JSON as markdown on stdout for `$GITHUB_STEP_SUMMARY`, including the behavioral-vs-structural counts from `scripts/structural-suites.tsv`. Wiring: 14 PR-corpus lanes in test.yml (10 matrix shards + serial + the three dedicated slow jobs) upload `coverage-*` artifacts and the advisory `coverage-report` job merges + renders + runs both gates report-only (deliberately absent from `test-status`/`cache-write` until graduation); schedule-only `coverage-full-{unit,serial,slow,e2e}` + `coverage-full-report` in e2e.yml produce the self-contained nightly fullCorpus number (full e2e glob included) and the `coverage-full-merged` trend artifact. Collection is `COVERAGE_DIR`-opt-in in `test-shard.sh`/`run-serial-tests.sh`/`run-e2e.sh` — unique coverage dir per bun process (a reused dir overwrites `lcov.info`), lane manifest written only on a green run, `run-e2e.sh` requires an ABSOLUTE `COVERAGE_DIR` and honors `E2E_FILE_TIMEOUT_SECS` (both deliberately non-`GBRAIN_`-prefixed to survive the hermetic env scrub). Bun/JSC emits line records only (function coverage is informational) and no subprocess coverage, so `src/cli.ts` undercounts. Pinned by `test/scripts/merge-lcov.test.ts`, `test/scripts/coverage-diff-gate.test.ts`, `test/scripts/render-coverage-summary.test.ts`.
- `scripts/merge-lcov.ts` + `scripts/coverage-diff-gate.ts` + `scripts/coverage-baseline-gate.ts` + `scripts/update-coverage-baseline.ts` + `scripts/render-coverage-summary.ts` + `scripts/coverage-gate-exemptions.txt` + `scripts/coverage-baseline.json` — the coverage measurement + gating cluster; the operating guide is docs/TESTING.md "Coverage lanes and gates". `merge-lcov.ts` walks artifact dirs for `lcov.info` + `lane-manifest.json`, sums DA hits per file:line, normalizes paths repo-relative, and emits a merged lcov + summary JSON (src-only totals/per-dir/per-file, the `lineHits` extension the diff gate consumes, and never-loaded src files as count + sorted list — deliberately never a percentage, since physical lines ≠ executable lines); `--manifest-expect` pins the lane set; SHA, duplicate identities, and actual per-lane LCOV counts are validated (`--sha` supports offline artifacts), and a missing/incomplete lane or a `shard` lane with `lcovCount != 1` (the xargs-batching tripwire) marks the summary `degraded: true` — still exit 0 (degraded is data, and both gates go report-only on it). `coverage-diff-gate.ts` gates added/changed gate-scoped lines (non-test, non-generated `src/**.ts`) at ≥80% covered plus zero changed-but-never-loaded files; report-only unless `COVERAGE_GATE_ENFORCE=1`; a `[coverage-exempt: reason]` commit trailer passes with a loud warning; `coverage-gate-exemptions.txt` rows (exact path or trailing-`/` prefix; SHRINK-ONLY — additions need a graduation review) are excluded from the gate but still reported (`[e2e-exempt]` / `[subprocess-undercount]`); exit contract: 0 = pass or report-only, 1 = fail while enforcing, 2 = infrastructure error (never conflated with a coverage verdict). `coverage-baseline-gate.ts` reads the baseline via `git show origin/master:scripts/coverage-baseline.json` (never the working tree, so a PR can't weaken its own bar) and compares corpus-matched sections only (`--corpus prCorpus|fullCorpus`), failing on >0.5pp global or >1.0pp per-dir drops; `provisional: true` in the baseline (the current state — both corpus sections unseeded) keeps it report-only regardless of enforcement; `update-coverage-baseline.ts` writes the working-tree baseline (per-file detail limited to the committed `watchlist`) and `--promote` flips `provisional: false`. `render-coverage-summary.ts` renders the summary JSON as markdown on stdout for `$GITHUB_STEP_SUMMARY`, including the behavioral-vs-structural counts from `scripts/structural-suites.tsv`. Wiring: 17 PR-corpus lanes in test.yml (10 matrix shards + four serial partitions + the three dedicated slow jobs) upload `coverage-*` artifacts and the advisory `coverage-report` job merges + renders + runs both gates report-only (deliberately absent from `test-status`/`cache-write` until graduation); schedule-only `coverage-full-{unit,serial,slow,e2e}` + `coverage-full-report` in e2e.yml produce the self-contained nightly fullCorpus number (full e2e glob included) and the `coverage-full-merged` trend artifact. Collection is `COVERAGE_DIR`-opt-in in `test-shard.sh`/`run-serial-tests.sh`/`run-e2e.sh` — unique coverage dir per bun process (a reused dir overwrites `lcov.info`), lane manifest written only on a green run, `run-e2e.sh` requires an ABSOLUTE `COVERAGE_DIR` and honors `E2E_FILE_TIMEOUT_SECS` (both deliberately non-`GBRAIN_`-prefixed to survive the hermetic env scrub). Bun/JSC emits line records only (function coverage is informational) and no subprocess coverage, so `src/cli.ts` undercounts. Pinned by `test/scripts/merge-lcov.test.ts`, `test/scripts/coverage-diff-gate.test.ts`, `test/scripts/render-coverage-summary.test.ts`.
- `scripts/check-module-size.sh` + `scripts/module-size-limits.tsv` — the module-size ratchet (`bun run check:module-size`, wired into `bun run verify`). The TSV commits a per-file `wc -l` ceiling (`path max_lines policy note`); four rules, all violations reported before a single exit 1: a file above its ceiling fails (raise a ceiling only as a conscious TSV edit); a ceiling more than 50 lines above the measured size fails (stale slack after a shrink — lower it so the ratchet holds); a TSV row whose path no longer exists fails (remove the row); an unlisted `src/**/*.ts` (excluding `*.generated.ts`/`*.test.ts`) above the 1500-line new-file cap fails (split it or add a row). Policy `region-exempt` (only `src/core/migrate.ts`) counts lines OUTSIDE the append-only `export const MIGRATIONS = [` … `];` region, so the migrations array grows freely while the surrounding runner logic stays ratcheted. Self-test seams: `GBRAIN_GUARD_ROOT`, `GBRAIN_MODULE_SIZE_SLACK`, `GBRAIN_MODULE_SIZE_NEWFILE_CAP`.
- `scripts/classify-tests.ts` + `scripts/structural-suites.tsv` — suite-level behavioral-vs-structural test classification (the intent axis described in docs/TESTING.md "File taxonomy"). Content-based detectors — repo-anchored `readFileSync`/`Bun.file` readers, exec-scan grep windows over `src|scripts|docs`, and the `doctorSource()`/`doctorFileSource()` helpers — mark a suite STRUCTURAL when its assertions read repo source/doc text rather than executing product code; tmpdir-anchored reads don't count, and files with detectors but no attributable suite land in an `unknown` bucket emitted as comment rows (surfaced, never silently dropped). Modes: bare = rewrite the TSV; `--check` = byte-for-byte regenerate-and-diff freshness (wired as `bun run check:structural-manifest` in `bun run verify` via `scripts/check-structural-manifest.sh`); `--summary` = counts only. Fix misclassifications in the detector list, never by hand-editing the TSV. `render-coverage-summary.ts` consumes the TSV for the behavioral-vs-structural line in the CI coverage report.
- `scripts/build-pglite-snapshot.ts` — `bun run build:pglite-snapshot`: bakes a post-`initSchema()` PGLite data dir into `test/fixtures/pglite-snapshot.tar` + a version file (schema hash line, then `dims=`/`model=` lines recording the embedding shape it was baked with). Idempotent (hash short-circuit ~40ms when fresh; rebuilds stale) and concurrency-safe (atomic `mkdir` lock at `test/fixtures/.pglite-snapshot.lock` with staleness-verified takeover — a live lock is never stolen; tar written first, version file last, so a crash can't leave a fresh-looking torn fixture; waiter bounded by `GBRAIN_SNAPSHOT_LOCK_TIMEOUT_MS`, default 120000; an exhausted waiter facing a still-live lock proceeds unlocked as a last resort — the loader's hash/shape gate validates the version file, not the tar bytes). Called through the shared `ensure_pglite_snapshot` helper in `scripts/lib/test-env.sh` (also home of `detect_cpus` + `detect_available_mem_mb`; sourced by `run-unit-parallel.sh`, `test-shard.sh`, `run-slow-tests.sh`, `run-serial-tests.sh`, `run-verify-parallel.sh`, and `run-e2e.sh` — default-on, opt out `GBRAIN_NO_SNAPSHOT=1`, no-op when a parent already exported the path, non-fatal on build failure with a one-line "active" echo so a silent cold-init fallback stays visible) and directly by `scripts/ci-local.sh`; every caller exports `GBRAIN_PGLITE_SNAPSHOT`. The loader side is `tryLoadSnapshot` + `computeSnapshotSchemaHash` (exported from `src/core/pglite-engine.ts`): the coverage-immune hash reads raw file bytes for the schema/migration entry modules and imported helpers, including grant SQL/policy/repair dependencies and withdrawal triggers (keep its dependency list and CI cache keys aligned when adding a helper); any hash or embedding-shape mismatch warns once and falls through to normal cold init — the snapshot is an optimization, never authoritative. The loader memoizes per process: the schema hash computes once (source bytes are static for the process lifetime) and the version file + ~42MB tar are read once per (path, process) instead of once per engine construction (a full suite constructs 600+ engines); a terminally-unusable path (missing/stale/torn) memoizes as null and is never retried, the tar blob loads lazily only after the FIRST caller passes the shape gate, and the dims/model shape gate itself is deliberately NOT memoized (tests reconfigure the gateway mid-process — a mismatched engine must still fall back to cold init). Accepted limitation: a snapshot rewritten mid-process is not observed; the only writer runs before test fan-out. Test seams `__snapshotMemoStatsForTests`/`__resetSnapshotMemoForTests`. Pinned by `test/snapshot-shape-guard.test.ts`.
- `docker-compose.ci.yml` + `scripts/ci-local.sh` — Local CI gate. `bun run ci:local` spins up four `pgvector/pgvector:pg16` services (postgres-1..4) + `oven/bun:1` with named volumes (`gbrain-ci-pg-data-{1..4}`, `gbrain-ci-node-modules`, `gbrain-ci-bun-cache`), runs gitleaks on host, smoke-tests `scripts/run-e2e.sh` argv handling, runs guards + typecheck, then the Tier 1 default: 4-shard parallel unit + E2E (`xargs -P4`, one Postgres per shard; unit phase keeps `DATABASE_URL` unset). `--no-shard` falls back to the legacy unsharded sequential flow (debug aid); `--diff` runs the diff-aware selector unsharded. Also runs a `pgbouncer` service (`edoburu/pgbouncer`, `POOL_MODE: transaction`, `AUTH_TYPE: plain` — pg16 stores SCRAM verifiers, so the userlist must hold the plaintext password; `IGNORE_STARTUP_PARAMETERS` whitelists gbrain's `statement_timeout`/`idle_in_transaction_session_timeout` startup params the way the Supabase pooler does) fronting postgres-1 on host port `GBRAIN_CI_PGBOUNCER_PORT` (default 6543); every E2E invocation exports `GBRAIN_PGBOUNCER_URL` (pooled; dedicated `gbrain_pgbouncer` database so it never races the `gbrain_test` TRUNCATE fixtures) + `GBRAIN_PGBOUNCER_DIRECT_URL`, consumed by `test/e2e/pgbouncer-teardown.test.ts` — which reproduces the transaction-mode teardown failure in the local gate. `--no-pull` skips upstream pulls; `--clean` nukes named volumes. Postgres host port defaults to 5434; override with `GBRAIN_CI_PG_PORT=NNNN`. Stronger gate than PR CI's 2-file Tier 1 set.
- `scripts/build-pglite-snapshot.ts` — `bun run build:pglite-snapshot [--profile legacy|default]` bakes post-schema PGLite snapshots. Legacy (the default) matches the unit preload; default follows canonical CLI embedding defaults. Profiles have disjoint tar/version/lock paths and CI cache keys. Version sidecars retain schema-byte hash, dims and model. The builder requires PID/token lock ownership, only reclaims confirmed dead owners using its retirement protocol, times out on live/unknown locks, and publishes temporary files via atomic rename with the version last. Normal release and crash recovery retain the same nonempty owner tombstone so a delayed observer cannot move a replacement lock. `scripts/lib/test-env.sh` exposes `ensure_pglite_snapshot` (legacy primary env) and `ensure_default_pglite_snapshot` (absolute auxiliary `GBRAIN_TEST_DEFAULT_SNAPSHOT`); failure visibly falls back to cold init. `GBRAIN_NO_SNAPSHOT=1` clears both and survives preloads. BrainBench overrides only CLI children, including run-all, to the default path. The production loader's per-process memoization, schema/shape refusal and in-memory-only gate remain authoritative and unchanged. Tests: `test/scripts/build-pglite-snapshot.test.ts`, `test/snapshot-shape-guard.test.ts`, BrainBench CLI E2E and operator preload tests.
- `docker-compose.ci.yml` + `scripts/ci-local.sh` — Local CI gate. `bun run ci:local` spins up four `pgvector/pgvector:pg16` services (postgres-1..4) + `oven/bun:${GBRAIN_CI_BUN_TAG:-1.3.13}` with named volumes (`gbrain-ci-pg-data-{1..4}`, `gbrain-ci-node-modules`, `gbrain-ci-bun-cache`), runs gitleaks on host, smoke-tests `scripts/run-e2e.sh` argv handling, runs guards + typecheck, then the Tier 1 default: 4-shard parallel unit + E2E (`xargs -P4`, one Postgres per shard; unit phase keeps `DATABASE_URL` unset). `--no-shard` falls back to the legacy unsharded sequential flow (debug aid); `--diff` runs the diff-aware selector unsharded. Also runs a `pgbouncer` service (`edoburu/pgbouncer`, `POOL_MODE: transaction`, `AUTH_TYPE: plain` — pg16 stores SCRAM verifiers, so the userlist must hold the plaintext password; `IGNORE_STARTUP_PARAMETERS` whitelists gbrain's `statement_timeout`/`idle_in_transaction_session_timeout` startup params the way the Supabase pooler does) fronting postgres-1 on host port `GBRAIN_CI_PGBOUNCER_PORT` (default 6543); every E2E invocation exports `GBRAIN_PGBOUNCER_URL` (pooled; dedicated `gbrain_pgbouncer` database so it never races the `gbrain_test` TRUNCATE fixtures) + `GBRAIN_PGBOUNCER_DIRECT_URL`, consumed by `test/e2e/pgbouncer-teardown.test.ts` — which reproduces the transaction-mode teardown failure in the local gate. `--no-pull` skips upstream pulls; `--clean` nukes named volumes. Postgres host port defaults to 5434; override with `GBRAIN_CI_PG_PORT=NNNN`. Stronger gate than PR CI's 2-file Tier 1 set.
- `scripts/select-e2e.ts` + `scripts/e2e-test-map.ts` — Diff-aware E2E test selector. Reads three git sources (committed `origin/master...HEAD`, working-tree `HEAD`, and `git ls-files --others --exclude-standard` for untracked NOT-gitignored files), classifies as EMPTY / DOC_ONLY / SRC. Fail-closed: EMPTY → all files; DOC_ONLY (every path matches the README/CLAUDE/AGENTS/CHANGELOG/TODOS allowlist) → empty stdout; SRC → escape-hatch paths (schema, package.json, skills/) trigger all, else the hand-tuned `E2E_TEST_MAP` glob narrows, and an unmapped src/ change still emits ALL files (never silently nothing). Pure-function exports `selectTests`, `classify`, `matchGlob`. `bun run ci:select-e2e` prints the current selection on stdout. `test/select-e2e.test.ts` covers all 4 branches plus 3 guards (skills/, untracked files, unmapped src/) — 24 cases.
- `scripts/run-e2e.sh` — Sequential E2E runner. Accepts an optional argv-driven file list (used by `ci:local:diff`) and a `--dry-run-list` flag that prints the resolved file list and exits (used by `ci-local.sh`'s startup smoke-test). Falls back to `test/e2e/*.test.ts` plus `test/phantom-redirect-engine-parity.test.ts` when invoked with no args (the phantom-redirect Postgres arm is only reachable through a DATABASE_URL-bearing lane; the unit wrappers strip the URL, so this lane must carry it). This wrapper is the database-URL opt-in boundary: it exports `GBRAIN_TEST_ALLOW_DATABASE_URL=1` so the bunfig preload guard (`test/helpers/database-url-guard-preload.ts`) lets the run start, unsets `GBRAIN_DATABASE_URL` (the e2e suite runs on `DATABASE_URL` only — an ambient `GBRAIN_DATABASE_URL` would pass the opt-in yet reach CLI-subprocess paths with no name floor), and its GBRAIN_* env scrub preserves `GBRAIN_E2E_ALLOW_DB` so the name-floor escape hatch the guard's own error message names stays usable. It also preserves `GBRAIN_CI_DISABLE_TEST_ENV_FILE=1`, keeping CI runs from loading checkout-local `.env.testing` credentials after the shell-to-Bun handoff. It also exports `GBRAIN_TEST_KEEP_PROVIDER_KEYS=1` so the unit-lane provider-key strip preload (`test/helpers/provider-keys-preload.ts`) leaves the real keys that live embed/parity e2e tests skip-gate on. Each file runs under a gtimeout/timeout wedge backstop (default signal: SIGTERM — the bun test child installs no JS-level handler for it, so kernel-default termination applies) — 180s default, with a per-file override for known-slow files (`skills.test.ts` gets 420s: the real ingest-skill run replays every migration, so its floor grows as master adds migrations); the cap is a wedge backstop, not a per-test budget, and bare `bun` (no outer cap) is the fallback when neither timeout binary is installed. It activates the PGLite schema snapshot like the other runners: sources `scripts/lib/test-env.sh` + `ensure_pglite_snapshot` after the `--dry-run-list` early exit (list mode stays instant; non-fatal on build failure), re-exports `GBRAIN_PGLITE_SNAPSHOT` as an ABSOLUTE path (e2e tests spawn CLI subprocesses with varying cwd — a relative path silently misses the tar there), and its env keep-list preserves the var; e2e files that assert the path TO post-`initSchema()` state carry per-file `delete process.env.GBRAIN_PGLITE_SNAPSHOT` opt-outs with one-line reasons.
- `scripts/sharding.ts` + `scripts/mine-shard-weights.ts` + `scripts/capture-test-log.ts` — shared deterministic weighted partitioning and timing refresh. CLI accepts lane-specific weights; missing entries use p75, equal loads prefer fewer files then shard index, and serial may explicitly opt into warning/fallback on bad advisory weights. `weights:mine --lane unit|serial|e2e` validates completed lane logs before replacement and records run/commit/unit metadata. Capture preserves stdout/stderr, records timestamped GitHub-format artifacts, forwards cancellation to its owned process group and marks failed runs unmineable. Unit and E2E artifact retention is 14 days.
- `scripts/e2e-matrix.ts` — freezes selected E2E paths after existing named/live exclusions into at most four weighted partitions. Explicit empty sentinel is the only no-op; malformed/missing input fails. Workers validate paths and filesystem confinement, pass argv directly, clear inherited SHARD and forward cancellation. `prepare-e2e` and every worker are required by e2e-status; each worker owns Postgres. Pinned by `test/scripts/e2e-matrix.test.ts` and workflow wiring tests.
- `test/e2e/helpers.ts` — shared guarded test database lifecycle. `setupDB()` preserves physical schema; fixed-width vector fixtures explicitly use `setupLegacyEmbeddingDB()` to align empty text-embedding columns with canonical legacy test configuration after truncation. This prevents earlier CLI-init fixtures from leaking a different width into weighted execution order. Custom-shape migration fixtures retain ordinary setup. `test/db-guard-coverage.test.ts` recognizes both guarded entry points.
- `scripts/llms-config.ts` + `scripts/build-llms.ts` — Generator for `llms.txt` (llmstxt.org-spec web index) + `llms-full.txt` (inlined single-fetch bundle). Curated config drives both. Run `bun run build:llms` after adding a new doc. `LLMS_REPO_BASE` env lets forks regenerate with their own URL base. `FULL_SIZE_BUDGET` (600KB) caps the inline bundle; generator WARNs if exceeded. Committed output has no runtime consumer; committed for GitHub browsing and fork-safe fetching.
- `AGENTS.md` — Local-clone entry point for non-Claude agents (Codex, Cursor, OpenClaw, Aider). Mirrors `CLAUDE.md` intent via relative links. Claude Code keeps using `CLAUDE.md`.
- `docs/UPGRADING_DOWNSTREAM_AGENTS.md` — Patches for downstream agent skill forks to apply when upgrading. Each release appends a new section; includes diffs for brain-ops, meeting-ingestion, signal-detector, enrich.

View File

@@ -1,7 +1,7 @@
{
"id": "gbrain-context-engine",
"name": "gbrain",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Personal knowledge brain with Postgres + pgvector hybrid search",
"family": "bundle-plugin",
"configSchema": {

View File

@@ -42,6 +42,7 @@
"build:llms": "bun run scripts/build-llms.ts",
"wave-security-scan": "bash scripts/wave-security-scan.sh",
"build:flag-registry": "bun run scripts/generate-flag-registry.ts",
"weights:mine": "bun run scripts/mine-shard-weights.ts",
"build:pglite-snapshot": "bun run scripts/build-pglite-snapshot.ts",
"test": "bash scripts/run-unit-parallel.sh",
"eval:autocut": "bun test test/search/autocut-eval.test.ts",
@@ -174,7 +175,7 @@
"bun": ">=1.3.11"
},
"license": "MIT",
"version": "0.50.2.0",
"version": "0.50.4.0",
"overrides": {
"@ai-sdk/provider-utils": "4.0.33",
"@hono/node-server": "^2.0.5",

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain-coding",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Brain-first coding agent working inside a repo: retrieval, routing, ingest discipline, correction hygiene. Default persona for the claude-code harness bridge; also published as the gbrain-coding marketplace variant. (persona variant of the gbrain plugin — 20 skills)",
"author": {
"name": "Garry Tan",

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain-coding",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Brain-first coding agent working inside a repo: retrieval, routing, ingest discipline, correction hygiene. Default persona for the claude-code harness bridge; also published as the gbrain-coding marketplace variant. (persona variant of the gbrain plugin — 20 skills)",
"author": {
"name": "Garry Tan",

View File

@@ -1,4 +1,4 @@
<!-- gbrain-plugin-tree-stamp: 0.50.2.0 -->
<!-- gbrain-plugin-tree-stamp: 0.50.4.0 -->
# gbrain-coding (generated persona variant — do not hand-edit)
Brain-first coding agent working inside a repo: retrieval, routing, ingest discipline, correction hygiene. Default persona for the claude-code harness bridge; also published as the gbrain-coding marketplace variant.

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain-daily",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Personal knowledge-brain daily use: meetings, tasks, briefings, reading, research. Published as the gbrain-daily marketplace variant. (persona variant of the gbrain plugin — 19 skills)",
"author": {
"name": "Garry Tan",

View File

@@ -1,6 +1,6 @@
{
"name": "gbrain-daily",
"version": "0.50.2.0",
"version": "0.50.4.0",
"description": "Personal knowledge-brain daily use: meetings, tasks, briefings, reading, research. Published as the gbrain-daily marketplace variant. (persona variant of the gbrain plugin — 19 skills)",
"author": {
"name": "Garry Tan",

View File

@@ -1,4 +1,4 @@
<!-- gbrain-plugin-tree-stamp: 0.50.2.0 -->
<!-- gbrain-plugin-tree-stamp: 0.50.4.0 -->
# gbrain-daily (generated persona variant — do not hand-edit)
Personal knowledge-brain daily use: meetings, tasks, briefings, reading, research. Published as the gbrain-daily marketplace variant.

View File

@@ -1,4 +1,4 @@
<!-- gbrain-plugin-tree-stamp: 0.50.2.0 -->
<!-- gbrain-plugin-tree-stamp: 0.50.4.0 -->
# gbrain plugin skill tree (generated — do not hand-edit)
This tree is the curated skill set for the gbrain Codex and Claude Code

View File

@@ -1,162 +1,180 @@
#!/usr/bin/env bun
// scripts/build-pglite-snapshot.ts
//
// TZ pinned to UTC BEFORE any PGLite work: dumpDataDir bakes this process's
// TimeZone into the tar's cluster defaults. Building under the host zone made
// restored engines run sessions in the build machine's zone (the engine also
// re-pins at restore — this is the belt to that suspender, and it keeps any
// OTHER zone-derived state baked into the tar deterministic across hosts).
process.env.TZ = 'UTC';
//
// Tier 3 fast-restore: boot a fresh PGLite, run the full initSchema (forward
// bootstrap + PGLITE_SCHEMA_SQL + every migration), dump the post-init state
// to a tar fixture. Test files that read GBRAIN_PGLITE_SNAPSHOT can skip the
// 1-3 seconds of cold init and load the post-schema state directly.
//
// Output: test/fixtures/pglite-snapshot.tar (binary, gitignored)
// test/fixtures/pglite-snapshot.version (SHA256 of schema/migration source FILE BYTES, including imported helpers)
//
// The version file lets the engine detect snapshot staleness — if the tar's
// recorded version doesn't match the current schema-file hash, the engine
// ignores the snapshot and runs a normal initSchema.
//
// Run: bun run scripts/build-pglite-snapshot.ts
// (or: bun run build:pglite-snapshot)
//
// Re-run whenever schema SQL or a migration helper changes.
/** Build validated schema fixtures; legacy is the unit-test shape, default the bare CLI shape. */
import { existsSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, renameSync, rmSync, writeFileSync } from 'node:fs';
import * as fsModule from 'node:fs';
import * as crypto from 'node:crypto';
import { join } from 'node:path';
import { hostname, tmpdir } from 'node:os';
import { configureGateway } from '../src/core/ai/gateway.ts';
import { DEFAULT_EMBEDDING_DIMENSIONS, DEFAULT_EMBEDDING_MODEL } from '../src/core/ai/defaults.ts';
import { PGLiteEngine, computeSnapshotSchemaHash } from '../src/core/pglite-engine.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../test/helpers/legacy-embedding-config.ts';
import { writeFileSync, mkdirSync, existsSync, readFileSync, rmdirSync, rmSync, mkdtempSync, statSync } from "node:fs";
import { dirname, join } from "node:path";
import { tmpdir } from "node:os";
import * as crypto from "node:crypto";
import * as fsModule from "node:fs";
export type SnapshotProfile = 'legacy' | 'default';
import { configureGateway, getEmbeddingDimensions, getEmbeddingModel } from "../src/core/ai/gateway.ts";
import { LEGACY_EMBEDDING_CONFIG } from "../test/helpers/legacy-embedding-config.ts";
import { PGLiteEngine, computeSnapshotSchemaHash } from "../src/core/pglite-engine.ts";
function computeSchemaHash(): string {
// File-bytes hash (see computeSnapshotSchemaHash) — identical between this
// plain-`bun run` builder and a coverage-instrumented test process.
const h = computeSnapshotSchemaHash(crypto, fsModule);
if (!h) throw new Error('build-pglite-snapshot: cannot read schema/migration source dependencies — run from a source checkout');
return h;
export function parseSnapshotProfile(args: string[]): SnapshotProfile {
if (args.length === 0) return 'legacy';
if (args.length === 2 && args[0] === '--profile' && (args[1] === 'legacy' || args[1] === 'default')) return args[1];
throw new Error('Usage: build-pglite-snapshot.ts [--profile legacy|default]');
}
async function main() {
const fixturePath = "test/fixtures/pglite-snapshot.tar";
const versionPath = "test/fixtures/pglite-snapshot.version";
const lockPath = "test/fixtures/.pglite-snapshot.lock";
mkdirSync(dirname(fixturePath), { recursive: true });
// W0 fix-wave: build under the EXACT embedding shape the test suite pins.
// bunfig.toml preloads test/helpers/legacy-embedding-preload.ts, which
// configures the gateway to the shared LEGACY_EMBEDDING_CONFIG (OpenAI
// 1536-d) for every `bun test` file — so the snapshot's baked vector(dims)
// columns MUST match that shape, not the builder machine's ambient config
// (nor the shipped 1280-d default an unconfigured gateway falls back to).
// Set in main(), not module scope: ESM hoists imports, so module-scope
// placement implied an ordering it never had — config reads are lazy.
configureGateway({ ...LEGACY_EMBEDDING_CONFIG, env: { ...process.env } });
const schemaHash = computeSchemaHash();
// W0 fix-wave (Tier-1 #16): idempotent short-circuit. Runners now call this
// script UNCONDITIONALLY (build-if-missing left stale-but-present snapshots
// permanently on the warn+slow path); a fresh snapshot exits in ~ms.
const isFresh = () => {
if (!existsSync(fixturePath) || !existsSync(versionPath)) return false;
const lines = readFileSync(versionPath, "utf-8").trim().split("\n");
return lines[0] === schemaHash
&& lines[1] === `dims=${getEmbeddingDimensions()}`
&& lines[2] === `model=${getEmbeddingModel()}`;
export function snapshotProfile(profile: SnapshotProfile, fixtureDir = 'test/fixtures') {
const stem = profile === 'legacy' ? 'pglite-snapshot' : 'pglite-snapshot-default';
return {
shape: profile === 'legacy' ? LEGACY_EMBEDDING_CONFIG : {
embedding_model: DEFAULT_EMBEDDING_MODEL,
embedding_dimensions: DEFAULT_EMBEDDING_DIMENSIONS,
},
tar: join(fixtureDir, `${stem}.tar`),
version: join(fixtureDir, `${stem}.version`),
lock: join(fixtureDir, `.${stem}.lock`),
};
if (isFresh()) {
console.log(`[build-pglite-snapshot] up to date (hash ${schemaHash.slice(0, 16)}...) — nothing to do`);
return;
}
type LockIdentity = { platform: NodeJS.Platform; hostname: string; pidNamespace: string | null };
type LockOwner = { pid: number; token: string; protocol?: number } & Partial<LockIdentity>;
export function snapshotLockIdentity(): LockIdentity {
let pidNamespace: string | null = null;
if (process.platform === 'linux') {
try { pidNamespace = readlinkSync('/proc/self/ns/pid'); } catch { /* unknown, never reclaim */ }
}
return { platform: process.platform, hostname: hostname(), pidNamespace };
}
// GBRAIN_HOME isolation is only needed once we actually BUILD (the engine
// boot reads config). Red-team catch: creating it before the isFresh()
// short-circuit leaked one temp dir per invocation on the COMMON path
// (this script runs on every `bun run test`).
const hermeticHome = mkdtempSync(join(tmpdir(), "gbrain-snapshot-hermetic-"));
process.env.GBRAIN_HOME = hermeticHome;
// W0 fix-wave (D5.8): concurrency lock. Parallel shard runners / concurrent
// Conductor workspaces invoking this simultaneously must not tear the tar.
// mkdir is atomic; the loser polls until the winner finishes, then
// re-checks freshness and exits.
let ownLock = false;
function readOwner(lock: string): LockOwner | null {
try {
mkdirSync(lockPath);
ownLock = true;
} catch {
console.log(`[build-pglite-snapshot] another builder holds ${lockPath}; waiting...`);
const timeoutMs = Number(process.env.GBRAIN_SNAPSHOT_LOCK_TIMEOUT_MS) || 120_000;
const deadline = Date.now() + timeoutMs;
while (existsSync(lockPath) && Date.now() < deadline) {
await new Promise(r => setTimeout(r, 250));
}
if (isFresh()) {
console.log(`[build-pglite-snapshot] concurrent builder finished; snapshot fresh`);
return;
}
// Stale lock (crashed builder) or still-stale snapshot: TAKE OVER.
// W0 ship-review catch: mkdirSync on a still-existing dir always throws
// EEXIST — the original retry could never acquire, so a single crashed
// builder left every future rebuild waiting the full deadline and then
// proceeding UNLOCKED forever (the stale dir was never removed).
// Red-team refinement: verify STALENESS (lock dir mtime older than the
// full wait window) before the rmdir — two exhausted waiters would
// otherwise each rmdir+mkdir and the second would steal the first's
// just-created LIVE lock, re-opening the torn-tar window.
try {
if (existsSync(lockPath)) {
const ageMs = Date.now() - statSync(lockPath).mtimeMs;
if (ageMs > timeoutMs) {
console.log(`[build-pglite-snapshot] stale lock (age ${Math.round(ageMs / 1000)}s > ${Math.round(timeoutMs / 1000)}s) — taking over`);
rmdirSync(lockPath);
}
}
mkdirSync(lockPath);
ownLock = true;
} catch { /* lock is LIVE (fresh mtime) or takeover raced; proceed unlocked as last resort */ }
}
const value = JSON.parse(readFileSync(join(lock, 'owner.json'), 'utf8'));
return Number.isSafeInteger(value.pid) && value.pid > 0 && typeof value.token === 'string' && value.token ? value : null;
} catch { return null; }
}
function isConfirmedDead(owner: LockOwner): boolean {
const here = snapshotLockIdentity();
// Older builders removed their lock on normal release. A delayed observer
// cannot safely reclaim those records after another owner acquires the path.
if (owner.protocol !== 1) return false;
// Host and container share the checkout, but their PIDs need not refer to
// the same processes. Missing/foreign identities are unknown, never dead.
if (!here.hostname || (here.platform === 'linux' && !here.pidNamespace) ||
owner.platform !== here.platform || owner.hostname !== here.hostname || owner.pidNamespace !== here.pidNamespace) return false;
try { process.kill(owner.pid, 0); return false; }
catch (err) { return (err as NodeJS.ErrnoException).code === 'ESRCH'; }
}
/** An observed owner may be stale: concurrent reapers must never move its replacement. */
export function reclaimDeadSnapshotOwner(lock: string, owner: LockOwner | null = readOwner(lock)): boolean {
if (!owner || !isConfirmedDead(owner)) return false;
return retireSnapshotOwner(lock, owner);
}
function retireSnapshotOwner(lock: string, owner: LockOwner): boolean {
// Normal release and every reaper target the SAME tombstone. It remains
// nonempty forever, so atomic directory rename cannot overwrite it. A late
// observer cannot move a new live lock after either release or recovery.
// No separate reaper mutex exists to become stranded by cancellation. These
// tiny ownership records are gitignored; do not delete them while builders run.
const tokenHash = crypto.createHash('sha256').update(owner.token).digest('hex');
const tombstone = `${lock}.dead-${tokenHash}`;
try {
console.log(`[build-pglite-snapshot] schema hash: ${schemaHash.slice(0, 16)}...`);
console.log(`[build-pglite-snapshot] booting PGLite (in-memory)...`);
const engine = new PGLiteEngine();
// Bypass the env-aware short-circuit: we WANT a real init here.
delete process.env.GBRAIN_PGLITE_SNAPSHOT;
await engine.connect({});
console.log(`[build-pglite-snapshot] running initSchema (forward bootstrap + full migration replay)...`);
const t0 = Date.now();
await engine.initSchema();
console.log(`[build-pglite-snapshot] initSchema completed in ${Date.now() - t0}ms`);
console.log(`[build-pglite-snapshot] dumping data dir...`);
const dump = await engine.db.dumpDataDir("none");
const buffer = Buffer.from(await dump.arrayBuffer());
// Write tar first, version LAST — the version file is the commit point, so
// a crash between the writes leaves a stale-hash (ignored) snapshot, never
// a fresh-looking torn one. Lines 2-3 record the embedding shape the
// snapshot was baked with; the loader refuses a shape-mismatched snapshot
// (the W0 1280-vs-1536 incident class).
writeFileSync(fixturePath, buffer);
writeFileSync(versionPath, `${schemaHash}\ndims=${getEmbeddingDimensions()}\nmodel=${getEmbeddingModel()}\n`);
await engine.disconnect();
console.log(`[build-pglite-snapshot] wrote ${fixturePath} (${buffer.length} bytes)`);
console.log(`[build-pglite-snapshot] wrote ${versionPath}`);
} finally {
if (ownLock) { try { rmdirSync(lockPath); } catch { /* best effort */ } }
try { rmSync(hermeticHome, { recursive: true, force: true }); } catch { /* best effort */ }
renameSync(lock, tombstone);
return true;
} catch (err) {
if (['ENOENT', 'EEXIST', 'ENOTEMPTY'].includes((err as NodeJS.ErrnoException).code ?? '')) return false;
throw err;
}
}
await main();
async function bakeData(shape: ReturnType<typeof snapshotProfile>['shape']): Promise<Uint8Array> {
const scratch = mkdtempSync(join(tmpdir(), 'gbrain-snapshot-hermetic-'));
const saved = { home: process.env.GBRAIN_HOME, snapshot: process.env.GBRAIN_PGLITE_SNAPSHOT, tz: process.env.TZ };
const engine = new PGLiteEngine();
try {
process.env.GBRAIN_HOME = scratch;
process.env.TZ = 'UTC';
delete process.env.GBRAIN_PGLITE_SNAPSHOT;
configureGateway({ ...shape, env: {} });
await engine.connect({});
await engine.initSchema();
return new Uint8Array(await (await engine.db.dumpDataDir('none')).arrayBuffer());
} finally {
try { await engine.disconnect(); }
finally {
for (const [key, value] of [['GBRAIN_HOME', saved.home], ['GBRAIN_PGLITE_SNAPSHOT', saved.snapshot], ['TZ', saved.tz]] as const) {
if (value === undefined) delete process.env[key]; else process.env[key] = value;
}
rmSync(scratch, { recursive: true, force: true });
}
}
}
/** The byte-producing seam lets lock/publication tests run without booting WASM. */
export async function buildPgliteSnapshot(profile: SnapshotProfile, opts: {
fixtureDir?: string;
lockTimeoutMs?: number;
buildData?: typeof bakeData;
log?: (line: string) => void;
} = {}): Promise<'fresh' | 'built'> {
const paths = snapshotProfile(profile, opts.fixtureDir);
const log = opts.log ?? console.log;
const hash = computeSnapshotSchemaHash(crypto, fsModule);
if (!hash) throw new Error('Cannot read snapshot schema dependencies; run from a source checkout.');
const version = `${hash}\ndims=${paths.shape.embedding_dimensions}\nmodel=${paths.shape.embedding_model}\n`;
const fresh = () => {
try { return existsSync(paths.tar) && readFileSync(paths.version, 'utf8') === version; }
catch { return false; }
};
mkdirSync(opts.fixtureDir ?? 'test/fixtures', { recursive: true });
if (fresh()) { log(`[build-pglite-snapshot] ${profile} up to date`); return 'fresh'; }
const owner: LockOwner = { ...snapshotLockIdentity(), pid: process.pid, token: crypto.randomUUID(), protocol: 1 };
const configuredTimeout = opts.lockTimeoutMs ?? Number(process.env.GBRAIN_SNAPSHOT_LOCK_TIMEOUT_MS ?? 120_000);
const timeout = Number.isFinite(configuredTimeout) && configuredTimeout >= 0 ? configuredTimeout : 120_000;
const deadline = Date.now() + timeout;
while (true) {
try { mkdirSync(paths.lock); }
catch (err) {
if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err;
if (fresh()) { log(`[build-pglite-snapshot] ${profile} concurrent builder finished`); return 'fresh'; }
reclaimDeadSnapshotOwner(paths.lock);
if (!existsSync(paths.lock)) continue;
if (Date.now() >= deadline) throw new Error(`Snapshot lock timeout: ${paths.lock}; refusing to build without ownership.`);
await Bun.sleep(Math.min(50, Math.max(1, deadline - Date.now())));
continue;
}
try { writeFileSync(join(paths.lock, 'owner.json'), JSON.stringify(owner), { flag: 'wx' }); }
catch (err) { rmSync(paths.lock, { recursive: true, force: true }); throw err; }
break;
}
const tarTemp = `${paths.tar}.${owner.token}.tmp`;
const versionTemp = `${paths.version}.${owner.token}.tmp`;
try {
if (fresh()) return 'fresh';
log(`[build-pglite-snapshot] building ${profile} (${paths.shape.embedding_model}@${paths.shape.embedding_dimensions})`);
const data = await (opts.buildData ?? bakeData)(paths.shape);
writeFileSync(tarTemp, data, { flag: 'wx' });
writeFileSync(versionTemp, version, { flag: 'wx' });
if (readOwner(paths.lock)?.token !== owner.token) throw new Error('Snapshot lock ownership lost; refusing publication.');
// Same-filesystem atomic renames; the sidecar is the last commit point.
// A crash cannot expose a partially written tar or a fresh-looking new version.
renameSync(tarTemp, paths.tar);
renameSync(versionTemp, paths.version);
log(`[build-pglite-snapshot] wrote ${paths.tar} (${data.byteLength} bytes)`);
return 'built';
} finally {
rmSync(tarTemp, { force: true });
rmSync(versionTemp, { force: true });
if (readOwner(paths.lock)?.token === owner.token) retireSnapshotOwner(paths.lock, owner);
}
}
if (import.meta.main) {
try { await buildPgliteSnapshot(parseSnapshotProfile(process.argv.slice(2))); }
catch (err) {
console.error(`[build-pglite-snapshot] ${(err as Error).message}`);
// PGLite can overwrite process.exitCode asynchronously; make a failed
// build unambiguously nonzero so runners select their cold-init fallback.
process.exit(1);
}
}

121
scripts/capture-test-log.ts Normal file
View File

@@ -0,0 +1,121 @@
#!/usr/bin/env bun
// Preserve live CI output while recording the timestamps used by the weight miner.
import { spawn } from 'node:child_process';
import { createWriteStream, mkdirSync } from 'node:fs';
import { dirname } from 'node:path';
import { once } from 'node:events';
import { constants } from 'node:os';
import type { Readable, Writable } from 'node:stream';
const MAX_PENDING_CHARS = 64 * 1024;
export async function captureTestLog(job: string, out: string, command: string[]): Promise<number> {
if (!job || /[\r\n\t]/.test(job) || !out || command.length === 0) {
throw new Error('job, output path and command are required; job must occupy one TSV field');
}
mkdirSync(dirname(out), { recursive: true });
const log = createWriteStream(out);
await once(log, 'open');
const grouped = process.platform !== 'win32';
const child = spawn(command[0]!, command.slice(1), {
detached: grouped,
stdio: ['inherit', 'pipe', 'pipe'],
});
let receivedSignal: 'SIGTERM' | 'SIGINT' | undefined;
const forward = (signal: 'SIGTERM' | 'SIGINT') => {
receivedSignal ??= signal;
try {
// The group belongs solely to this invocation, including shell pipelines.
if (grouped && child.pid) process.kill(-child.pid, signal);
else child.kill(signal);
} catch { /* child exited before the signal arrived */ }
};
const onTerm = () => forward('SIGTERM');
const onInt = () => forward('SIGINT');
process.on('SIGTERM', onTerm);
process.on('SIGINT', onInt);
let spawnError: Error | undefined;
const exited = new Promise<{ code: number | null; signal: NodeJS.Signals | null }>(resolve => {
child.once('error', error => { spawnError = error; });
child.once('close', (code, signal) => resolve({ code, signal }));
});
let logError: Error | undefined;
const onLogError = (error: Error) => { logError = error; forward('SIGTERM'); };
log.on('error', onLogError);
const write = async (stream: Writable, data: string | Buffer) => {
if (stream === log && logError) throw logError;
if (!stream.write(data)) await once(stream, 'drain');
};
const record = (line: string) => write(log, `${job}\tcapture\t${new Date().toISOString()} ${line}\n`);
const consume = async (source: Readable, mirror: Writable) => {
const decoder = new TextDecoder();
let pending = '';
for await (const chunk of source) {
await write(mirror, chunk);
pending += decoder.decode(chunk, { stream: true });
let newline: number;
while ((newline = pending.indexOf('\n')) >= 0) {
// Large single-line diagnostics are split only in the artifact. Live
// stdout/stderr remain byte-for-byte unchanged and buffering is bounded.
const length = Math.min(newline, MAX_PENDING_CHARS);
await record(pending.slice(0, length).replace(/\r$/, ''));
pending = pending.slice(length + (length === newline ? 1 : 0));
}
while (pending.length > MAX_PENDING_CHARS) {
await record(pending.slice(0, MAX_PENDING_CHARS));
pending = pending.slice(MAX_PENDING_CHARS);
}
}
pending += decoder.decode();
if (pending) await record(pending);
};
const guardedConsume = (source: Readable, mirror: Writable) => consume(source, mirror).catch(error => {
// Stop the owned process group immediately if recording or mirroring fails;
// waiting for its other pipe first could leave a long-running child alive.
forward('SIGTERM');
throw error;
});
try {
await record('##[gbrain-capture-start]');
const streams = await Promise.allSettled([
guardedConsume(child.stdout!, process.stdout), guardedConsume(child.stderr!, process.stderr),
]);
const failed = streams.find(result => result.status === 'rejected');
const result = await exited;
if (spawnError) throw spawnError;
if (failed?.status === 'rejected') throw failed.reason;
if (logError) throw logError;
const code = receivedSignal ? (receivedSignal === 'SIGINT' ? 130 : 143)
: result.signal ? 128 + (constants.signals[result.signal] ?? 1)
: result.code ?? 1;
// A Bun summary can pass before its outer runner detects another failure.
// Keep failed artifacts fail-closed when mined without GitHub run metadata.
if (code !== 0) await record(`##[error]captured command exited ${code}`);
await record(`##[gbrain-capture-complete] exit=${code}`);
await new Promise<void>((resolve, reject) => {
log.once('error', reject);
log.end(resolve);
});
return code;
} finally {
process.off('SIGTERM', onTerm);
process.off('SIGINT', onInt);
log.destroy();
}
}
async function main(): Promise<number> {
const args = process.argv.slice(2);
let job = '', out = '';
let i = 0;
for (; i < args.length && args[i] !== '--'; i++) {
if (args[i] === '--job') job = args[++i] ?? '';
else if (args[i] === '--out') out = args[++i] ?? '';
else throw new Error(`unknown option ${args[i]}`);
}
return captureTestLog(job, out, args[i] === '--' ? args.slice(i + 1) : []);
}
if (import.meta.main) main().then(code => { process.exitCode = code; }).catch(error => {
console.error(`capture-test-log: ${error.message}`);
process.exitCode = 2;
});

View File

@@ -15,6 +15,14 @@
set -euo pipefail
. scripts/lib/test-env.sh
run_brainbench() {
ensure_default_pglite_snapshot brainbench-gate
# Only bare CLI children use this profile; never inherit a caller's legacy
# unit-test snapshot. Build after the baseline/ref preflight has succeeded.
GBRAIN_PGLITE_SNAPSHOT="${GBRAIN_TEST_DEFAULT_SNAPSHOT:-}" bun src/cli.ts eval brainbench "$@"
}
BASELINE_PATH="evals/brainbench/baselines/main.json"
MAIN_REF="${BRAINBENCH_MAIN_REF:-origin/master}"
# mktemp default (review finding): a fixed world-writable /tmp path is a
@@ -50,7 +58,7 @@ if git show "${MAIN_REF}:${BASELINE_PATH}" > "$MAIN_BASELINE" 2>/dev/null; then
exit 2
fi
echo "[brainbench-gate] comparing against ${MAIN_REF}:${BASELINE_PATH}"
bun src/cli.ts eval brainbench --compare "$MAIN_BASELINE" --out "$OUT"
run_brainbench --compare "$MAIN_BASELINE" --out "$OUT"
else
# First landing: the ref exists but carries no baseline yet. The COMMITTED
# baseline still gets verified against the actual run (codex adversarial
@@ -59,9 +67,9 @@ else
# inside --compare does the verification.
if [ -f "$BASELINE_PATH" ]; then
echo "[brainbench-gate] no baseline on ${MAIN_REF} yet — verifying the run against the COMMITTED baseline (first-landing path)"
bun src/cli.ts eval brainbench --compare "$BASELINE_PATH" --out "$OUT"
run_brainbench --compare "$BASELINE_PATH" --out "$OUT"
else
echo "[brainbench-gate] no baseline on ${MAIN_REF} and none committed — running ungated (pre-baseline tree)"
bun src/cli.ts eval brainbench --out "$OUT"
run_brainbench --out "$OUT"
fi
fi

88
scripts/e2e-matrix.ts Normal file
View File

@@ -0,0 +1,88 @@
#!/usr/bin/env bun
// Freeze selection before setup; workers execute exactly the supplied argv.
// selector -> exclusions -> weighted matrix -> isolated sequential workers
import { realpathSync, statSync } from "node:fs";
import { resolve, relative, isAbsolute } from "node:path";
import { loadWeights, partition, type WeightMap } from "./sharding.ts";
export const E2E_EXCLUSIONS = new Set([
'test/e2e/op-checkpoint-jsonb-parity.test.ts',
'test/e2e/jsonb-roundtrip.test.ts',
'test/e2e/mechanical.test.ts',
'test/e2e/mcp.test.ts',
'test/e2e/job-isolation.test.ts',
'test/e2e/sync-reconcile-postgres.test.ts',
'test/e2e/engine-parity.test.ts',
'test/e2e/serve-http-multi-agent.test.ts',
'test/e2e/postgres-bootstrap.test.ts',
'test/e2e/sync-delegation-under-serve.serial.test.ts',
'test/e2e/dream-synthesize-pglite.test.ts',
'test/e2e/skills.test.ts',
'test/e2e/zeroentropy-live.test.ts',
'test/e2e/voyage-rerank-live.test.ts',
'test/e2e/voyage-multimodal.test.ts',
]);
export interface E2ERow { shard: number; files: string[]; empty: boolean }
function validatePath(file: unknown): asserts file is string {
if (typeof file !== "string" || !/^test\/e2e\/[a-zA-Z0-9_./-]+\.test\.ts$/.test(file) || file.split("/").some(p => p === ".." || p === "." || p === "")) {
throw new Error(`invalid E2E test path: ${JSON.stringify(file)}`);
}
}
export function prepareMatrix(files: string[], weights: WeightMap): { include: E2ERow[] } {
for (const file of files) validatePath(file);
if (new Set(files).size !== files.length) throw new Error("duplicate selected E2E file");
const selected = files.filter(file => {
if (!E2E_EXCLUSIONS.has(file)) return true;
console.error(`excluded (named-job / live-key lane): ${file}`);
return false;
});
if (!selected.length) return { include: [{ shard: 1, files: [], empty: true }] };
return { include: partition(selected, weights, Math.min(4, selected.length)).map((files, i) => ({ shard: i + 1, files, empty: false })) };
}
export function validateRow(value: unknown, root: string): E2ERow {
if (!value || typeof value !== "object") throw new Error("missing E2E matrix row");
const row = value as E2ERow;
if (!Number.isInteger(row.shard) || row.shard < 1 || row.shard > 4 || !Array.isArray(row.files) || typeof row.empty !== "boolean") throw new Error("malformed E2E matrix row");
if (row.empty !== (row.files.length === 0) || (row.empty && row.shard !== 1)) throw new Error("invalid empty E2E sentinel");
if (new Set(row.files).size !== row.files.length) throw new Error("duplicate E2E worker file");
const base = realpathSync(root);
for (const file of row.files) {
validatePath(file);
if (E2E_EXCLUSIONS.has(file)) throw new Error(`excluded E2E worker file: ${file}`);
const target = realpathSync(resolve(base, file));
const path = relative(base, target);
if (isAbsolute(path) || !path.startsWith("test/e2e/") || !statSync(target).isFile()) throw new Error(`E2E path escapes corpus: ${file}`);
}
return row;
}
export async function runRow(value: unknown, root = process.cwd()): Promise<number> {
const row = validateRow(value, root);
if (row.empty) {
console.log("selected E2E: explicit empty selection; no tests launched");
return 0;
}
const env = { ...process.env };
delete env.SHARD; // The prepared list is already partitioned.
console.log(`selected E2E shard ${row.shard}: ${row.files.length} frozen files`);
const child = Bun.spawn(["bash", "scripts/run-e2e.sh", ...row.files], { cwd: root, env, stdout: "inherit", stderr: "inherit" });
const term = () => child.kill("SIGTERM");
const interrupt = () => child.kill("SIGINT");
process.once("SIGTERM", term);
process.once("SIGINT", interrupt);
try { return await child.exited; }
finally { process.off("SIGTERM", term); process.off("SIGINT", interrupt); }
}
async function main() {
if (process.argv[2] === "prepare") {
if (process.stdin.isTTY) throw new Error("prepare requires selector output on stdin");
const raw = await new Response(Bun.stdin.stream()).text();
const files = raw.split("\n").filter(Boolean);
const matrix = prepareMatrix(files, loadWeights(resolve(import.meta.dir, "e2e-weights.json")));
console.error(`selected E2E: ${files.length} selected, ${matrix.include.reduce((n, row) => n + row.files.length, 0)} executable, ${matrix.include.length} workers`);
console.log(JSON.stringify(matrix));
} else if (process.argv[2] === "run") {
if (!process.env.E2E_MATRIX_ROW) throw new Error("E2E_MATRIX_ROW is required");
process.exitCode = await runRow(JSON.parse(process.env.E2E_MATRIX_ROW));
} else throw new Error("usage: bun scripts/e2e-matrix.ts prepare|run");
}
if (import.meta.main) main().catch(error => { console.error(`e2e-matrix: ${error.message}`); process.exitCode = 1; });

223
scripts/e2e-weights.json Normal file
View File

@@ -0,0 +1,223 @@
{
"test/e2e/anomalies-pglite.test.ts": 1510,
"test/e2e/auth-permissions.test.ts": 1407,
"test/e2e/auth-takes-holders-pglite.test.ts": 947,
"test/e2e/autopilot-cooldown-parity.test.ts": 1900,
"test/e2e/autopilot-fanout-postgres.test.ts": 1357,
"test/e2e/autopilot-linux-lifecycle.serial.test.ts": 11590,
"test/e2e/backfill-perf-pglite.test.ts": 4300,
"test/e2e/bootstrap-attach.serial.test.ts": 4350,
"test/e2e/bootstrap-compiled-binary.serial.test.ts": 3900,
"test/e2e/bootstrap-degraded-modes.serial.test.ts": 4490,
"test/e2e/bootstrap-harness-lifecycle.serial.test.ts": 7480,
"test/e2e/bootstrap-hook-under-serve.serial.test.ts": 6160,
"test/e2e/bootstrap-keyed-postgres.serial.test.ts": 897,
"test/e2e/bootstrap-lifecycle.serial.test.ts": 5450,
"test/e2e/bootstrap-magic-moment.serial.test.ts": 4270,
"test/e2e/bootstrap-persistence.serial.test.ts": 781,
"test/e2e/bootstrap-real-claude.serial.test.ts": 381,
"test/e2e/bootstrap-real-codex.serial.test.ts": 388,
"test/e2e/brainstorm-resume.test.ts": 1830,
"test/e2e/cache-gate-pglite.test.ts": 2380,
"test/e2e/calibration-profile-write.test.ts": 1900,
"test/e2e/capture-generation-regression.test.ts": 1920,
"test/e2e/chronicle-event-projection-parity.test.ts": 1890,
"test/e2e/chronicle-last-seen-postgres.test.ts": 1016,
"test/e2e/chunk-canonical-text-privacy.test.ts": 1880,
"test/e2e/chunker-takes-strip.test.ts": 1166,
"test/e2e/cjk-roundtrip.test.ts": 1499,
"test/e2e/claude-plugin-install-real.serial.test.ts": 364,
"test/e2e/claw-test.test.ts": 67970,
"test/e2e/cli-source-scoping-pglite.test.ts": 1520,
"test/e2e/client-grants.test.ts": 1630,
"test/e2e/code-edges-jsonb-postgres.test.ts": 998,
"test/e2e/code-edges-read-parity.test.ts": 1970,
"test/e2e/code-indexing.test.ts": 2530,
"test/e2e/code-intel-mcp-ops-pglite.test.ts": 3720,
"test/e2e/codex-plugin-install-real.serial.test.ts": 363,
"test/e2e/compile-context-pglite.test.ts": 1400,
"test/e2e/concurrent-embed-race.test.ts": 1131,
"test/e2e/connect-bearer.test.ts": 9870,
"test/e2e/connector-sync-handler-pglite.test.ts": 1610,
"test/e2e/connectors-sync-pglite.test.ts": 5840,
"test/e2e/contextual-retrieval-pglite.test.ts": 2270,
"test/e2e/conversation-parser-pglite.test.ts": 4690,
"test/e2e/cross-modal-eval.test.ts": 157,
"test/e2e/cycle-consolidate-postgres.test.ts": 1040,
"test/e2e/cycle-recompute-emotional-weight-pglite.test.ts": 1187,
"test/e2e/cycle.test.ts": 1730,
"test/e2e/db-guard.test.ts": 279,
"test/e2e/db-singleton-shared-recovery.test.ts": 213,
"test/e2e/delegated-grants-withdrawal.test.ts": 1385,
"test/e2e/delegated-http-worker.test.ts": 2230,
"test/e2e/doctor-connectors-pglite.test.ts": 1730,
"test/e2e/doctor-progress.test.ts": 7050,
"test/e2e/doctor-silent-death-parity.test.ts": 2700,
"test/e2e/dream-allow-list-pglite.test.ts": 1433,
"test/e2e/dream-cycle-phase-order-pglite.test.ts": 2740,
"test/e2e/dream-patterns-pglite.test.ts": 3410,
"test/e2e/dream-synthesize-chunking.test.ts": 8300,
"test/e2e/dream-synthesize-concurrency-postgres.test.ts": 9550,
"test/e2e/dream-triage-postgres.test.ts": 1088,
"test/e2e/dream.test.ts": 1610,
"test/e2e/embed-stale-pagination.test.ts": 1490,
"test/e2e/embedding-column-pglite.test.ts": 1202,
"test/e2e/embedding-column-postgres.test.ts": 378,
"test/e2e/engine-content-privacy.test.ts": 4180,
"test/e2e/engine-parity-cjk.test.ts": 2260,
"test/e2e/engine-parity-salience.test.ts": 4490,
"test/e2e/enrich-pglite.test.ts": 5340,
"test/e2e/eval-capture.test.ts": 387,
"test/e2e/eval-contradictions-postgres.test.ts": 533,
"test/e2e/eval-loop.test.ts": 1165,
"test/e2e/eval-replay-column.test.ts": 989,
"test/e2e/eval-takes-quality.test.ts": 983,
"test/e2e/extract-atoms-discovery-sql.test.ts": 1016,
"test/e2e/extraction-review-postgres.test.ts": 1173,
"test/e2e/facts-context-injection-postgres.test.ts": 969,
"test/e2e/facts-cross-source-isolation.test.ts": 975,
"test/e2e/facts-fence-reconcile-postgres.test.ts": 980,
"test/e2e/facts-forget.test.ts": 1010,
"test/e2e/facts-notability-roundtrip.test.ts": 1015,
"test/e2e/facts-recall-event-time-postgres.test.ts": 1004,
"test/e2e/facts-recall-render.test.ts": 1041,
"test/e2e/facts-separation-postgres.test.ts": 1018,
"test/e2e/fresh-install-pglite.test.ts": 19980,
"test/e2e/frontmatter-migration.test.ts": 3260,
"test/e2e/global-basename-pglite.test.ts": 1700,
"test/e2e/graph-quality.test.ts": 1810,
"test/e2e/graph-signals-engine.test.ts": 1004,
"test/e2e/graph-signals-eval.test.ts": 3220,
"test/e2e/harness-access.test.ts": 4770,
"test/e2e/health-parity-postgres.test.ts": 2040,
"test/e2e/http-transport.test.ts": 1317,
"test/e2e/import-credential-preflight.test.ts": 1530,
"test/e2e/ingestion-roundtrip.test.ts": 2850,
"test/e2e/init-fresh-pglite.test.ts": 39380,
"test/e2e/install-real-grok.serial.test.ts": 300,
"test/e2e/install-real-hermes.serial.test.ts": 278,
"test/e2e/install-real-opencode.serial.test.ts": 282,
"test/e2e/integrity-batch.test.ts": 1236,
"test/e2e/jobs-agent-scope-postgres.test.ts": 998,
"test/e2e/jobs-watch-readsnapshot.test.ts": 1120,
"test/e2e/jsonb-batch-poison-postgres.test.ts": 1620,
"test/e2e/legacy-chunk-privacy.test.ts": 4480,
"test/e2e/link-source-check-repair-postgres.test.ts": 958,
"test/e2e/list-all-sources-postgres.test.ts": 985,
"test/e2e/list-pages-regression.test.ts": 1013,
"test/e2e/mcp-budget-reservation-postgres.test.ts": 478,
"test/e2e/migrate-chain.test.ts": 2970,
"test/e2e/migrate-embeddings-postgres.test.ts": 1241,
"test/e2e/migrate-engine-pglite-to-postgres.test.ts": 5970,
"test/e2e/migrate-engine-sources-postgres.test.ts": 1800,
"test/e2e/migration-drop-invalid-concurrent-index.test.ts": 962,
"test/e2e/migration-flow.test.ts": 3940,
"test/e2e/migration-v35-auto-rls.test.ts": 971,
"test/e2e/migration-v35-event-trigger.test.ts": 949,
"test/e2e/migration-v47-notability.test.ts": 947,
"test/e2e/migration-v50-ingest-log-source-id.test.ts": 952,
"test/e2e/minions-authority-parity.test.ts": 1082,
"test/e2e/minions-budget-cathedral.test.ts": 1308,
"test/e2e/minions-concurrency.test.ts": 1444,
"test/e2e/minions-controller-bounce-only.test.ts": 1233,
"test/e2e/minions-field-report-repro.test.ts": 26100,
"test/e2e/minions-prefix-strip-smoke.test.ts": 1081,
"test/e2e/minions-resilience.test.ts": 12360,
"test/e2e/minions-self-fix-flow.test.ts": 1163,
"test/e2e/minions-shell-pglite.test.ts": 1780,
"test/e2e/minions-shell.test.ts": 1343,
"test/e2e/mounts-routing-pglite.test.ts": 27570,
"test/e2e/multi-source-bug-class.test.ts": 4130,
"test/e2e/multi-source-emotional-weight-pglite.test.ts": 979,
"test/e2e/multi-source.test.ts": 4950,
"test/e2e/multimodal-postgres.test.ts": 1282,
"test/e2e/non-tty-output.serial.test.ts": 5710,
"test/e2e/oauth-grant-transactions.test.ts": 1090,
"test/e2e/onboard-full-flow.test.ts": 1168,
"test/e2e/ontology-merge-parity.test.ts": 1810,
"test/e2e/openclaw-context-engine-plugin.test.ts": 215,
"test/e2e/openclaw-plugin-load-real.test.ts": 154,
"test/e2e/openclaw-reference-compat.test.ts": 979,
"test/e2e/openrouter-anthropic-subagent-replay.live.test.ts": 1255,
"test/e2e/openrouter-deepseek-subagent-replay.live.test.ts": 1255,
"test/e2e/orphan-reduction.test.ts": 3170,
"test/e2e/pgbouncer-teardown.test.ts": 185,
"test/e2e/pglite-cli-exit.serial.test.ts": 32299.999999999996,
"test/e2e/phantom-redirect.test.ts": 31570,
"test/e2e/postgres-engine-disconnect-idempotency.test.ts": 245,
"test/e2e/postgres-jsonb.test.ts": 945,
"test/e2e/postgres-reconnect-singleton.test.ts": 265,
"test/e2e/propose-takes-jsonb-postgres.test.ts": 962,
"test/e2e/purge-deleted-dryrun-postgres.test.ts": 993,
"test/e2e/qm-provisioning.test.ts": 24710,
"test/e2e/quarantine-search-exclusion.test.ts": 2350,
"test/e2e/read-enrichment-privacy.test.ts": 3630,
"test/e2e/remote-privacy-journeys.test.ts": 17670,
"test/e2e/restore-source-config-jsonb-postgres.test.ts": 954,
"test/e2e/salience-anomalies-source-isolation-pglite.test.ts": 1257,
"test/e2e/salience-llm-routing.test.ts": 156,
"test/e2e/salience-pglite.test.ts": 1445,
"test/e2e/schema-cathedral.test.ts": 4390,
"test/e2e/schema-drift.test.ts": 4190,
"test/e2e/search-exclude.test.ts": 1290,
"test/e2e/search-quality.test.ts": 1140,
"test/e2e/search-swamp.test.ts": 1110,
"test/e2e/self-upgrade-binary-swap.test.ts": 177,
"test/e2e/self-upgrade-marker.test.ts": 3800,
"test/e2e/serve-http-consent.test.ts": 3830,
"test/e2e/serve-http-ingest-webhook.test.ts": 25750,
"test/e2e/serve-http-meta.test.ts": 928,
"test/e2e/serve-http-oauth.test.ts": 16490,
"test/e2e/serve-http-source-grant.test.ts": 3170,
"test/e2e/serve-http-surface-ceiling.test.ts": 14820,
"test/e2e/serve-http-takes-holders.test.ts": 3190,
"test/e2e/serve-stdio-roundtrip.test.ts": 16309.999999999998,
"test/e2e/skill-brain-first.test.ts": 488,
"test/e2e/skillopt-loop.serial.test.ts": 4090,
"test/e2e/skillopt-pglite.serial.test.ts": 1650,
"test/e2e/skillpack-flow.test.ts": 10440,
"test/e2e/skillpack-third-party.test.ts": 5490,
"test/e2e/source-boundary-mutation-postgres.test.ts": 1049,
"test/e2e/source-isolation-pglite.test.ts": 4040,
"test/e2e/source-routing.test.ts": 1240,
"test/e2e/sources-remote-mcp.test.ts": 7160,
"test/e2e/status-pglite.test.ts": 2160,
"test/e2e/storage-tiering.test.ts": 1530,
"test/e2e/subagent-crash-replay-multi-provider.test.ts": 3080,
"test/e2e/subagent-gateway-path.test.ts": 3220,
"test/e2e/subagent-gateway-resume-reconciliation.test.ts": 2250,
"test/e2e/symbol-resolver-pglite.test.ts": 1830,
"test/e2e/sync-cjk-git.test.ts": 311,
"test/e2e/sync-credential-preflight.test.ts": 2230,
"test/e2e/sync-lock-recovery.test.ts": 4750,
"test/e2e/sync-parallel.test.ts": 4750,
"test/e2e/sync-sigkill-resume-postgres.test.ts": 11380,
"test/e2e/sync-status-pglite.test.ts": 1269,
"test/e2e/sync.test.ts": 3100,
"test/e2e/synthesize-bigint-job-id-postgres.test.ts": 963,
"test/e2e/system-of-record-invariant.test.ts": 2150,
"test/e2e/takes-postgres.test.ts": 1075,
"test/e2e/takes-scorecard-parity.test.ts": 1900,
"test/e2e/takes-weight-rounding-postgres.test.ts": 990,
"test/e2e/takes-write-ops-postgres.test.ts": 3040,
"test/e2e/thin-client.test.ts": 13280,
"test/e2e/think-source-isolation-pglite.test.ts": 1550,
"test/e2e/think-trajectory-pglite.test.ts": 1800,
"test/e2e/transcripts-ingest-pglite.test.ts": 7610,
"test/e2e/transcripts-writeback-fidelity.test.ts": 2450,
"test/e2e/type-unification-full-flow.test.ts": 2530,
"test/e2e/upgrade-bun-link-arc.serial.test.ts": 5330,
"test/e2e/upgrade.test.ts": 3120,
"test/e2e/upsert-chunks-registry-column.test.ts": 1740,
"test/e2e/v030_1-integration-pglite.test.ts": 28770,
"test/e2e/v0_28_5-fix-wave.test.ts": 11550,
"test/e2e/v0_29-mcp-dispatch-pglite.test.ts": 1247,
"test/e2e/v0_30_3-fix-wave.test.ts": 10300,
"test/e2e/vector-ef-search-postgres.test.ts": 2390,
"test/e2e/volunteer-context-postgres.test.ts": 952,
"test/e2e/whoknows.test.ts": 1690,
"test/e2e/worker-abort-recovery.test.ts": 4770,
"test/e2e/worker-lock-renewal-starvation.test.ts": 3740,
"test/e2e/workspace-generic-compat.test.ts": 601,
"test/e2e/zombie-reaping.test.ts": 9510
}

View File

@@ -0,0 +1,10 @@
{
"lane": "e2e",
"unit": "milliseconds",
"run": "34998157990",
"commit": "f1fbdfba193383785b555b507f5c817003944aea",
"source": "github",
"measuredFiles": 221,
"totalFiles": 221,
"mergeExisting": true
}

View File

@@ -65,7 +65,10 @@ detect_available_mem_mb() {
# ──────────────────────────────────────────────────────────────────────────
ensure_pglite_snapshot() {
local label="${1:-test-env}"
[ "${GBRAIN_NO_SNAPSHOT:-0}" = "1" ] && return 0
if [ "${GBRAIN_NO_SNAPSHOT:-0}" = "1" ]; then
unset GBRAIN_PGLITE_SNAPSHOT GBRAIN_TEST_DEFAULT_SNAPSHOT
return 0
fi
if [ -n "${GBRAIN_PGLITE_SNAPSHOT:-}" ]; then
echo "[$label] PGLite snapshot active (inherited): $GBRAIN_PGLITE_SNAPSHOT" >&2
return 0
@@ -77,3 +80,27 @@ ensure_pglite_snapshot() {
echo "[$label] snapshot build failed (non-fatal) — tests run with cold init" >&2
fi
}
# Bare BrainBench CLI children use the shipped embedding shape. Keep this
# auxiliary path separate from their parent bun test process's legacy shape.
ensure_default_pglite_snapshot() {
local label="${1:-test-env}"
if [ "${GBRAIN_NO_SNAPSHOT:-0}" = "1" ]; then
unset GBRAIN_PGLITE_SNAPSHOT GBRAIN_TEST_DEFAULT_SNAPSHOT
return 0
fi
if [ -z "${GBRAIN_TEST_DEFAULT_SNAPSHOT:-}" ]; then
if bun run build:pglite-snapshot --profile default >/dev/null 2>&1; then
export GBRAIN_TEST_DEFAULT_SNAPSHOT="$PWD/test/fixtures/pglite-snapshot-default.tar"
else
unset GBRAIN_TEST_DEFAULT_SNAPSHOT
echo "[$label] default snapshot build failed (non-fatal) — CLI children run with cold init" >&2
return 0
fi
fi
case "$GBRAIN_TEST_DEFAULT_SNAPSHOT" in
/*) ;;
*) export GBRAIN_TEST_DEFAULT_SNAPSHOT="$PWD/$GBRAIN_TEST_DEFAULT_SNAPSHOT" ;;
esac
echo "[$label] default PGLite snapshot active: $GBRAIN_TEST_DEFAULT_SNAPSHOT" >&2
}

View File

@@ -6,7 +6,7 @@
*
* Usage:
* bun scripts/merge-lcov.ts --out-lcov <path> --out-json <path> \
* [--manifest-expect <lane,lane,...>] <dir-or-file>...
* [--manifest-expect <lane,lane,...>] [--sha <commit>] <dir-or-file>...
*
* Behavior:
* - Recursively finds every lcov.info under the input dirs (a file input
@@ -30,6 +30,8 @@
* - Lane manifests: each lane dir may contain lane-manifest.json
* {lane, sha, lcovCount, complete}. With --manifest-expect, a
* missing/incomplete manifest for an expected lane marks degraded.
* Duplicate lane identities, a SHA other than --sha (default HEAD),
* or an lcovCount unlike the files under that manifest also degrade.
* A manifest whose lane name contains 'shard' with lcovCount != 1
* marks degraded: the shard lane runs ONE bun process via xargs -x;
* a second bun process reusing the coverage dir would have OVERWRITTEN
@@ -63,7 +65,8 @@ import {
statSync,
writeFileSync,
} from "node:fs";
import { dirname, join } from "node:path";
import { dirname, join, resolve, sep } from "node:path";
import { spawnSync } from "node:child_process";
// ---------------------------------------------------------------------------
// Parsing
@@ -322,7 +325,7 @@ function warn(msg: string): void {
function usage(msg: string): never {
process.stderr.write(`merge-lcov: ${msg}\n`);
process.stderr.write(
"usage: bun scripts/merge-lcov.ts --out-lcov <path> --out-json <path> [--manifest-expect <lane,lane,...>] <dir-or-file>...\n",
"usage: bun scripts/merge-lcov.ts --out-lcov <path> --out-json <path> [--manifest-expect <lane,lane,...>] [--sha <expected-commit>] <dir-or-file>...\n",
);
process.exit(2);
}
@@ -362,11 +365,16 @@ function main(): void {
let outLcov = "";
let outJson = "";
let manifestExpect: string[] = [];
let expectedSha = "";
const inputs: string[] = [];
for (let i = 0; i < argv.length; i++) {
const a = argv[i]!;
if (a === "--out-lcov") outLcov = argv[++i] ?? "";
else if (a === "--out-json") outJson = argv[++i] ?? "";
else if (a === "--sha") {
expectedSha = argv[++i] ?? "";
if (!expectedSha || expectedSha.startsWith("--")) usage("--sha requires a commit");
}
else if (a === "--manifest-expect") {
manifestExpect = (argv[++i] ?? "")
.split(",")
@@ -400,7 +408,15 @@ function main(): void {
}
// --- lane manifests ---
const manifests: LaneManifest[] = [];
if (!expectedSha && manifestFiles.length > 0) {
const git = spawnSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" });
if (git.status === 0) expectedSha = git.stdout.trim();
if (!expectedSha) {
warn("cannot resolve expected commit — pass --sha; marking degraded");
degraded = true;
}
}
const manifests: Array<LaneManifest & { root: string; valid: boolean }> = [];
for (const mf of manifestFiles) {
try {
const parsed = JSON.parse(readFileSync(mf, "utf8")) as Partial<LaneManifest>;
@@ -414,6 +430,8 @@ function main(): void {
sha: typeof parsed.sha === "string" ? parsed.sha : undefined,
lcovCount: typeof parsed.lcovCount === "number" ? parsed.lcovCount : undefined,
complete: parsed.complete === true,
root: resolve(dirname(mf)),
valid: true,
});
} catch {
warn(`manifest ${mf} is not valid JSON — marking degraded`);
@@ -421,11 +439,27 @@ function main(): void {
}
}
for (const m of manifests) {
const invalidate = (reason: string) => {
warn(`lane '${m.lane}' ${reason} — marking degraded`);
m.valid = false;
degraded = true;
};
if (!expectedSha || m.sha !== expectedSha) {
invalidate(`has sha=${m.sha ?? "missing"}, expected ${expectedSha || "unavailable"}`);
}
if (manifests.filter(other => other.lane === m.lane).length !== 1) {
invalidate("has duplicate manifests");
}
const actualCount = new Set(lcovFiles.filter(file => resolve(file).startsWith(m.root + sep)).map(file => resolve(file))).size;
if (!Number.isInteger(m.lcovCount) || m.lcovCount! < 0 || m.lcovCount !== actualCount) {
invalidate(`has lcovCount=${m.lcovCount ?? "missing"}, actual=${actualCount}`);
}
// xargs-batching tripwire: a shard lane must have written EXACTLY ONE
// lcov.info (one bun process). Anything else means data was overwritten
// or never written.
if (m.lane.includes("shard") && m.lcovCount !== 1) {
warn(`shard lane '${m.lane}' has lcovCount=${m.lcovCount ?? "missing"} (expected 1) — marking degraded`);
m.valid = false;
degraded = true;
}
}
@@ -507,7 +541,7 @@ function main(): void {
lanes: {
expected: manifestExpect,
complete: manifests
.filter((m) => m.complete === true)
.filter((m) => m.complete === true && m.valid)
.map((m) => m.lane)
.sort(),
},

View File

@@ -1,247 +1,210 @@
#!/usr/bin/env bun
/**
* scripts/mine-shard-weights.ts — extract per-file test wallclock from a
* real CI run's logs, write scripts/test-weights.json.
*
* Why this exists: scripts/sharding.ts does LPT bin-packing over per-file
* weights, but the weights have to come from somewhere. The original
* design ran each test file in isolation (`bun test <file>` per file)
* which (a) takes ~57min to run all 676 files, and (b) measures cold-
* start dominantly because each invocation pays a fresh `bun test`
* startup. CI shards run ~150 files in ONE bun process — cold-start is
* amortized away. Per-file isolated profiles are wrong-by-methodology.
*
* This script scrapes per-file wallclock from a real CI shard's log via
* GitHub's `gh run view --log` output. bun emits an `##[group]test/foo.
* test.ts:` header before each file with an ISO timestamp; the
* difference between consecutive headers = how long the previous file
* took. This is the actual CI shard runtime per file, in the right
* execution mode, for free on every green run.
*
* Usage:
* bun run scripts/mine-shard-weights.ts --run <RUN_ID> [--out PATH]
* bun run scripts/mine-shard-weights.ts --from-file <LOG_FILE> [--out PATH]
* gh run view <RUN_ID> --log | bun run scripts/mine-shard-weights.ts [--out PATH]
*
* Default output: scripts/test-weights.json (overwrites). Use --out to
* write elsewhere (useful for diffing before commit). Output is JSON
* with sorted keys for stable diffs.
*
* Regen cadence: there is none. Weights drift continuously but missing
* files fall back to the corpus median in sharding.ts, so stale weights
* degrade gracefully. Run this script when you notice a specific shard
* starts running long, or after a wave that added many heavy tests.
*
* Exit codes:
* 0 wrote weights file (count > 0)
* 1 internal error
* 2 usage error
* 3 no usable timing data found (parsed log but extracted 0 weights)
/** Mine timings from successful CI execution in its real execution mode.
* Unit: consecutive headers + final Bun summary, milliseconds.
* Serial: per-file runner PASS records, seconds (not buffered log timestamps).
* E2E: per-file Bun summaries, milliseconds. Partial selections merge old weights.
* Refresh after large test additions or sustained shard imbalance.
*/
import { spawnSync } from "node:child_process";
import { readFileSync, writeFileSync } from "node:fs";
import { readFileSync, writeFileSync, renameSync, rmSync, existsSync } from "node:fs";
import { dirname, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { loadWeights } from "./sharding.ts";
const __dirname = dirname(fileURLToPath(import.meta.url));
const REPO_ROOT = resolve(__dirname, "..");
const DEFAULT_OUT = resolve(REPO_ROOT, "scripts/test-weights.json");
interface Args {
runId?: string;
fromFile?: string;
fromStdin: boolean;
out: string;
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), "..");
export type Lane = "unit" | "serial" | "e2e";
const OUTPUTS = { unit: "test-weights", serial: "serial-weights", e2e: "e2e-weights" };
interface TimingEvent { job: string; timestampMs: number; file: string; summaryCount?: number; failures?: number }
interface LogLine { job: string; step: string; timestampMs: number; text: string }
function logLines(raw: string): LogLine[] {
return raw.split("\n").flatMap(line => {
const m = /^([^\t]+)\t([^\t]*)\t(\d{4}-\d{2}-\d{2}T\S+Z)\s+(.*)$/.exec(line);
if (!m || !Number.isFinite(Date.parse(m[3]!))) return [];
return [{ job: m[1]!.trim(), step: m[2]!.trim(), timestampMs: Date.parse(m[3]!), text: m[4]!.replace(/\x1b\[[0-9;]*m/g, "") }];
});
}
function isJob(job: string, lane: Lane): boolean {
if (lane === "unit") return /^test \(\d+\)$/.test(job);
if (lane === "serial") return /^serial-tests(?: \(\d+\))?$/.test(job);
return /^Selected E2E \(diff-relevant\)(?: \(\d+\)| \d+)?$/.test(job);
}
function parseArgs(argv: string[]): Args {
const out: Args = { fromStdin: false, out: DEFAULT_OUT };
for (let i = 0; i < argv.length; i++) {
const a = argv[i];
if (a === "--run" || a === "--run-id") {
out.runId = argv[++i];
} else if (a === "--from-file") {
out.fromFile = argv[++i];
} else if (a === "--out") {
out.out = resolve(argv[++i] ?? "");
} else if (a === "--help" || a === "-h") {
console.log(
"usage: bun run scripts/mine-shard-weights.ts (--run <ID> | --from-file <PATH> | <stdin>) [--out <PATH>]",
);
process.exit(0);
} else {
console.error(`error: unknown arg: ${a}`);
process.exit(2);
}
}
if (!out.runId && !out.fromFile) {
out.fromStdin = true;
}
return out;
}
async function readSource(args: Args): Promise<string> {
if (args.runId) {
const r = spawnSync("gh", ["run", "view", args.runId, "--log"], {
encoding: "utf8",
maxBuffer: 256 * 1024 * 1024, // CI logs can be 50-80MB
});
if (r.status !== 0) {
throw new Error(
`gh run view ${args.runId} --log failed (exit ${r.status}): ${r.stderr}`,
);
}
return r.stdout;
}
if (args.fromFile) {
return readFileSync(args.fromFile, "utf8");
}
// Read stdin
const chunks: Buffer[] = [];
for await (const chunk of process.stdin) {
chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk);
}
return Buffer.concat(chunks).toString("utf8");
}
/**
* Parsed timing event from a CI log line. timestamp is ms-since-epoch.
*/
interface TimingEvent {
job: string;
timestampMs: number;
file: string;
}
/**
* Parse a CI log into a list of `##[group]test/X.test.ts:` events keyed
* by job (so timing deltas don't cross shard boundaries).
*
* GH log line shape:
* <job-name>\tUNKNOWN STEP\t<ISO-timestamp> ##[group]test/foo.test.ts:
* or:
* <job-name>\t<step-name>\t<ISO-timestamp> ##[group]test/foo.test.ts:
*
* Exported for unit testing.
*/
/** File starts and the LAST summary per unit job; nested CLI summaries are ignored. */
export function parseLog(raw: string): TimingEvent[] {
const events: TimingEvent[] = [];
const lines = raw.split("\n");
// Match: <job>TAB<step>TAB<iso-ts> ##[group]<path>:
const re = /^([^\t]+)\t[^\t]*\t(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d+Z)\s+##\[group\](test\/[^\s:]+\.test\.ts):?\s*$/;
for (const line of lines) {
const m = re.exec(line);
if (!m) continue;
const job = m[1]!.trim();
const ts = Date.parse(m[2]!);
const file = m[3]!;
if (Number.isNaN(ts)) continue;
events.push({ job, timestampMs: ts, file });
const ends = new Map<string, TimingEvent>();
const failures = new Map<string, number>();
for (const line of logLines(raw)) {
if (!isJob(line.job, "unit")) continue;
// Captured artifacts retain Bun's raw workflow command; GitHub's rendered
// logs rewrite that same command to the bracketed form.
const m = /^(?:##\[group\]|::group::)((?:test|evals)\/[^\s:]+\.test\.ts):?\s*$/.exec(line.text);
if (m) events.push({ job: line.job, timestampMs: line.timestampMs, file: m[1]! });
const fail = /^\s*(\d+) fail\s*$/.exec(line.text);
if (fail) failures.set(line.job, Number(fail[1]));
const summary = /^Ran \d+ tests? across (\d+) files?\./.exec(line.text);
if (summary) ends.set(line.job, { job: line.job, timestampMs: line.timestampMs, file: "", summaryCount: Number(summary[1]), failures: failures.get(line.job) });
}
return events;
return [...events, ...ends.values()];
}
/**
* From a list of file-start events grouped by job, compute per-file
* runtime as (timestamp[i+1] - timestamp[i]) within each job. The last
* file in each job is dropped (we don't know when it ended without
* also parsing the bun summary line; the loss is acceptable since
* sharding.ts's median fallback covers missing files).
*
* When the same file appears in multiple jobs (shouldn't happen, but
* defensive against shard remix during the in-flight CI run that
* generated this log), take the max — heaviest observation wins.
*
* Exported for unit testing.
*/
export function computeWeights(events: TimingEvent[]): Map<string, number> {
// Group events by job, in stream order.
const byJob = new Map<string, TimingEvent[]>();
for (const e of events) {
let bucket = byJob.get(e.job);
if (!bucket) {
bucket = [];
byJob.set(e.job, bucket);
}
bucket.push(e);
if (!byJob.has(e.job)) byJob.set(e.job, []);
byJob.get(e.job)!.push(e);
}
const weights = new Map<string, number>();
for (const [, jobEvents] of byJob) {
for (let i = 0; i + 1 < jobEvents.length; i++) {
const file = jobEvents[i]!.file;
const delta = jobEvents[i + 1]!.timestampMs - jobEvents[i]!.timestampMs;
if (delta < 0) continue; // log out-of-order; defensive
// Round to nearest ms; sub-ms doesn't matter for shard balancing.
const ms = Math.round(delta);
const prev = weights.get(file);
if (prev === undefined || ms > prev) {
weights.set(file, ms);
}
for (const list of byJob.values()) {
for (let i = 0; i + 1 < list.length; i++) {
const a = list[i]!, b = list[i + 1]!;
if (!a.file || b.timestampMs < a.timestampMs) continue;
weights.set(a.file, Math.max(weights.get(a.file) ?? 0, Math.round(b.timestampMs - a.timestampMs)));
}
// Drop the last event's file (no successor → unknown duration).
}
return weights;
}
/**
* Serialize a weights map to canonical JSON (keys sorted asc) so the
* committed file produces stable diffs run-to-run.
*
* Exported for unit testing.
*/
/** Refuse partial/failed sources before any output is replaced. */
export function mineWeights(raw: string, lane: Lane, opts: { expectedJobs?: readonly string[] } = {}): Map<string, number> {
const lines = logLines(raw).filter(line => isJob(line.job, lane));
if (lines.some(l => /^##\[error\]/.test(l.text))) throw new Error(`${lane}: failed job log`);
const byJob = new Map<string, LogLine[]>();
for (const line of lines) {
if (!byJob.has(line.job)) byJob.set(line.job, []);
byJob.get(line.job)!.push(line);
}
if (opts.expectedJobs) {
const expected = new Set(opts.expectedJobs);
const missing = [...expected].filter(job => !byJob.has(job));
const unexpected = [...byJob.keys()].filter(job => !expected.has(job));
if (missing.length || unexpected.length) throw new Error(`${lane}: timing job set differs from run metadata (missing: ${missing.join(', ') || 'none'}; unexpected: ${unexpected.join(', ') || 'none'})`);
}
for (const [job, records] of byJob) {
const captured = records.filter(line => line.step === 'capture');
if (!captured.length) continue;
if (captured[0]!.text !== '##[gbrain-capture-start]' ||
captured.at(-1)!.text !== '##[gbrain-capture-complete] exit=0' ||
captured.filter(line => line.text === '##[gbrain-capture-start]').length !== 1 ||
captured.filter(line => line.text.startsWith('##[gbrain-capture-complete]')).length !== 1) {
throw new Error(`${job}: missing or unsuccessful capture completion`);
}
}
if (lane === "unit") {
const events = parseLog(raw);
for (const job of byJob.keys()) {
const files = events.filter(e => e.job === job && e.file);
const end = events.find(e => e.job === job && e.summaryCount !== undefined);
if (!end || end.failures !== 0 || end.summaryCount !== files.length || !files.length) throw new Error(`${job}: missing, failed, or incomplete Bun summary`);
const ordered = [...files, end];
if (ordered.some((e, i) => i > 0 && e.timestampMs < ordered[i - 1]!.timestampMs)) throw new Error(`${job}: out-of-order timing records`);
if (new Set(files.map(e => e.file)).size !== files.length) throw new Error(`${job}: duplicate file headers`);
}
const weights = computeWeights(events);
if (!weights.size) throw new Error("unit: no complete timing data");
return weights;
}
const duration = (text: string, multiplier: number, job: string) => {
const value = Number(text) * multiplier;
if (!Number.isFinite(value) || value < 0) throw new Error(`${job}: invalid duration ${text}`);
return value;
};
const weights = new Map<string, number>();
for (const [job, records] of byJob) {
const found = new Map<string, number>();
let expected: number | undefined, current: string | undefined;
let failures: number | undefined;
for (const { text } of records) {
if (lane === "serial") {
const pass = /^\[serial-tests\] PASS ([\d.]+)s (test\/\S+\.serial\.test\.ts)(?:\s|$)/.exec(text);
if (pass) found.set(pass[2]!, Math.max(found.get(pass[2]!) ?? 0, duration(pass[1]!, 1, job)));
const end = /^\[serial-tests\] all (\d+) file\(s\) passed/.exec(text);
if (end) expected = Number(end[1]);
} else {
const start = /^=== ([^/]+\.test\.ts) ===$/.exec(text);
if (start) {
if (current) throw new Error(`${job}: file missing Bun summary: ${current}`);
current = `test/e2e/${start[1]}`;
failures = undefined;
}
const fail = /^\s*(\d+) fail\s*$/.exec(text);
if (fail) failures = Number(fail[1]);
const summary = /^Ran \d+ tests? across 1 file\. \[([\d.]+)(ms|s)\]/.exec(text);
if (current && summary) {
if (failures !== 0 || found.has(current)) throw new Error(`${job}: failed or duplicate file ${current}`);
found.set(current, duration(summary[1]!, summary[2] === "s" ? 1000 : 1, job));
current = undefined;
}
const end = /^Files: (\d+) total, \d+ passed, (\d+) failed$/.exec(text);
if (end && Number(end[2]) === 0) expected = Number(end[1]);
if (/^ERROR: HOME isolation breach/.test(text)) throw new Error(`${job}: isolation failure`);
}
}
// Only the runner's explicit no-work sentinel permits a job without a
// summary. Setup-only/truncated jobs must not disappear from the evidence.
if (lane === 'e2e' && !found.size && expected === undefined && !current &&
records.some(line => line.text === 'selected E2E: explicit empty selection; no tests launched')) continue;
if (current || expected === undefined || expected !== found.size) throw new Error(`${job}: incomplete ${lane} execution`);
for (const [file, duration] of found) weights.set(file, Math.max(weights.get(file) ?? 0, duration));
}
if (!weights.size) throw new Error(`${lane}: no complete timing data`);
return weights;
}
export function serializeWeights(weights: Map<string, number>): string {
const sorted = Array.from(weights.entries()).sort((a, b) =>
a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0,
);
const obj: Record<string, number> = {};
for (const [k, v] of sorted) obj[k] = v;
return JSON.stringify(obj, null, 2) + "\n";
for (const [file, value] of weights) {
if (!Number.isFinite(value) || value < 0) throw new Error(`invalid weight for ${file}: ${value}`);
}
return JSON.stringify(Object.fromEntries([...weights].sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))), null, 2) + "\n";
}
async function main(): Promise<number> {
const args = parseArgs(process.argv.slice(2));
console.error(
`[mine-shard-weights] source=${args.runId ?? args.fromFile ?? "<stdin>"}`,
);
let raw: string;
function gh(args: string[]): string {
const r = spawnSync("gh", args, { encoding: "utf8", maxBuffer: 256 * 1024 * 1024 });
if (r.status !== 0) throw new Error(`gh ${args.join(" ")} failed: ${r.stderr}`);
return r.stdout;
}
async function main(): Promise<void> {
let lane: Lane = "unit", run: string | undefined, input: string | undefined, out: string | undefined;
const argv = process.argv.slice(2);
for (let i = 0; i < argv.length; i++) {
const arg = argv[i]!;
if (arg === "--help" || arg === "-h") {
console.log("usage: bun run weights:mine [--lane unit|serial|e2e] [--run ID | --from-file PATH | <stdin>] [--out PATH]");
return;
}
if (!["--lane", "--run", "--run-id", "--from-file", "--out"].includes(arg) || !argv[i + 1] || argv[i + 1]!.startsWith("--")) throw new Error(`unknown/incomplete option: ${arg}`);
const value = argv[++i]!;
if (arg === "--lane") {
if (!["unit", "serial", "e2e"].includes(value)) throw new Error(`invalid lane: ${value}`);
lane = value as Lane;
} else if (arg === "--run" || arg === "--run-id") run = value;
else if (arg === "--from-file") input = value;
else out = resolve(value);
}
if (run && input) throw new Error("choose --run or --from-file, not both");
out ??= resolve(ROOT, `scripts/${OUTPUTS[lane]}.json`);
let commit: string | null = null;
let expectedJobs: string[] | undefined;
if (run) {
const info = JSON.parse(gh(["run", "view", run, "--json", "conclusion,headSha,jobs"]));
if (info.conclusion !== "success") throw new Error(`run ${run} is not successful`);
const jobs = info.jobs.filter((j: { name: string }) => isJob(j.name, lane));
if (!jobs.length || jobs.some((j: { conclusion: string }) => j.conclusion !== "success")) throw new Error(`run ${run} has no complete ${lane} lane`);
expectedJobs = jobs.map((j: { name: string }) => j.name);
commit = info.headSha;
}
const raw = run ? gh(["run", "view", run, "--log"]) : input ? readFileSync(input, "utf8") : await new Response(Bun.stdin.stream()).text();
const measured = mineWeights(raw, lane, { expectedJobs });
// Downloaded artifacts/stdin may cover only one shard. Only an authoritative
// complete GitHub unit/serial run replaces the full map; selected E2E always
// merges because its executed corpus depends on the diff.
const mergeExisting = !run || lane === "e2e";
const weights = mergeExisting && existsSync(out) ? loadWeights(out) : new Map<string, number>();
for (const [file, duration] of measured) weights.set(file, duration);
const metadata = { lane, unit: lane === "serial" ? "seconds" : "milliseconds", run: run ?? null, commit, source: run ? "github" : input ? "file" : "stdin", measuredFiles: measured.size, totalFiles: weights.size, mergeExisting };
const temp = `${out}.${process.pid}.tmp`;
try {
raw = await readSource(args);
} catch (e) {
console.error(`error: ${e instanceof Error ? e.message : String(e)}`);
return 1;
}
const events = parseLog(raw);
console.error(`[mine-shard-weights] parsed ${events.length} file-start events`);
if (events.length === 0) {
console.error(
"error: no ##[group]test/*.test.ts: events found in input. Was this a CI test run log?",
);
return 3;
}
const weights = computeWeights(events);
if (weights.size === 0) {
console.error("error: parsed events but extracted 0 weights (every job had ≤1 file?)");
return 3;
}
const json = serializeWeights(weights);
writeFileSync(args.out, json);
// Summary: min/median/max/total. Useful for spot-checking the file.
const values = Array.from(weights.values()).sort((a, b) => a - b);
const min = values[0]!;
const max = values[values.length - 1]!;
const median = values[Math.floor(values.length / 2)]!;
const total = values.reduce((a, b) => a + b, 0);
console.error(
`[mine-shard-weights] wrote ${weights.size} weights to ${args.out}`,
);
console.error(
`[mine-shard-weights] stats: min=${min}ms median=${median}ms max=${max}ms total=${(total / 1000).toFixed(1)}s`,
);
return 0;
}
if (import.meta.main) {
main().then((code) => process.exit(code));
writeFileSync(temp, serializeWeights(weights));
renameSync(temp, out);
writeFileSync(out.endsWith(".json") ? out.replace(/\.json$/, ".metadata.json") : `${out}.metadata.json`, JSON.stringify(metadata, null, 2) + "\n");
} finally { rmSync(temp, { force: true }); }
console.error(`[weights:mine] ${lane}: ${measured.size} measured, ${weights.size} total (${metadata.unit}); ${out}`);
}
if (import.meta.main) main().catch(error => { console.error(`weights:mine: ${error.message}`); process.exitCode = 1; });

View File

@@ -15,7 +15,7 @@
# shard, per-file bun startup (~1-2s) amortizes under the natural per-file
# test time of 5-10s.
#
# Exits non-zero on the first failing file so CI fails fast.
# Reports every file failure, then exits non-zero if any file failed.
#
# `--timeout=60000` matches the unit test suite. Bun's default is 5s,
# which is too tight for setupDB's TRUNCATE CASCADE on ~30 tables on
@@ -34,6 +34,8 @@
# Trap cleans up the tmpdir even on test failure.
set -euo pipefail
RUNNER_SHARD="${SHARD:-}"
unset SHARD
cd "$(dirname "$0")/.."
@@ -91,6 +93,33 @@ fi
E2E_TMP_HOME=$(mktemp -d "${TMPDIR:-/tmp}/gbrain-e2e.XXXXXX")
trap 'rm -rf "$E2E_TMP_HOME"' EXIT
# The foreground file runs as an owned child so signals interrupt wait promptly.
# Snapshot descendants before signalling: reparented children cannot be found later.
ACTIVE_E2E_PID=""
e2e_descendants() {
local child
for child in $(pgrep -P "$1" 2>/dev/null || true); do
e2e_descendants "$child"
printf '%s\n' "$child"
done
}
interrupt_e2e() {
local code="$1" descendants="" pid
trap '' INT TERM
if [ -n "$ACTIVE_E2E_PID" ]; then
descendants=$(e2e_descendants "$ACTIVE_E2E_PID")
kill -TERM "$ACTIVE_E2E_PID" 2>/dev/null || true
for pid in $descendants; do kill -TERM "$pid" 2>/dev/null || true; done
sleep 1
for pid in $descendants; do kill -KILL "$pid" 2>/dev/null || true; done
kill -KILL "$ACTIVE_E2E_PID" 2>/dev/null || true
wait "$ACTIVE_E2E_PID" 2>/dev/null || true
fi
exit "$code"
}
trap 'interrupt_e2e 130' INT
trap 'interrupt_e2e 143' TERM
export HOME="$E2E_TMP_HOME"
export GBRAIN_HOME="$E2E_TMP_HOME"
mkdir -p "$E2E_TMP_HOME/.gbrain"
@@ -115,7 +144,7 @@ mkdir -p "$E2E_TMP_HOME/.gbrain"
for _e2e_var in $(env | grep -oE '^(CONDUCTOR_|MCP_|OPENCLAW_|HERMES_|GROK_|OPENCODE_|GBRAIN_)[A-Za-z0-9_]*' | sort -u); do
case "$_e2e_var" in
GBRAIN_HOME) ;; # required for HOME isolation (set above) — keep
GBRAIN_PGLITE_SNAPSHOT) ;; # snapshot fast-path fixture (exported by ci-local.sh / runners) — keep
GBRAIN_PGLITE_SNAPSHOT|GBRAIN_NO_SNAPSHOT) ;; # snapshot fast-path fixture (exported by ci-local.sh / runners) — keep
GBRAIN_TEST_ALLOW_DATABASE_URL) ;; # #3485 preload opt-in (set above) — keep
GBRAIN_TEST_KEEP_PROVIDER_KEYS) ;; # provider-keys preload opt-in (set above) — keep
GBRAIN_CI_DISABLE_TEST_ENV_FILE) ;; # CI forbids loading checkout-local .env.testing — keep through Bun startup
@@ -147,33 +176,19 @@ else
files=(test/e2e/*.test.ts test/phantom-redirect-engine-parity.test.ts)
fi
# SHARD env (e.g. SHARD=1/4) keeps every M-th file starting at index N (1-indexed).
# Used by scripts/ci-local.sh to fan 4 shards in parallel against 4 postgres
# containers. Sequential execution within a shard is preserved (the TRUNCATE
# CASCADE no-race rationale at the top of this file still holds).
if [ -n "${SHARD:-}" ]; then
shard_n=${SHARD%/*}
shard_m=${SHARD#*/}
if ! printf '%s' "$shard_n" | grep -qE '^[0-9]+$' || \
! printf '%s' "$shard_m" | grep -qE '^[0-9]+$' || \
[ "$shard_n" -lt 1 ] || [ "$shard_m" -lt 1 ] || [ "$shard_n" -gt "$shard_m" ]; then
echo "ERROR: invalid SHARD=$SHARD (expected N/M with 1<=N<=M, both integers)" >&2
exit 1
fi
filtered=()
i=0
for f in "${files[@]}"; do
if [ $((i % shard_m + 1)) -eq "$shard_n" ]; then
filtered+=("$f")
fi
i=$((i + 1))
done
# ${filtered[@]:-} avoids "unbound variable" under `set -u` when no files matched.
files=("${filtered[@]:-}")
# If the empty placeholder slipped in, drop it.
if [ "${#files[@]}" -eq 1 ] && [ -z "${files[0]}" ]; then
files=()
# Weighted across isolated databases, sequential within each shard.
if [ -n "$RUNNER_SHARD" ]; then
if ! [[ "$RUNNER_SHARD" =~ ^[0-9]+/[0-9]+$ ]]; then
echo "ERROR: invalid SHARD=$RUNNER_SHARD (expected N/M)" >&2
exit 2
fi
shard_n=${RUNNER_SHARD%/*}
shard_m=${RUNNER_SHARD#*/}
selected=$(printf '%s\n' "${files[@]}" | bun scripts/sharding.ts "$shard_n" "$shard_m" --weights scripts/e2e-weights.json)
files=()
while IFS= read -r f; do
[ -n "$f" ] && files+=("$f")
done <<< "$selected"
fi
if [ "$DRY_RUN_LIST" = "1" ]; then
@@ -186,7 +201,7 @@ fi
if [ "${#files[@]}" -eq 0 ]; then
# Empty shard (e.g. SHARD=4/4 with only 3 files): nothing to do.
echo "No files for shard ${SHARD:-(unsharded)}; exiting clean."
echo "No files for shard ${RUNNER_SHARD:-(unsharded)}; exiting clean."
exit 0
fi
@@ -272,7 +287,13 @@ for f in "${files[@]}"; do
else
TIMEOUT_CMD=""
fi
if output=$($TIMEOUT_CMD bun test --timeout=60000 ${COVERAGE_ARGS[@]+"${COVERAGE_ARGS[@]}"} "$f" 2>&1); then
rc=0
$TIMEOUT_CMD bun test --timeout=60000 ${COVERAGE_ARGS[@]+"${COVERAGE_ARGS[@]}"} "$f" > "$E2E_TMP_HOME/current.log" 2>&1 &
ACTIVE_E2E_PID=$!
wait "$ACTIVE_E2E_PID" || rc=$?
ACTIVE_E2E_PID=""
output=$(cat "$E2E_TMP_HOME/current.log")
if [ "$rc" -eq 0 ]; then
if [ "$f" = "test/e2e/pgbouncer-teardown.test.ts" ] && \
[ "${GBRAIN_CI_REQUIRE_PGBOUNCER:-0}" = "1" ] && \
! printf '%s\n' "$output" | grep -qE '^[[:space:]]*[1-9][0-9]* pass$'; then

View File

@@ -20,6 +20,7 @@
# Invoked separately by run-unit-parallel.sh after the parallel pass succeeds.
#
# Knobs:
# SHARD=N/M weighted shard; unset runs every file
# GBRAIN_SERIAL_POOL=N pool width (default min(detect_cpus, 4),
# then memory-adapted; 1 restores the old
# fully-sequential behavior)
@@ -42,6 +43,25 @@ export GIT_CONFIG_COUNT=1 GIT_CONFIG_KEY_0="commit.gpgsign" GIT_CONFIG_VALUE_0="
# parallel runner) so the bunfig preload guard passes and nothing can
# reach a real brain.
unset DATABASE_URL GBRAIN_DATABASE_URL
SERIAL_SHARD="${SHARD:-}"
# Routing belongs to this invocation, never to tests that start nested runners.
unset SHARD
shard_n=1
shard_m=1
LANE=serial
if [ -n "$SERIAL_SHARD" ]; then
if ! [[ "$SERIAL_SHARD" =~ ^[0-9]+/[0-9]+$ ]]; then
echo "[serial-tests] ERROR: invalid SHARD=$SERIAL_SHARD (expected N/M)" >&2
exit 2
fi
shard_n=${SERIAL_SHARD%/*}
shard_m=${SERIAL_SHARD#*/}
if [ "$shard_n" -lt 1 ] || [ "$shard_m" -lt 1 ] || [ "$shard_n" -gt "$shard_m" ]; then
echo "[serial-tests] ERROR: invalid SHARD=$SERIAL_SHARD (need 1 <= N <= M)" >&2
exit 2
fi
LANE="serial-$shard_n"
fi
cd "$(dirname "$0")/.."
. scripts/lib/test-env.sh
@@ -89,60 +109,38 @@ while IFS= read -r f; do
files+=("$f")
done < <(find test -name '*.serial.test.ts' -not -path 'test/e2e/*' | sort)
if [ "${#files[@]}" -eq 0 ]; then
echo "[serial-tests] no *.serial.test.ts files found"
exit 0
fi
# --dry-run-list mirrors run-unit-shard.sh for inline checks/tests. Lists
# ALL discovered files, pooled and exclusive alike.
if [ "${1:-}" = "--dry-run-list" ]; then
printf '%s\n' "${files[@]}"
exit 0
fi
ensure_pglite_snapshot "serial-tests"
# Partition into pooled vs exclusive (exclusive entries missing from the
# discovered set are simply ignored — the list names repo files, and a
# sandbox copy of this script won't have them).
pool_files=()
exclusive_present=()
for f in "${files[@]}"; do
for f in ${files[@]+"${files[@]}"}; do
if is_exclusive "$f"; then
exclusive_present+=("$f")
[ "$shard_n" -ne 1 ] || exclusive_present+=("$f")
else
pool_files+=("$f")
fi
done
# LPT dispatch: heaviest-first into the work-stealing pool (descending-weight
# dispatch into a width-P pool IS longest-processing-time-first). Weights are
# ADVISORY (scripts/serial-weights.json, seconds, mined from
# .context/serial-durations.txt below); absent file / corrupt JSON / missing
# bun keep discovery order (bun, not node: bun-only dev machines are the
# common case — the script runs `bun test` right after). Absent key → corpus
# p75 (same doctrine as
# scripts/sharding.ts). Rank stability is all that matters — a wrong order
# costs idle tail, never correctness. The --dry-run-list output above stays
# discovery-ordered on purpose (pinned by test/scripts/run-serial-pool.test.ts).
if [ "${#pool_files[@]}" -gt 1 ] && [ -f scripts/serial-weights.json ] && command -v bun >/dev/null 2>&1; then
lpt_sorted=$(printf '%s\n' "${pool_files[@]}" | bun -e '
const fs = require("fs");
let w = {};
try { w = JSON.parse(fs.readFileSync("scripts/serial-weights.json", "utf8")); } catch {}
const files = fs.readFileSync(0, "utf8").split("\n").filter(Boolean);
const vals = Object.values(w).filter((v) => typeof v === "number").sort((a, b) => a - b);
const p75 = vals.length ? vals[Math.floor(vals.length * 0.75)] : 0;
const wt = (f) => (typeof w[f] === "number" ? w[f] : p75);
files.sort((a, b) => (wt(b) - wt(a)) || (a < b ? -1 : 1));
process.stdout.write(files.join("\n"));
' 2>/dev/null) || lpt_sorted=""
if [ -n "$lpt_sorted" ]; then
pool_files=()
while IFS= read -r f; do pool_files+=("$f"); done <<< "$lpt_sorted"
# One scheduler owns membership AND heaviest-first dispatch. Bad weights are
# advisory; a missing/broken scheduler is an execution error, never an empty run.
if [ "${#pool_files[@]}" -gt 0 ]; then
selected=$(printf '%s\n' "${pool_files[@]}" | bun scripts/sharding.ts \
"$shard_n" "$shard_m" --weights scripts/serial-weights.json --fallback-on-error)
pool_files=()
if [ -n "$selected" ]; then
while IFS= read -r f; do pool_files+=("$f"); done <<< "$selected"
fi
fi
ordered_files=()
if [ "${#pool_files[@]}" -gt 0 ]; then ordered_files+=("${pool_files[@]}"); fi
if [ "${#exclusive_present[@]}" -gt 0 ]; then ordered_files+=("${exclusive_present[@]}"); fi
if [ "${1:-}" = "--dry-run-list" ]; then
if [ "${#ordered_files[@]}" -gt 0 ]; then printf '%s\n' "${ordered_files[@]}" | sort; fi
exit 0
fi
if [ "${#ordered_files[@]}" -gt 0 ]; then ensure_pglite_snapshot "serial-tests"; fi
# ──────────────────────────────────────────────────────────────────────────
# Pool sizing: min(detect_cpus, 4) — each pooled bun process can hold a
@@ -180,14 +178,99 @@ command -v timeout >/dev/null 2>&1 && TIMEOUT_BIN="timeout"
[ -z "$TIMEOUT_BIN" ] && command -v gtimeout >/dev/null 2>&1 && TIMEOUT_BIN="gtimeout"
LOG_DIR=$(mktemp -d "${TMPDIR:-/tmp}/gbrain-serial.XXXXXX")
trap 'rm -rf "$LOG_DIR"' EXIT
RUN_SHA=$(git rev-parse HEAD 2>/dev/null || echo "${GITHUB_SHA:-unknown}")
if [ "${#ordered_files[@]}" -gt 0 ]; then
printf '%s\n' "${ordered_files[@]}" > "$LOG_DIR/selected.txt"
else
: > "$LOG_DIR/selected.txt"
fi
if [ -n "${COVERAGE_DIR:-}" ]; then
mkdir -p "$COVERAGE_DIR"
rm -f "$COVERAGE_DIR/lane-manifest.json"
fi
now_ms() {
local stamp
stamp=$(date +%s%3N)
case "$stamp" in
*[!0-9]*|'') bun -e 'console.log(Date.now())' ;;
*) printf '%s\n' "$stamp" ;;
esac
}
# Every attempt has its own files; concurrent workers never append shared JSON.
# The parent assembles one artifact on success, failure, or cancellation.
write_timings() {
local final_rc="$1"
mkdir -p .context 2>/dev/null || return 0
bun -e '
const fs = require("fs"), path = require("path");
const [dir, lane, sha, finalRc, timeout] = process.argv.slice(1);
const read = (name) => { try { return fs.readFileSync(path.join(dir, name), "utf8").trim(); } catch { return ""; } };
const selected = read("selected.txt").split("\n").filter(Boolean);
const byFile = new Map(selected.map(file => [file, []]));
for (const name of fs.readdirSync(dir).filter(n => /^\d+\.file$/.test(n)).sort((a,b) => parseInt(a)-parseInt(b))) {
const key = name.slice(0, -5), file = read(name), rawExit = read(key + ".exit");
const exitCode = rawExit === "" ? null : Number(rawExit);
const rawMs = read(key + ".duration-ms"), start = Number(read(key + ".start-ms"));
const durationMs = rawMs ? Number(rawMs) : start ? Math.max(0, Date.now() - start) : null;
const status = exitCode === 0 ? "pass" : exitCode === null ? "cancelled"
: exitCode === 124 || (exitCode === 137 && durationMs >= Number(timeout) * 1000) ? "timeout"
: exitCode === 137 || exitCode === 143 ? "external-kill" : "fail";
const attempts = byFile.get(file) ?? [];
attempts.push({ attempt: attempts.length + 1, durationMs, status, exitCode });
byFile.set(file, attempts);
}
const files = [...byFile].map(([file, attempts]) => {
const last = attempts.at(-1);
return { file, durationMs: last?.durationMs ?? null, status: last?.status ?? "not-run", attempts };
});
const result = { version: 1, lane, sha, complete: Number(finalRc) === 0, files };
const out = ".context/serial-timings.json";
fs.writeFileSync(out + ".tmp", JSON.stringify(result, null, 2) + "\n");
fs.renameSync(out + ".tmp", out);
' "$LOG_DIR" "$LANE" "$RUN_SHA" "$final_rc" "$PER_FILE_TIMEOUT" || \
echo "[serial-tests] warning: could not write timing artifact" >&2
}
running_exclusive=0
handle_interrupt() {
local code="$1" pid children="" roots=""
trap '' INT TERM
collect_descendants() {
local child
for child in $(pgrep -P "$1" 2>/dev/null); do
collect_descendants "$child"
printf '%s\n' "$child"
done
}
# Only live children owned by this runner: never signal by process name.
roots=$(jobs -rp)
for pid in $roots; do children="$children $(collect_descendants "$pid")"; done
for pid in $children $roots; do kill -TERM "$pid" 2>/dev/null || true; done
if [ "$running_exclusive" -eq 0 ]; then
sleep 1
for pid in $children $roots; do kill -KILL "$pid" 2>/dev/null || true; done
fi
for pid in $roots; do wait "$pid" 2>/dev/null || true; done
exit "$code"
}
finish() {
local code="$?"
trap - EXIT
write_timings "$code"
rm -rf "$LOG_DIR"
}
trap 'handle_interrupt 130' INT
trap 'handle_interrupt 143' TERM
trap finish EXIT
if [ -n "$TIMEOUT_BIN" ]; then
TIMEOUT_DESC="${PER_FILE_TIMEOUT}s via $TIMEOUT_BIN"
else
TIMEOUT_DESC="none (no timeout/gtimeout on PATH)"
fi
echo "[serial-tests] ${#files[@]} file(s): pool=$POOL (${#exclusive_present[@]} exclusive), per-file timeout=$TIMEOUT_DESC"
echo "[serial-tests] ${#ordered_files[@]} file(s): pool=$POOL (${#exclusive_present[@]} exclusive), lane=$LANE, per-file timeout=$TIMEOUT_DESC"
# Per-test timeout is 120s (not the fast-loop 60s): pooled contention can
# push a 30-50s file past 60s — the same flake class the slow lane hardened
@@ -196,15 +279,19 @@ echo "[serial-tests] ${#files[@]} file(s): pool=$POOL (${#exclusive_present[@]}
run_one_file() {
# $1 file, $2 log path, $3 exit-sentinel path, $4 wrap ("wrap"|"nowrap")
local f="$1" log="$2" exitf="$3" wrap="$4" rc=0
local key="${log%.log}" started finished
printf '%s\n' "$f" > "$key.file"
started=$(now_ms)
printf '%s\n' "$started" > "$key.start-ms"
# COVERAGE_DIR (opt-in): every bun process needs its OWN coverage dir — a
# second process reusing a dir OVERWRITES lcov.info. The log basename is
# unique per file (pool idx / exclusive i), so it keys the dir. Empty/unset
# COVERAGE_DIR leaves the exec line byte-identical to pre-coverage behavior.
local cov_args=()
if [ -n "${COVERAGE_DIR:-}" ]; then
local key
key=$(basename "$log" .log)
cov_args=(--coverage --coverage-reporter=lcov --coverage-dir="$COVERAGE_DIR/serial-$key")
local cov_key
cov_key=$(basename "$log" .log)
cov_args=(--coverage --coverage-reporter=lcov --coverage-dir="$COVERAGE_DIR/serial-$cov_key")
fi
if [ "$wrap" = "wrap" ] && [ -n "$TIMEOUT_BIN" ]; then
"$TIMEOUT_BIN" -k 15 "$PER_FILE_TIMEOUT" \
@@ -213,6 +300,9 @@ run_one_file() {
bun test --max-concurrency=1 --timeout=120000 ${cov_args[@]+"${cov_args[@]}"} "$f" > "$log" 2>&1 || rc=$?
fi
echo "$rc" > "$exitf"
finished=$(now_ms)
echo "$((finished - started))" > "$key.duration-ms"
echo "$(((finished - started) / 1000))" > "$key.dur"
}
start_epoch=$(date +%s)
@@ -225,10 +315,7 @@ if [ "${#pool_files[@]}" -gt 0 ]; then
sleep 0.05
done
(
s=$(date +%s)
run_one_file "$f" "$LOG_DIR/$idx.log" "$LOG_DIR/$idx.exit" "wrap"
e=$(date +%s)
echo "$((e - s))" > "$LOG_DIR/$idx.dur"
) &
idx=$((idx + 1))
done
@@ -238,10 +325,10 @@ fi
# Exclusive lane: sequential, unwrapped (see EXCLUSIVE_FILES comment).
if [ "${#exclusive_present[@]}" -gt 0 ]; then
for f in "${exclusive_present[@]}"; do
s=$(date +%s)
run_one_file "$f" "$LOG_DIR/$idx.log" "$LOG_DIR/$idx.exit" "nowrap"
e=$(date +%s)
echo "$((e - s))" > "$LOG_DIR/$idx.dur"
running_exclusive=1
run_one_file "$f" "$LOG_DIR/$idx.log" "$LOG_DIR/$idx.exit" "nowrap" &
wait "$!"
running_exclusive=0
idx=$((idx + 1))
done
fi
@@ -259,10 +346,6 @@ fi
# fails again is a real failure. Never a silent pass.
# anything else → real failure
# ──────────────────────────────────────────────────────────────────────────
ordered_files=()
if [ "${#pool_files[@]}" -gt 0 ]; then ordered_files+=("${pool_files[@]}"); fi
if [ "${#exclusive_present[@]}" -gt 0 ]; then ordered_files+=("${exclusive_present[@]}"); fi
fail_count=0
failed_files=()
rescue_files=()
@@ -273,7 +356,7 @@ rescue_files=()
# directly), so only PASSING files accumulate here — no double counting.
pass_total=0
i=0
for f in "${ordered_files[@]}"; do
for f in ${ordered_files[@]+"${ordered_files[@]}"}; do
dur="?"
[ -f "$LOG_DIR/$i.dur" ] && dur=$(cat "$LOG_DIR/$i.dur")
if [ ! -f "$LOG_DIR/$i.exit" ]; then
@@ -320,7 +403,10 @@ if [ "${#rescue_files[@]}" -gt 0 ]; then
wrap_mode="wrap"
is_exclusive "$f" && wrap_mode="nowrap"
s=$(date +%s)
run_one_file "$f" "$LOG_DIR/$i.log" "$LOG_DIR/$i.exit" "$wrap_mode"
[ "$wrap_mode" != "nowrap" ] || running_exclusive=1
run_one_file "$f" "$LOG_DIR/$i.log" "$LOG_DIR/$i.exit" "$wrap_mode" &
wait "$!"
running_exclusive=0
e=$(date +%s)
rc=$(cat "$LOG_DIR/$i.exit" 2>/dev/null || echo 1)
if [ "$rc" = "0" ]; then
@@ -341,7 +427,7 @@ fi
# Slowest-file table: feeds flake triage + future weight mining.
echo "[serial-tests] slowest files:"
i=0
for f in "${ordered_files[@]}"; do
for f in ${ordered_files[@]+"${ordered_files[@]}"}; do
[ -f "$LOG_DIR/$i.dur" ] && echo "$(cat "$LOG_DIR/$i.dur") $f"
i=$((i + 1))
done | sort -rn | awk 'NR<=10' | sed 's/^/ /'
@@ -352,7 +438,7 @@ done | sort -rn | awk 'NR<=10' | sed 's/^/ /'
if mkdir -p .context 2>/dev/null; then
{
di=0
for f in "${ordered_files[@]}"; do
for f in ${ordered_files[@]+"${ordered_files[@]}"}; do
[ -f "$LOG_DIR/$di.dur" ] && echo "$(cat "$LOG_DIR/$di.dur") $f"
di=$((di + 1))
done
@@ -373,8 +459,8 @@ fi
# treats a missing manifest as a degraded lane.
if [ -n "${COVERAGE_DIR:-}" ]; then
LCOV_COUNT=$(find "$COVERAGE_DIR" -name 'lcov.info' 2>/dev/null | grep -c '^' || true)
printf '{"lane":"serial","sha":"%s","lcovCount":%s,"complete":true}\n' \
"$(git rev-parse HEAD)" "${LCOV_COUNT:-0}" > "$COVERAGE_DIR/lane-manifest.json"
printf '{"lane":"%s","sha":"%s","lcovCount":%s,"complete":true}\n' \
"$LANE" "$RUN_SHA" "${LCOV_COUNT:-0}" > "$COVERAGE_DIR/lane-manifest.json"
fi
# bun-summary-format aggregate: run-unit-parallel.sh's headline counter
# (bun_summary_count awk: $1 numeric, $2 == "pass") reads this line — without

View File

@@ -18,6 +18,7 @@ cd "$(dirname "$0")/.."
. scripts/lib/test-env.sh
ensure_pglite_snapshot "run-slow-tests"
ensure_default_pglite_snapshot "run-slow-tests"
slow_files=()
while IFS= read -r f; do

View File

@@ -43,6 +43,7 @@
# .context/test-shards/ per-shard logs + exit codes (cleared at start)
set -uo pipefail
unset SHARD # This wrapper assigns its own children; ambient routing must not reach nested runners.
# Fixture tests that `git commit` in temp repos must not inherit the developer's
# global commit.gpgsign — a signing gpg-agent can OOM under full-suite memory

View File

@@ -3,7 +3,7 @@
#
# Runs the unit suite for a single shard. Excludes test/e2e/* (those are run
# by scripts/run-e2e.sh in the E2E phase). When SHARD=N/M is set, keeps every
# M-th file starting at index N (1-indexed); otherwise runs the full unit set.
# weighted partition N (1-indexed); otherwise runs the full unit set.
#
# Used by scripts/ci-local.sh to fan 4 unit-shard workers in parallel inside
# the runner container, each pinned to its own postgres shard for the
@@ -14,6 +14,9 @@
set -euo pipefail
RUNNER_SHARD="${SHARD:-}"
unset SHARD
# #3485: unit/slow tests need no database — strip ambient DB URLs at this
# wrapper boundary so the bunfig preload guard passes and nothing can reach a
# real brain. The e2e wrapper (run-e2e.sh) is the only lane that keeps them.
@@ -51,28 +54,25 @@ while IFS= read -r f; do
done < <(find test -name '*.test.ts' -not -path 'test/e2e/*' -not -name '*.slow.test.ts' -not -name '*.serial.test.ts' | sort)
files=()
if [ -n "${SHARD:-}" ]; then
shard_n=${SHARD%/*}
shard_m=${SHARD#*/}
if ! printf '%s' "$shard_n" | grep -qE '^[0-9]+$' || \
if [ -n "$RUNNER_SHARD" ]; then
shard_n=${RUNNER_SHARD%/*}
shard_m=${RUNNER_SHARD#*/}
if ! [[ "$RUNNER_SHARD" =~ ^[0-9]+/[0-9]+$ ]] || ! printf '%s' "$shard_n" | grep -qE '^[0-9]+$' || \
! printf '%s' "$shard_m" | grep -qE '^[0-9]+$' || \
[ "$shard_n" -lt 1 ] || [ "$shard_m" -lt 1 ] || [ "$shard_n" -gt "$shard_m" ]; then
echo "ERROR: invalid SHARD=$SHARD (expected N/M with 1<=N<=M, both integers)" >&2
echo "ERROR: invalid SHARD=$RUNNER_SHARD (expected N/M with 1<=N<=M, both integers)" >&2
exit 1
fi
i=0
for f in "${all_files[@]}"; do
if [ $((i % shard_m + 1)) -eq "$shard_n" ]; then
files+=("$f")
fi
i=$((i + 1))
done
selected=$(printf '%s\n' "${all_files[@]}" | bun scripts/sharding.ts "$shard_n" "$shard_m")
while IFS= read -r f; do
[ -n "$f" ] && files+=("$f")
done <<< "$selected"
else
files=("${all_files[@]}")
fi
if [ "${#files[@]}" -eq 0 ]; then
echo "[unit-shard ${SHARD:-(unsharded)}] no files; exiting clean."
echo "[unit-shard ${RUNNER_SHARD:-(unsharded)}] no files; exiting clean."
exit 0
fi
@@ -92,7 +92,7 @@ if ! printf '%s' "$MULT" | grep -qE '^[0-9]+$' || [ "$MULT" -lt 1 ]; then
fi
TEST_TIMEOUT_MS=$((60000 * MULT))
echo "[unit-shard ${SHARD:-(unsharded)}] running ${#files[@]} files (timeout=${TEST_TIMEOUT_MS}ms)"
echo "[unit-shard ${RUNNER_SHARD:-(unsharded)}] running ${#files[@]} files (timeout=${TEST_TIMEOUT_MS}ms)"
if [ -n "$MAX_CONC" ]; then
exec bun test --max-concurrency="$MAX_CONC" --timeout="$TEST_TIMEOUT_MS" "${files[@]}"
fi

View File

@@ -1,222 +1,262 @@
{
"test/admin-embed-spawn.serial.test.ts": 30,
"test/advisor-backup-coverage.serial.test.ts": 1,
"test/agent-register-guards.serial.test.ts": 13,
"test/agent-scheduler-contract.serial.test.ts": 26,
"test/agent-voice-cors.serial.test.ts": 2,
"test/admin-embed-spawn.serial.test.ts": 55,
"test/advisor-backup-coverage.serial.test.ts": 2,
"test/agent-install-backup.serial.test.ts": 114,
"test/agent-register-guards.serial.test.ts": 22,
"test/agent-scheduler-contract.serial.test.ts": 39,
"test/agent-voice-cors.serial.test.ts": 1,
"test/ai/header-transport.serial.test.ts": 0,
"test/ai/no-batch-cap-suppression.serial.test.ts": 1,
"test/ambient-recall-hooks.serial.test.ts": 1,
"test/ai/no-batch-cap-suppression.serial.test.ts": 0,
"test/ambient-recall-hooks.serial.test.ts": 2,
"test/ambient-writeback-lifecycle.serial.test.ts": 29,
"test/apply-migrations-list-db-state.serial.test.ts": 1,
"test/apply-migrations-pglite-spawn.serial.test.ts": 30,
"test/audit-parser-probe.serial.test.ts": 0,
"test/audit-slug-fallback.serial.test.ts": 0,
"test/apply-migrations-pglite-spawn.serial.test.ts": 49,
"test/audit-parser-probe.serial.test.ts": 1,
"test/audit-slug-fallback.serial.test.ts": 1,
"test/audit-synopsis.serial.test.ts": 0,
"test/audit/connection-audit.serial.test.ts": 0,
"test/autopilot-fanout-clamp.serial.test.ts": 2,
"test/autopilot-install-wrapper.serial.test.ts": 1,
"test/autopilot-launchd-lifecycle.serial.test.ts": 10,
"test/backfill-concurrency-clamp.serial.test.ts": 0,
"test/backup-cli.serial.test.ts": 0,
"test/backup-coverage.serial.test.ts": 1,
"test/backup-cli.serial.test.ts": 1,
"test/backup-coverage.serial.test.ts": 2,
"test/backup-invalidation.serial.test.ts": 2,
"test/backup-status-file.serial.test.ts": 0,
"test/binary-self-update-compiled.serial.test.ts": 0,
"test/backup-status-file.serial.test.ts": 1,
"test/binary-self-update-compiled.serial.test.ts": 1,
"test/bootstrap-dispatcher.serial.test.ts": 1,
"test/bootstrap-harness.serial.test.ts": 0,
"test/bootstrap-interview.serial.test.ts": 1,
"test/bootstrap-opencode-door.serial.test.ts": 2,
"test/bootstrap-harness.serial.test.ts": 1,
"test/bootstrap-interview.serial.test.ts": 0,
"test/bootstrap-opencode-door.serial.test.ts": 3,
"test/bootstrap-plugin-lane.serial.test.ts": 1,
"test/bootstrap-status.serial.test.ts": 0,
"test/bootstrap-status.serial.test.ts": 1,
"test/bootstrap-subcommand-help.serial.test.ts": 1,
"test/bootstrap-uninstall.serial.test.ts": 0,
"test/bootstrap-verify.serial.test.ts": 9,
"test/brain-allowlist.serial.test.ts": 2,
"test/brain-flag-routing.serial.test.ts": 13,
"test/brain-registry.serial.test.ts": 0,
"test/brainstorm/checkpoint.serial.test.ts": 1,
"test/bootstrap-verify.serial.test.ts": 13,
"test/brain-allowlist.serial.test.ts": 3,
"test/brain-durability-hook.serial.test.ts": 5,
"test/brain-flag-routing.serial.test.ts": 24,
"test/brain-registry.serial.test.ts": 1,
"test/brain-repo-durability.serial.test.ts": 2,
"test/brainstorm/checkpoint.serial.test.ts": 0,
"test/check-update-refresh.serial.test.ts": 0,
"test/checkpoint-harvest.serial.test.ts": 5,
"test/cli-help-without-brain.serial.test.ts": 10,
"test/cli-util-prompt.serial.test.ts": 2,
"test/code-callers-pin.serial.test.ts": 3,
"test/config-db-plane.serial.test.ts": 1,
"test/config-set-unknown-flag.serial.test.ts": 34,
"test/checkpoint-harvest.serial.test.ts": 9,
"test/cli-help-without-brain.serial.test.ts": 16,
"test/cli-util-prompt.serial.test.ts": 3,
"test/code-callers-pin.serial.test.ts": 11,
"test/code-def-refs-pin.serial.test.ts": 12,
"test/config-db-plane.serial.test.ts": 0,
"test/config-set-unknown-flag.serial.test.ts": 58,
"test/connection-manager.serial.test.ts": 0,
"test/connectors-credentials.serial.test.ts": 1,
"test/context-engine-checkpoint.serial.test.ts": 0,
"test/connectors-credentials.serial.test.ts": 0,
"test/context-engine-checkpoint.serial.test.ts": 1,
"test/context-pack-wb-lane.serial.test.ts": 3,
"test/contextual-retrieval-doctor.serial.test.ts": 2,
"test/contextual-synopsis-model.serial.test.ts": 0,
"test/core/audit-week-file.serial.test.ts": 0,
"test/core/cycle.serial.test.ts": 5,
"test/core/remediation-checkpoint.serial.test.ts": 0,
"test/creds-vault.serial.test.ts": 1,
"test/cross-modal-hybrid-integration.serial.test.ts": 4,
"test/cycle-lock-release-diagnostic.serial.test.ts": 1,
"test/cycle-lock-steal.serial.test.ts": 3,
"test/cycle-patterns-completed-outcome.serial.test.ts": 2,
"test/cycle-pglite-lock-ordering.serial.test.ts": 1,
"test/cycle-start-recovery-throw.serial.test.ts": 1,
"test/cycle/cycle-sync-uncommitted-warn.serial.test.ts": 2,
"test/contextual-synopsis-model.serial.test.ts": 1,
"test/core/audit-week-file.serial.test.ts": 1,
"test/core/cycle.serial.test.ts": 9,
"test/core/remediation-checkpoint.serial.test.ts": 1,
"test/creds-vault.serial.test.ts": 0,
"test/cross-modal-hybrid-integration.serial.test.ts": 6,
"test/cycle-lock-release-diagnostic.serial.test.ts": 3,
"test/cycle-lock-steal.serial.test.ts": 5,
"test/cycle-patterns-completed-outcome.serial.test.ts": 4,
"test/cycle-patterns-evidence-gate.serial.test.ts": 7,
"test/cycle-pglite-lock-ordering.serial.test.ts": 4,
"test/cycle-start-recovery-throw.serial.test.ts": 3,
"test/cycle/cycle-sync-uncommitted-warn.serial.test.ts": 4,
"test/db-repair.serial.test.ts": 0,
"test/delete-write-through-commit.serial.test.ts": 6,
"test/dispatch-response-meta.serial.test.ts": 1,
"test/doctor-autopilot-fanout-concurrency.serial.test.ts": 0,
"test/doctor-backup-coverage.serial.test.ts": 0,
"test/doctor-cli-smoke.serial.test.ts": 10,
"test/doctor-connection-classified.serial.test.ts": 1,
"test/doctor-cli-smoke.serial.test.ts": 14,
"test/doctor-connection-classified.serial.test.ts": 2,
"test/doctor-engine-fit.serial.test.ts": 0,
"test/doctor-no-migrate.serial.test.ts": 15,
"test/doctor-remote.serial.test.ts": 0,
"test/doctor-report-remote.serial.test.ts": 4,
"test/dream-owner-id-threading.serial.test.ts": 1,
"test/dream-postgres.serial.test.ts": 5,
"test/embed-backfill-cli-admission.serial.test.ts": 39,
"test/embed-exit-code-3037.serial.test.ts": 11,
"test/embed-input-type-wire.serial.test.ts": 0,
"test/embed-oversize-heal-drain.serial.test.ts": 3,
"test/doctor-eval-drift-branches.serial.test.ts": 1,
"test/doctor-memory-writeback.serial.test.ts": 3,
"test/doctor-no-migrate.serial.test.ts": 23,
"test/doctor-remote.serial.test.ts": 1,
"test/doctor-report-remote.serial.test.ts": 10,
"test/dream-drain-failure-summary.serial.test.ts": 3,
"test/dream-owner-id-threading.serial.test.ts": 3,
"test/dream-postgres.serial.test.ts": 9,
"test/embed-backfill-cli-admission.serial.test.ts": 77,
"test/embed-exit-code-3037.serial.test.ts": 21,
"test/embed-input-type-wire.serial.test.ts": 1,
"test/embed-oversize-heal-drain.serial.test.ts": 6,
"test/embed-partial-failure-3037.serial.test.ts": 6,
"test/embed-quarantine.serial.test.ts": 0,
"test/embed-stale-chunkless-pages.serial.test.ts": 5,
"test/embed-stale.serial.test.ts": 6,
"test/embed.serial.test.ts": 10,
"test/embedding-truth-predicates.serial.test.ts": 5,
"test/embed-quarantine.serial.test.ts": 1,
"test/embed-stale-chunkless-pages.serial.test.ts": 11,
"test/embed-stale-signature-pagination.serial.test.ts": 32,
"test/embed-stale.serial.test.ts": 15,
"test/embed.serial.test.ts": 11,
"test/embedding-truth-predicates.serial.test.ts": 12,
"test/eval-capture-db-plane.serial.test.ts": 1,
"test/eval-takes-quality-runner.serial.test.ts": 1,
"test/extract-cross-source-links-db.serial.test.ts": 2,
"test/extract-facts-embed-unavailable.serial.test.ts": 1,
"test/extract-facts-embed-warn.serial.test.ts": 1,
"test/facts-backstop-outcome.serial.test.ts": 3,
"test/facts-context-injection.serial.test.ts": 2,
"test/facts-mcp-allowlist.serial.test.ts": 2,
"test/facts-valid-from-context.serial.test.ts": 2,
"test/eval-takes-quality-runner.serial.test.ts": 3,
"test/extract-cross-source-links-db.serial.test.ts": 4,
"test/extract-facts-embed-unavailable.serial.test.ts": 3,
"test/extract-facts-embed-warn.serial.test.ts": 3,
"test/extract-include-frontmatter-config.serial.test.ts": 4,
"test/facts-backstop-outcome.serial.test.ts": 5,
"test/facts-context-injection.serial.test.ts": 4,
"test/facts-mcp-allowlist.serial.test.ts": 3,
"test/facts-valid-from-context.serial.test.ts": 3,
"test/filing-rules-resolution.serial.test.ts": 1,
"test/foreground-chat-gateway-init.serial.test.ts": 1,
"test/friction-diff.serial.test.ts": 0,
"test/fts-language-cache-isolation.serial.test.ts": 1,
"test/friction-diff.serial.test.ts": 1,
"test/fts-language-cache-isolation.serial.test.ts": 3,
"test/fts-language-migration.serial.test.ts": 0,
"test/fts-language-schema-replay.serial.test.ts": 5,
"test/fts-language.serial.test.ts": 1,
"test/fts-language-schema-replay.serial.test.ts": 9,
"test/fts-language.serial.test.ts": 0,
"test/gateway-reconfigure-clobber.serial.test.ts": 0,
"test/git-remote-durable.serial.test.ts": 2,
"test/git-remote-pull-file-transport.serial.test.ts": 0,
"test/git-remote-pull-file-transport.serial.test.ts": 1,
"test/google-connect-cmd.serial.test.ts": 1,
"test/graph-query-thin-client-source-flag.serial.test.ts": 1,
"test/guarded-http-tls.serial.test.ts": 1,
"test/harness-access.serial.test.ts": 20,
"test/harness-delivery-recovery.serial.test.ts": 4,
"test/hook-backup-notice.serial.test.ts": 0,
"test/hook-command.serial.test.ts": 2,
"test/hybrid-cache-hit-limit-honors-mode.serial.test.ts": 3,
"test/hybrid-cache-scope-poison.serial.test.ts": 2,
"test/hybrid-cached-hit-budget-meta.serial.test.ts": 2,
"test/hybrid-degraded-cache-meta.serial.test.ts": 3,
"test/hybrid-exclude-private-cache-fold.serial.test.ts": 2,
"test/hybrid-meta.serial.test.ts": 2,
"test/hybrid-pool-floor.serial.test.ts": 2,
"test/hybrid-salvage.serial.test.ts": 2,
"test/hybrid-search-lite.serial.test.ts": 3,
"test/hybrid-since-relative.serial.test.ts": 2,
"test/hybrid-types-cache-skip.serial.test.ts": 3,
"test/import-configured-root-guard.serial.test.ts": 11,
"test/import-json-stdout.serial.test.ts": 11,
"test/import-signature-stamp.serial.test.ts": 3,
"test/import-source-cr-mode.serial.test.ts": 2,
"test/init-picker-pty.serial.test.ts": 9,
"test/jobs-autopilot-cycle-braindir.serial.test.ts": 2,
"test/jobs-embed-background-dryrun.serial.test.ts": 2,
"test/jobs-embed-background-parity.serial.test.ts": 6,
"test/jobs-embed-stall-wiring.serial.test.ts": 2,
"test/hook-command.serial.test.ts": 3,
"test/hook-writeback-stop.serial.test.ts": 1,
"test/hybrid-adaptive-dedup-cache-plane.serial.test.ts": 5,
"test/hybrid-cache-hit-limit-honors-mode.serial.test.ts": 6,
"test/hybrid-cache-scope-poison.serial.test.ts": 7,
"test/hybrid-cached-hit-budget-meta.serial.test.ts": 3,
"test/hybrid-degraded-cache-meta.serial.test.ts": 5,
"test/hybrid-exclude-private-cache-fold.serial.test.ts": 5,
"test/hybrid-meta.serial.test.ts": 4,
"test/hybrid-pool-floor.serial.test.ts": 6,
"test/hybrid-reranker-skipped.serial.test.ts": 4,
"test/hybrid-salvage.serial.test.ts": 4,
"test/hybrid-search-lite.serial.test.ts": 5,
"test/hybrid-search-single-mode-read.serial.test.ts": 4,
"test/hybrid-since-relative.serial.test.ts": 4,
"test/hybrid-types-cache-skip.serial.test.ts": 4,
"test/import-cancellation.serial.test.ts": 5,
"test/import-configured-root-guard.serial.test.ts": 19,
"test/import-connection-budget.serial.test.ts": 1,
"test/import-json-stdout.serial.test.ts": 24,
"test/import-signature-stamp.serial.test.ts": 6,
"test/import-source-cr-mode.serial.test.ts": 4,
"test/init-picker-pty.serial.test.ts": 14,
"test/jobs-autopilot-cycle-braindir.serial.test.ts": 4,
"test/jobs-embed-background-dryrun.serial.test.ts": 3,
"test/jobs-embed-background-parity.serial.test.ts": 11,
"test/jobs-embed-stall-wiring.serial.test.ts": 3,
"test/jobs-gateway-refresh.serial.test.ts": 1,
"test/jobs-list-get-json.serial.test.ts": 2,
"test/jobs-stats-backpressure.serial.test.ts": 1,
"test/jobs-stats-divergence.serial.test.ts": 2,
"test/jobs-stats-private-queue.serial.test.ts": 2,
"test/jobs-subcommand-help.serial.test.ts": 8,
"test/llm-intent-hybrid-integration.serial.test.ts": 3,
"test/loops-extract-run.serial.test.ts": 2,
"test/mcp-backup-nag.serial.test.ts": 1,
"test/migrate-embeddings-boundary.serial.test.ts": 6,
"test/migrate-embeddings-flow.serial.test.ts": 8,
"test/migrate-embeddings-hardening.serial.test.ts": 19,
"test/migrate-embeddings-op-contract.serial.test.ts": 7,
"test/migrate-engine-completeness.serial.test.ts": 13,
"test/migrate-engine-page-copy-failure.serial.test.ts": 17,
"test/migrate-quiet-replay.serial.test.ts": 7,
"test/migration-in-process.serial.test.ts": 10,
"test/migration-orchestrator-v0_46_3.serial.test.ts": 7,
"test/migration-v0-29-1.serial.test.ts": 7,
"test/jobs-import-errors.serial.test.ts": 4,
"test/jobs-list-get-json.serial.test.ts": 4,
"test/jobs-stats-backpressure.serial.test.ts": 4,
"test/jobs-stats-divergence.serial.test.ts": 5,
"test/jobs-stats-private-queue.serial.test.ts": 4,
"test/jobs-subcommand-help.serial.test.ts": 12,
"test/keyword-relaxed-fusion.serial.test.ts": 3,
"test/llm-intent-hybrid-integration.serial.test.ts": 6,
"test/loops-extract-run.serial.test.ts": 5,
"test/mcp-backup-nag.serial.test.ts": 2,
"test/memorable-relay.serial.test.ts": 8,
"test/migrate-embeddings-boundary.serial.test.ts": 12,
"test/migrate-embeddings-flow.serial.test.ts": 11,
"test/migrate-embeddings-hardening.serial.test.ts": 26,
"test/migrate-embeddings-op-contract.serial.test.ts": 13,
"test/migrate-engine-completeness.serial.test.ts": 23,
"test/migrate-engine-page-copy-failure.serial.test.ts": 30,
"test/migrate-quiet-replay.serial.test.ts": 16,
"test/migration-in-process.serial.test.ts": 16,
"test/migration-orchestrator-v0_46_3.serial.test.ts": 12,
"test/migration-v0-29-1.serial.test.ts": 10,
"test/minions/delegated-execution.serial.test.ts": 15,
"test/model-config.serial.test.ts": 1,
"test/models-per-task-extract-atoms.serial.test.ts": 0,
"test/models-probe-max-output-tokens.serial.test.ts": 8,
"test/native-base-url-fold.serial.test.ts": 0,
"test/models-per-task-extract-atoms.serial.test.ts": 1,
"test/models-probe-max-output-tokens.serial.test.ts": 17,
"test/native-base-url-fold.serial.test.ts": 1,
"test/openai-latest.serial.test.ts": 0,
"test/ops-run-onboard-scope-gate.serial.test.ts": 1,
"test/page-summary-length.serial.test.ts": 1,
"test/pglite-disconnect-watchdog.serial.test.ts": 21,
"test/pglite-engine-disconnect.serial.test.ts": 34,
"test/ops-run-onboard-scope-gate.serial.test.ts": 0,
"test/page-summary-length.serial.test.ts": 0,
"test/pglite-disconnect-watchdog.serial.test.ts": 32,
"test/pglite-engine-disconnect.serial.test.ts": 66,
"test/pglite-hoisted-install.serial.test.ts": 1,
"test/pglite-reconnect.serial.test.ts": 4,
"test/pglite-repair-command.serial.test.ts": 4,
"test/pglite-snapshot-timezone.serial.test.ts": 6,
"test/pglite-wal-repair.serial.test.ts": 14,
"test/process-watchdog.serial.test.ts": 4,
"test/provider-sunset-doctor.serial.test.ts": 6,
"test/put-page-push-reporting.serial.test.ts": 2,
"test/query-cache-knobs-hash.serial.test.ts": 2,
"test/query-image-flag.serial.test.ts": 2,
"test/pglite-reconnect.serial.test.ts": 10,
"test/pglite-repair-command.serial.test.ts": 7,
"test/pglite-snapshot-timezone.serial.test.ts": 11,
"test/pglite-wal-repair.serial.test.ts": 29,
"test/process-watchdog.serial.test.ts": 5,
"test/provider-sunset-doctor.serial.test.ts": 13,
"test/put-page-push-reporting.serial.test.ts": 6,
"test/query-cache-knobs-hash.serial.test.ts": 4,
"test/query-image-flag.serial.test.ts": 5,
"test/query-image-mode-limit.serial.test.ts": 16,
"test/query-op-limit-mode-4356.serial.test.ts": 0,
"test/reconcile-links.serial.test.ts": 1,
"test/reindex-code-max-cost.serial.test.ts": 1,
"test/reindex-code-model-source.serial.test.ts": 2,
"test/reindex-code-nudge.serial.test.ts": 1,
"test/reindex-frontmatter-pglite-spawn.serial.test.ts": 12,
"test/recall-thin-client-fallback.serial.test.ts": 1,
"test/reconcile-links.serial.test.ts": 3,
"test/reindex-code-max-cost.serial.test.ts": 3,
"test/reindex-code-model-source.serial.test.ts": 4,
"test/reindex-code-nudge.serial.test.ts": 3,
"test/reindex-frontmatter-pglite-spawn.serial.test.ts": 19,
"test/reindex-search-vector.serial.test.ts": 0,
"test/remediation-run-extras.serial.test.ts": 2,
"test/remediation-run-loop.serial.test.ts": 1,
"test/remediation-run-extras.serial.test.ts": 3,
"test/remediation-run-loop.serial.test.ts": 0,
"test/rerank-no-key.serial.test.ts": 0,
"test/rerank-sunset-short-circuit.serial.test.ts": 0,
"test/schema-cli-database-path.serial.test.ts": 7,
"test/schema-cli-database-path.serial.test.ts": 11,
"test/schema-cli-db-config.serial.test.ts": 18,
"test/schema-pack-find-pack-successors.serial.test.ts": 0,
"test/schema-pack-load-active.serial.test.ts": 1,
"test/search-multimodal-no-embed.serial.test.ts": 2,
"test/search-telemetry-cache-wiring.serial.test.ts": 2,
"test/search-telemetry-disconnect-hang.serial.test.ts": 6,
"test/search/autocut-integration.serial.test.ts": 3,
"test/search-multimodal-no-embed.serial.test.ts": 5,
"test/search-telemetry-cache-wiring.serial.test.ts": 4,
"test/search-telemetry-disconnect-hang.serial.test.ts": 8,
"test/search/autocut-integration.serial.test.ts": 6,
"test/search/crag-escalation-limit.serial.test.ts": 3,
"test/search/embedding-column.serial.test.ts": 0,
"test/search/hybrid-reranker-integration.serial.test.ts": 3,
"test/seed-pglite.serial.test.ts": 13,
"test/self-upgrade-checkonly.serial.test.ts": 0,
"test/serve-source-guard.serial.test.ts": 0,
"test/serve-sync-ipc-killswitch.serial.test.ts": 10,
"test/serve-sync-runner.serial.test.ts": 6,
"test/skillopt/adversarial/concurrent-runs.serial.test.ts": 3,
"test/skillopt/cycle-phase-caps.serial.test.ts": 3,
"test/skillopt/run-skillopt-op.serial.test.ts": 3,
"test/sources-push-cli.serial.test.ts": 2,
"test/stdio-stderr-redirect.serial.test.ts": 0,
"test/sync-cost-gate.serial.test.ts": 15,
"test/sync-deferred-extract-queue.serial.test.ts": 16,
"test/sync-delete-trailing-hyphen.serial.test.ts": 6,
"test/sync-failure-ledger.serial.test.ts": 0,
"test/sync-inline-extract-stamps.serial.test.ts": 4,
"test/sync-malformed-path.serial.test.ts": 3,
"test/sync-metafile-skip.serial.test.ts": 3,
"test/sync-ops-pages.serial.test.ts": 3,
"test/sync-pull-failed-anchor.serial.test.ts": 4,
"test/sync-reconcile-db-only.serial.test.ts": 2,
"test/sync-rename-reconcile.serial.test.ts": 25,
"test/sync-resumable-import.serial.test.ts": 6,
"test/sync-soft-delete.serial.test.ts": 7,
"test/takes-fence-read-ops.serial.test.ts": 1,
"test/takes-mcp-allowlist.serial.test.ts": 2,
"test/search/hybrid-reranker-integration.serial.test.ts": 6,
"test/search/relational-rerank-pin-hybrid.serial.test.ts": 7,
"test/seed-pglite.serial.test.ts": 24,
"test/self-upgrade-checkonly.serial.test.ts": 1,
"test/serve-idle-sweep-source.serial.test.ts": 1,
"test/serve-source-guard.serial.test.ts": 1,
"test/serve-sync-ipc-killswitch.serial.test.ts": 16,
"test/serve-sync-runner.serial.test.ts": 11,
"test/skillopt/adversarial/concurrent-runs.serial.test.ts": 5,
"test/skillopt/cycle-phase-caps.serial.test.ts": 7,
"test/skillopt/run-skillopt-op.serial.test.ts": 7,
"test/source-resolver-default-write-guard.serial.test.ts": 6,
"test/sources-push-cli.serial.test.ts": 4,
"test/stdio-stderr-redirect.serial.test.ts": 1,
"test/sync-cost-gate.serial.test.ts": 28,
"test/sync-default-write-guard.serial.test.ts": 8,
"test/sync-deferred-extract-queue.serial.test.ts": 30,
"test/sync-delete-trailing-hyphen.serial.test.ts": 11,
"test/sync-failure-ledger.serial.test.ts": 1,
"test/sync-index-matches-tree.serial.test.ts": 7,
"test/sync-inline-extract-stamps.serial.test.ts": 9,
"test/sync-malformed-path.serial.test.ts": 8,
"test/sync-metafile-skip.serial.test.ts": 7,
"test/sync-ops-pages.serial.test.ts": 6,
"test/sync-pull-failed-anchor.serial.test.ts": 8,
"test/sync-reconcile-db-only.serial.test.ts": 5,
"test/sync-rename-reconcile.serial.test.ts": 46,
"test/sync-resumable-import.serial.test.ts": 13,
"test/sync-soft-delete.serial.test.ts": 11,
"test/takes-fence-read-ops.serial.test.ts": 3,
"test/takes-mcp-allowlist.serial.test.ts": 4,
"test/think-cli-source-flag.serial.test.ts": 1,
"test/think-extractive.serial.test.ts": 5,
"test/think-pipeline.serial.test.ts": 3,
"test/think-temporal-window.serial.test.ts": 1,
"test/unified-multimodal.serial.test.ts": 2,
"test/think-embed-question-wiring.serial.test.ts": 1,
"test/think-extractive.serial.test.ts": 11,
"test/think-gather-autocut.serial.test.ts": 1,
"test/think-pipeline.serial.test.ts": 5,
"test/think-temporal-window.serial.test.ts": 4,
"test/unified-multimodal.serial.test.ts": 6,
"test/upgrade-checkpoint.serial.test.ts": 0,
"test/upgrade.serial.test.ts": 3,
"test/v0_37_fix_wave.serial.test.ts": 0,
"test/v0_37_gap_fill.serial.test.ts": 6,
"test/watch-sigint.serial.test.ts": 6,
"test/worker-lock-renewal-e2e.serial.test.ts": 10,
"test/upgrade.serial.test.ts": 5,
"test/v0_37_fix_wave.serial.test.ts": 1,
"test/v0_37_gap_fill.serial.test.ts": 13,
"test/watch-sigint.serial.test.ts": 13,
"test/worker-lock-renewal-e2e.serial.test.ts": 13,
"test/worker-registry.serial.test.ts": 0,
"test/workspace-push.serial.test.ts": 5,
"test/write-through-commit.serial.test.ts": 2,
"test/autopilot-launchd-lifecycle.serial.test.ts": 12,
"test/brain-durability-hook.serial.test.ts": 5,
"test/brain-repo-durability.serial.test.ts": 4
"test/workspace-push.serial.test.ts": 9,
"test/write-through-commit.serial.test.ts": 4,
"test/writeback-nudge.serial.test.ts": 3,
"test/writeback-review-fixes.serial.test.ts": 3
}

View File

@@ -0,0 +1,10 @@
{
"lane": "serial",
"unit": "seconds",
"run": "34998157700",
"commit": "f1fbdfba193383785b555b507f5c817003944aea",
"source": "github",
"measuredFiles": 260,
"totalFiles": 260,
"mergeExisting": false
}

View File

@@ -12,7 +12,7 @@
*
* Weights live in scripts/test-weights.json — committed, mined from real
* CI run logs via scripts/mine-shard-weights.ts. Files absent from the
* weights map fall back to the corpus median (not zero — that would
* weights map fall back to the corpus p75 (not zero — that would
* favor unknown new files into the smallest shard, defeating balance).
*
* CLI:
@@ -20,8 +20,9 @@
* Reads test file list from stdin (one path per line). Prints the
* subset assigned to <shard-index> to stdout, one per line.
*
* bun run scripts/sharding.ts <shard-index> <total-shards> --files <glob>
* Walks the filesystem for matching files instead of reading stdin.
* bun run scripts/sharding.ts <shard-index> <total-shards> --weights <path>
* Uses lane-specific weights. --fallback-on-error warns and uses uniform
* weights if the advisory file is malformed (serial runner compatibility).
*
* Exit codes:
* 0 success
@@ -54,7 +55,10 @@ export class WeightsLoadError extends Error {
* caller decides whether to fall through to defaults or surface.
*/
export function loadWeights(path: string = DEFAULT_WEIGHTS_PATH): WeightMap {
if (!existsSync(path)) return new Map();
if (!existsSync(path)) {
console.error(`warning: weights missing at ${path}; using uniform fallback weights`);
return new Map();
}
let raw: string;
try {
raw = readFileSync(path, "utf8");
@@ -110,7 +114,7 @@ export function computeQuantile(values: number[], q: number): number {
export interface PartitionOpts {
/**
* Weight to assign files that are absent from the weights map. Defaults
* to the median of present weights (computed inside `partition`) so
* to the p75 of present weights (computed inside `partition`) so
* unknown new files cluster around the typical file's cost.
*
* Override mainly for tests; production callers should use the default.
@@ -130,7 +134,7 @@ export interface PartitionOpts {
* - Every file in `files` appears in exactly one returned shard.
* - If `files` is empty, returns `n` empty arrays.
* - If `n <= 0`, throws RangeError.
* - Files missing from `weights` get `opts.fallbackWeight` (or median).
* - Files missing from `weights` get `opts.fallbackWeight` (or p75).
*/
export function partition(
files: string[],
@@ -164,12 +168,8 @@ export function partition(
} else {
fallback = computeQuantile(Array.from(weights.values()), 0.75);
}
// Cold-start guard: if the weights map is empty AND no explicit
// fallback was supplied, every effective weight would be 0 and LPT
// collapses (all ties → lowest-index wins → every file in shard 0).
// Normalize fallback to 1 so LPT degenerates to round-robin, which is
// a strictly better default than "everything in shard 1" until
// test-weights.json gets mined.
// Keep a nonzero estimate for unknown files until timings are available.
// Equal-load file-count ties also distribute explicitly zero-valued weights.
if (fallback === 0 && opts.fallbackWeight === undefined) {
fallback = 1;
}
@@ -185,12 +185,14 @@ export function partition(
return a.path < b.path ? -1 : a.path > b.path ? 1 : 0;
});
// Running per-shard totals. argmin tiebreaker: lowest index (stable).
// Equal loads prefer fewer files, then lowest index. In particular, measured
// zero-duration files must not all collapse into the first shard.
const totals = new Array<number>(n).fill(0);
for (const t of tuples) {
let minIdx = 0;
for (let i = 1; i < n; i++) {
if (totals[i]! < totals[minIdx]!) minIdx = i;
if (totals[i]! < totals[minIdx]! ||
(totals[i] === totals[minIdx] && shards[i]!.length < shards[minIdx]!.length)) minIdx = i;
}
shards[minIdx]!.push(t.path);
totals[minIdx] = totals[minIdx]! + t.weight;
@@ -236,16 +238,28 @@ async function readStdinLines(): Promise<string[]> {
async function main(): Promise<number> {
const argv = process.argv.slice(2);
if (argv.length < 2) {
console.error("usage: bun run scripts/sharding.ts <shard-index> <total-shards>");
const positional: string[] = [];
let weightsPath: string | undefined;
let fallbackOnError = false;
for (let i = 0; i < argv.length; i++) {
const arg = argv[i]!;
if (arg === "--weights" && argv[i + 1] && !argv[i + 1]!.startsWith("--")) weightsPath = argv[++i];
else if (arg === "--fallback-on-error") fallbackOnError = true;
else if (arg.startsWith("--")) {
console.error(`error: unknown or incomplete option ${arg}`);
return 2;
} else positional.push(arg);
}
if (positional.length !== 2) {
console.error("usage: bun run scripts/sharding.ts <shard-index> <total-shards> [--weights PATH] [--fallback-on-error]");
console.error(" (reads file list from stdin, one path per line)");
return 2;
}
const idx = Number.parseInt(argv[0]!, 10);
const total = Number.parseInt(argv[1]!, 10);
if (!Number.isInteger(idx) || !Number.isInteger(total) || idx < 1 || total < 1 || idx > total) {
const idx = Number(positional[0]);
const total = Number(positional[1]);
if (!positional.every((v) => /^[0-9]+$/.test(v)) || !Number.isSafeInteger(idx) || !Number.isSafeInteger(total) || idx < 1 || total < 1 || idx > total) {
console.error(
`error: shard index ${argv[0]} / total ${argv[1]} invalid (need 1 <= index <= total, both ints)`,
`error: shard index ${positional[0]} / total ${positional[1]} invalid (need 1 <= index <= total, both ints)`,
);
return 2;
}
@@ -257,10 +271,11 @@ async function main(): Promise<number> {
}
let weights: WeightMap;
try {
weights = loadWeights();
weights = loadWeights(weightsPath);
} catch (e) {
console.error(`error: ${e instanceof Error ? e.message : String(e)}`);
return 1;
console.error(`${fallbackOnError ? "warning" : "error"}: ${e instanceof Error ? e.message : String(e)}`);
if (!fallbackOnError) return 1;
weights = new Map();
}
const shards = partition(files, weights, total);
for (const f of shards[idx - 1]!) {

View File

@@ -260,6 +260,7 @@ test/scripts/e2e-wiring.test.ts e2e file claim ratchet 5 readFileSync
test/scripts/e2e-wiring.test.ts selected-e2e job wiring 6 readFileSync
test/scripts/run-verify-parallel.test.ts guard registration ⇒ execution coverage 1 readFileSync
test/scripts/test-shard.slow.test.ts test-shard.sh — LPT balance contract 5 readFileSync
test/scripts/test-shard.slow.test.ts test-shard.sh — exclusion contract 6 readFileSync
test/serve-http-admin-route-guard.test.ts (file-level) 4 readFileSync
test/serve-http-github-webhook.test.ts handler wiring (source pin) 1 readFileSync
test/serve-http-mcp-transport-cleanup.test.ts POST /mcp transport cleanup (#2844) 2 readFileSync
Can't render this file because it contains an unexpected character in line 44 and column 63.

View File

@@ -30,6 +30,7 @@
# same assignment, so retries are reproducible.
set -euo pipefail
unset SHARD # Routing belongs to this wrapper, never to nested test runners.
DRY_RUN_LIST=0
if [ "${1:-}" = "--dry-run-list" ]; then
@@ -87,6 +88,7 @@ ALL_FILES=$(find test evals -name '*.test.ts' \
-not -name '*.serial.test.ts' \
-not -name 'eval-longmemeval-e2e.slow.test.ts' \
-not -name 'entity-resolve-perf.slow.test.ts' \
-not -name 'entity-card-perf.slow.test.ts' \
-not -name 'eval-brainbench-e2e.slow.test.ts' \
-not -path 'test/e2e/*' | sort)

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,10 @@
{
"lane": "unit",
"unit": "milliseconds",
"run": "34998157700",
"commit": "f1fbdfba193383785b555b507f5c817003944aea",
"source": "github",
"measuredFiles": 1745,
"totalFiles": 1745,
"mergeExisting": false
}

View File

@@ -1,6 +1,6 @@
# gbrain agent workspace — template
<!-- gbrain-template-stamp: 0.50.2.0 -->
<!-- gbrain-template-stamp: 0.50.4.0 -->
This repository is the **"Use this template"** distribution artifact for a
[gbrain](https://github.com/garrytan/gbrain) personal-agent workspace — the same

View File

@@ -105,7 +105,8 @@ function runsDestructiveSql(src: string): boolean {
}
/**
* Guarded either by the shared helper, by setupDB() (which calls it), or by an
* Guarded either by the shared helper, by setupDB()/setupLegacyEmbeddingDB()
* (which call it), or by an
* inline db-name floor. schema-drift.test.ts uses the last form: its pattern is
* deliberately different from the shared one (it also accepts *_e2e), so it is
* recognized rather than rewritten.
@@ -117,7 +118,7 @@ function runsDestructiveSql(src: string): boolean {
*/
function isGuarded(src: string): boolean {
for (const line of codeLines(src)) {
if (/\b(assertSafeE2eDatabaseUrl|setupDB)\s*\(/.test(line)) return true;
if (/\b(assertSafeE2eDatabaseUrl|setupDB|setupLegacyEmbeddingDB)\s*\(/.test(line)) return true;
if (/looksLikeTestDb/.test(line)) return true;
}
return false;
@@ -194,12 +195,14 @@ describe('scan classifiers (the gate must be able to fire)', () => {
test('isGuarded recognizes each accepted guard form and nothing else', () => {
expect(isGuarded('assertSafeE2eDatabaseUrl(url);')).toBe(true);
expect(isGuarded('await setupDB();')).toBe(true);
expect(isGuarded('await setupLegacyEmbeddingDB();')).toBe(true);
expect(isGuarded('if (!looksLikeTestDb(name)) return;')).toBe(true);
expect(isGuarded('// totally unguarded')).toBe(false);
});
test('isGuarded rejects comment-only guard mentions (fail-open hardening)', () => {
expect(isGuarded('// unlike setupDB() we connect directly')).toBe(false);
expect(isGuarded('// unlike setupLegacyEmbeddingDB() we connect directly')).toBe(false);
expect(isGuarded('/* assertSafeE2eDatabaseUrl( would go here */')).toBe(false);
expect(isGuarded(' * setupDB() runs SCHEMA_SQL — JSDoc mention')).toBe(false);
// multi-line block comment WITHOUT leading * per line — must not count

View File

@@ -17,6 +17,7 @@ import * as gateway from '../../src/core/ai/gateway.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../helpers/legacy-embedding-config.ts';
import { assertSafeE2eDatabaseUrl } from '../helpers/db-guard.ts';
import { withEnv } from '../helpers/with-env.ts';
import { readContentChunksEmbeddingDim } from '../../src/core/embedding-dim-check.ts';
const SOURCE = 'canonical-chunk-privacy-fixture';
const NUL = String.fromCharCode(0);
@@ -78,7 +79,13 @@ for (const kind of ['pglite', 'postgres'] as const) {
await imported(slug, 'Public canonical replacement <20> control.');
const before = await engine.getChunks(slug, { sourceId: SOURCE, requireSafeChunks: true });
expect(before).toHaveLength(1);
const vector = new Float32Array(1536); vector[0] = 1;
// This checks canonical storage, independent of the shared Postgres
// database's embedding profile. The isolated PGLite arm stays legacy.
const dimensions = kind === 'postgres'
? (await readContentChunksEmbeddingDim(engine)).dims
: LEGACY_EMBEDDING_CONFIG.embedding_dimensions;
if (!dimensions) throw new Error('Fixture requires a dimensioned text embedding column');
const vector = new Float32Array(dimensions); vector[0] = 1;
const poison = before[0].chunk_text.replace('<27>', LONE_HI) + NUL;
// Embedding refresh supplies body fields only; omitted code metadata
// retains its stored value under the upsert contract.

View File

@@ -11,13 +11,13 @@
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { setupDB, teardownDB, hasDatabase, getEngine } from './helpers.ts';
import { setupLegacyEmbeddingDB, teardownDB, hasDatabase, getEngine } from './helpers.ts';
import { runPhaseConsolidate } from '../../src/core/cycle/phases/consolidate.ts';
const RUN = hasDatabase();
const d = RUN ? describe : describe.skip;
beforeAll(async () => { if (RUN) await setupDB(); });
beforeAll(async () => { if (RUN) await setupLegacyEmbeddingDB(); });
afterAll(async () => { if (RUN) await teardownDB(); });
const oldDate = () => new Date(Date.now() - 30 * 60 * 60 * 1000).toISOString();

View File

@@ -16,7 +16,7 @@ import { mkdtempSync, writeFileSync, rmSync, mkdirSync } from 'fs';
import { join } from 'path';
import { execSync } from 'child_process';
import { tmpdir } from 'os';
import { hasDatabase, setupDB, teardownDB, getEngine, getConn } from './helpers.ts';
import { hasDatabase, setupLegacyEmbeddingDB, teardownDB, getEngine, getConn } from './helpers.ts';
// Mock embedBatch BEFORE importing runCycle so no real OpenAI calls happen
// even when the full cycle's embed phase runs.
@@ -61,7 +61,7 @@ describeE2E('E2E: runCycle against real Postgres', () => {
let repo: string;
beforeAll(async () => {
await setupDB();
await setupLegacyEmbeddingDB();
repo = makeGitRepo();
}, 30_000);

View File

@@ -14,7 +14,7 @@ import { mkdtempSync, writeFileSync, rmSync, mkdirSync } from 'fs';
import { join } from 'path';
import { execSync } from 'child_process';
import { tmpdir } from 'os';
import { hasDatabase, setupDB, teardownDB, getEngine, getConn } from './helpers.ts';
import { hasDatabase, setupLegacyEmbeddingDB, teardownDB, getEngine, getConn } from './helpers.ts';
// Mock embedBatch so embed phase doesn't call OpenAI.
mock.module('../../src/core/embedding.ts', () => ({
@@ -66,7 +66,7 @@ describeE2E('E2E: gbrain dream CLI against real Postgres', () => {
let repo: string;
beforeAll(async () => {
await setupDB();
await setupLegacyEmbeddingDB();
repo = makeGitRepo();
}, 30_000);

View File

@@ -20,7 +20,7 @@ import type { BrainEngine } from '../../src/core/engine.ts';
import { getSessionContextState, upsertSessionContextState } from '../../src/core/context/session-state.ts';
import { linkEntityIdentity, listEntityIdentities } from '../../src/core/entity-identity.ts';
import { buildEntityCard } from '../../src/core/verbs/entity-card.ts';
import { hasDatabase, setupDB, teardownDB, getEngine } from './helpers.ts';
import { hasDatabase, setupDB, setupLegacyEmbeddingDB, teardownDB, getEngine } from './helpers.ts';
import { TRAVERSE_PATH_ROW_CAP } from '../../src/core/engine-constants.ts';
import { DENSE_HUB_SLUG, DENSE_HUB_SPOKES, seedDenseHub } from '../helpers/dense-hub.ts';
@@ -115,7 +115,7 @@ describeBoth('Engine parity — Postgres vs PGLite', () => {
let pgliteEngine: PGLiteEngine;
beforeAll(async () => {
pgEngine = await setupDB();
pgEngine = await setupLegacyEmbeddingDB();
await seedEngine(pgEngine);
pgliteEngine = new PGLiteEngine();
@@ -1305,7 +1305,7 @@ describeBoth('Engine parity — relationalFanout', () => {
let pgliteEngine: PGLiteEngine;
beforeAll(async () => {
pgEngine = await setupDB();
pgEngine = await setupLegacyEmbeddingDB();
await seedRelational(pgEngine);
pgliteEngine = new PGLiteEngine();
await pgliteEngine.connect({});
@@ -1838,7 +1838,7 @@ describeBoth('Engine parity — CJK keyword fallback (#3986)', () => {
}
beforeAll(async () => {
pgEngine = await setupDB();
pgEngine = await setupLegacyEmbeddingDB();
await seedCJK(pgEngine);
pgliteEngine = new PGLiteEngine();
await pgliteEngine.connect({});
@@ -2351,7 +2351,7 @@ describeBoth('Engine parity — facts TTL read-time validity (WP5)', () => {
let pgliteEngine: PGLiteEngine;
beforeAll(async () => {
pgEngine = await setupDB();
pgEngine = await setupLegacyEmbeddingDB();
pgliteEngine = new PGLiteEngine();
await pgliteEngine.connect({});
await pgliteEngine.initSchema();

View File

@@ -14,7 +14,7 @@
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import type { PostgresEngine } from '../../src/core/postgres-engine.ts';
import { hasDatabase, setupDB, teardownDB } from './helpers.ts';
import { hasDatabase, setupLegacyEmbeddingDB, teardownDB } from './helpers.ts';
import { enrichEntity } from '../../src/core/enrichment-service.ts';
import { isUnverifiedExtraction, STATUS_VERIFIED, EXTRACTION_STATUS_KEY } from '../../src/core/extraction-review.ts';
import { operationsByName, type OperationContext } from '../../src/core/operations.ts';
@@ -38,7 +38,7 @@ function ctx(over: Partial<OperationContext> = {}): OperationContext {
d('extraction quarantine lane (live Postgres)', () => {
beforeAll(async () => {
engine = await setupDB();
engine = await setupLegacyEmbeddingDB();
}, 60_000);
afterAll(async () => {

View File

@@ -14,6 +14,9 @@ import * as db from '../../src/core/db.ts';
import { importFromContent } from '../../src/core/import-file.ts';
import { parseMarkdown } from '../../src/core/markdown.ts';
import { assertSafeE2eDatabaseUrl } from '../helpers/db-guard.ts';
import { configureGateway } from '../../src/core/ai/gateway.ts';
import { runSchemaTransition } from '../../src/core/embedding-migration.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../helpers/legacy-embedding-config.ts';
// Local opt-in configuration; container CI must not import developer credentials.
const envPath = resolve(import.meta.dir, '../../.env.testing');
@@ -141,6 +144,40 @@ export async function setupDB(): Promise<PostgresEngine> {
return engine;
}
/**
* Opt-in setup for fixtures that seed legacy-width text vectors. Bare CLI
* init tests can create the shared database at the new-install width; row
* truncation alone cannot make those columns fit a later 1536-d fixture.
* Ordinary setupDB preserves custom shapes for schema/migration tests.
*/
export async function setupLegacyEmbeddingDB(): Promise<PostgresEngine> {
configureGateway({ ...LEGACY_EMBEDDING_CONFIG, env: {} });
const target = await setupDB();
const dims = LEGACY_EMBEDDING_CONFIG.embedding_dimensions;
const columns = await target.executeRaw<{ table_name: string; type_name: string; dims: number }>(`
SELECT c.relname AS table_name, t.typname AS type_name, a.atttypmod AS dims
FROM pg_attribute a
JOIN pg_class c ON c.oid = a.attrelid
JOIN pg_namespace n ON n.oid = c.relnamespace
JOIN pg_type t ON t.oid = a.atttypid
WHERE n.nspname = 'public'
AND c.relname IN ('content_chunks', 'query_cache', 'facts', 'takes')
AND a.attname = 'embedding' AND a.attnum > 0 AND NOT a.attisdropped`);
if (columns.length !== 4 || columns.some(column => !['vector', 'halfvec'].includes(column.type_name))) {
throw new Error('Legacy embedding fixture requires all four text embedding columns');
}
if (columns.some(column => column.table_name !== 'takes' && Number(column.dims) !== dims)) {
await runSchemaTransition(target, dims);
}
const takes = columns.find(column => column.table_name === 'takes')!;
if (Number(takes.dims) !== dims) {
// Production transition deliberately leaves takes alone (search is
// trigram-based). This empty test table also receives fixed-width seeds.
await target.executeRaw(`ALTER TABLE takes ALTER COLUMN embedding TYPE ${takes.type_name}(${dims}) USING NULL`);
}
return target;
}
/**
* Disconnect from DB. Call in afterAll() of each test file.
*/

View File

@@ -13,12 +13,13 @@ import { join } from 'path';
import { execSync } from 'child_process';
import { tmpdir } from 'os';
import {
hasDatabase, setupDB, teardownDB, getEngine, getConn,
hasDatabase, setupDB, setupLegacyEmbeddingDB, teardownDB, getEngine, getConn,
importFixtures, importFixture, time, dumpDBState, FIXTURES_PATH,
} from './helpers.ts';
import { operationsByName, operations } from '../../src/core/operations.ts';
import type { OperationContext } from '../../src/core/operations.ts';
import { importFromContent } from '../../src/core/import-file.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../helpers/legacy-embedding-config.ts';
// Skip all E2E tests if no database is configured
const skip = !hasDatabase();
@@ -834,7 +835,7 @@ describeE2E('E2E: Idempotency', () => {
describeE2E('E2E: Setup Journey', () => {
beforeAll(async () => {
await setupDB();
await setupLegacyEmbeddingDB();
}, 30_000);
afterAll(teardownDB);
@@ -850,7 +851,8 @@ describeE2E('E2E: Setup Journey', () => {
// inits in the file honor persisted config per D5 (no flag needed).
const result = Bun.spawnSync({
cmd: ['bun', 'run', 'src/cli.ts', 'init', '--non-interactive', '--url', process.env.DATABASE_URL!,
'--embedding-model', 'openai:text-embedding-3-large'],
'--embedding-model', LEGACY_EMBEDDING_CONFIG.embedding_model,
'--embedding-dimensions', String(LEGACY_EMBEDDING_CONFIG.embedding_dimensions)],
cwd: cliCwd,
env: cliEnv(),
timeout: 15_000,
@@ -1345,7 +1347,7 @@ describeE2E('E2E: Doctor Command', () => {
let gbrainHome: string;
beforeAll(async () => {
await setupDB();
await setupLegacyEmbeddingDB();
await importFixtures();
// Isolate GBRAIN_HOME to a per-block tempdir so the developer's
// ~/.gbrain/migrations/completed.jsonl ledger doesn't leak in. Without
@@ -1381,12 +1383,14 @@ describeE2E('E2E: Doctor Command', () => {
// when ZEROENTROPY_API_KEY is in env) that mismatches the 1536d schema
// setupDB initialized, producing a WARN-status embedding_width_consistency
// check and exit 1. Mirrors the same pattern in 'Setup Journey'.
Bun.spawnSync({
const init = Bun.spawnSync({
cmd: ['bun', 'run', 'src/cli.ts', 'init', '--non-interactive',
'--url', process.env.DATABASE_URL!,
'--embedding-model', 'openai:text-embedding-3-large'],
'--embedding-model', LEGACY_EMBEDDING_CONFIG.embedding_model,
'--embedding-dimensions', String(LEGACY_EMBEDDING_CONFIG.embedding_dimensions)],
cwd: cliCwd, env: cliEnv(), timeout: 15_000,
});
expect(init.exitCode, new TextDecoder().decode(init.stderr)).toBe(0);
const result = Bun.spawnSync({
cmd: ['bun', 'run', 'src/cli.ts', 'doctor'],
cwd: cliCwd,

View File

@@ -19,10 +19,9 @@
* postgres engine" `process.exit(1)`. helpers.ts captures DATABASE_URL at
* module load, so this file deletes it from process.env for the duration
* (restored in afterAll) and passes the target URL explicitly via --url.
* - The live Postgres schema sizes content_chunks.embedding at vector(1536)
* while an unconfigured gateway defaults PGLite to 1280d. The gateway is
* configured at 1536 (and the fixture config.json pins it) so the seeded
* vectors land on the target without a dims mismatch.
* - Both fixture engines explicitly use the legacy test embedding shape.
* Earlier CLI-init files can create the shared Postgres at a different
* width, so setup transitions the cleared target before seeding vectors.
*/
import { afterAll, beforeAll, describe, expect, test } from 'bun:test';
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'fs';
@@ -30,16 +29,17 @@ import { tmpdir } from 'os';
import { join, resolve } from 'path';
import { PGLiteEngine } from '../../src/core/pglite-engine.ts';
import { runMigrateEngine } from '../../src/commands/migrate-engine.ts';
import { configureGateway, resetGateway } from '../../src/core/ai/gateway.ts';
import { resetGateway } from '../../src/core/ai/gateway.ts';
import type { BrainEngine } from '../../src/core/engine.ts';
import { hasDatabase, setupDB, teardownDB, getEngine } from './helpers.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../helpers/legacy-embedding-config.ts';
import { hasDatabase, setupLegacyEmbeddingDB, teardownDB, getEngine } from './helpers.ts';
const describePg = hasDatabase() ? describe : describe.skip;
// Captured at module load, before beforeAll deletes it from process.env.
const DB_URL = process.env.DATABASE_URL ?? '';
const REPO_ROOT = resolve(import.meta.dir, '../..');
const EMBED_DIMS = 1536;
const EMBED_DIMS = LEGACY_EMBEDDING_CONFIG.embedding_dimensions;
/** Deterministic 1536-d vector; v[0] = seed/8 is float4-exact for the
* round-trip spot check on the Postgres side. */
@@ -102,20 +102,14 @@ describePg('migrate-engine whole-brain PGLite to Postgres (D2)', () => {
beforeAll(async () => {
if (!DB_URL) throw new Error('DATABASE_URL must be set for this e2e file');
// Postgres clean slate FIRST (helpers captured DATABASE_URL at import).
await setupDB();
// Pin embedding sizing to the live Postgres schema (vector(1536)) so the
// fresh PGLite brain sizes its columns identically.
configureGateway({ embedding_model: 'openai:text-embedding-3-small', embedding_dimensions: EMBED_DIMS, env: {} });
await setupLegacyEmbeddingDB();
// Isolated gbrain home with a real pglite file config — the SOURCE brain.
mkdirSync(gbrainDir, { recursive: true });
writeFileSync(configFile, JSON.stringify({
engine: 'pglite',
database_path: pgliteDir,
embedding_model: 'openai:text-embedding-3-small',
embedding_dimensions: EMBED_DIMS,
...LEGACY_EMBEDDING_CONFIG,
}, null, 2));
process.env.GBRAIN_HOME = tmpBase;
// See header: an exported DATABASE_URL makes loadConfig() infer postgres,

View File

@@ -16,6 +16,7 @@
import { afterAll, beforeAll, beforeEach, describe, expect, test } from 'bun:test';
import { PostgresEngine } from '../../src/core/postgres-engine.ts';
import { assertSafeE2eDatabaseUrl } from '../helpers/db-guard.ts';
import { readContentChunksEmbeddingDim } from '../../src/core/embedding-dim-check.ts';
const DATABASE_URL = process.env.DATABASE_URL;
const skip = !DATABASE_URL;
@@ -26,12 +27,16 @@ if (skip) {
describe.skipIf(skip)('multimodal v0.27.1 against real Postgres', () => {
let pg: PostgresEngine;
let textDimensions: number;
beforeAll(async () => {
pg = new PostgresEngine();
assertSafeE2eDatabaseUrl(DATABASE_URL!);
await pg.connect({ database_url: DATABASE_URL! });
await pg.initSchema();
const { dims } = await readContentChunksEmbeddingDim(pg);
if (!dims) throw new Error('Fixture requires a dimensioned text embedding column');
textDimensions = dims;
}, 60_000);
afterAll(async () => {
@@ -174,10 +179,10 @@ describe.skipIf(skip)('multimodal v0.27.1 against real Postgres', () => {
}, 30_000);
test('searchVector with embeddingColumn=embedding_image returns image rows on Postgres', async () => {
// Seed: one text page (1536-dim primary embedding) and two image pages
// (1024-dim embedding_image).
const textVec = new Float32Array(1536);
for (let i = 0; i < 1536; i++) textVec[i] = i / 1536;
// Text uses the shared database's primary width; images always use
// their separate 1024-dim embedding_image column.
const textVec = new Float32Array(textDimensions);
for (let i = 0; i < textDimensions; i++) textVec[i] = i / textDimensions;
await pg.putPage('notes/text-only', {
type: 'note', title: 'text only', compiled_truth: 'body', timeline: '',
});
@@ -215,7 +220,7 @@ describe.skipIf(skip)('multimodal v0.27.1 against real Postgres', () => {
});
const slugs = hits.map(h => h.slug);
expect(slugs).toContain('photos/b');
// Modality filter excludes the text page even though dim mismatches.
// Column routing excludes the text page, regardless of its vector width.
expect(slugs).not.toContain('notes/text-only');
// Nearest-first ordering.
expect(hits[0].slug).toBe('photos/b');
@@ -223,8 +228,8 @@ describe.skipIf(skip)('multimodal v0.27.1 against real Postgres', () => {
test('searchKeyword hides image rows by default (modality filter on Postgres)', async () => {
// Seed text + image pages with chunk_text the FTS would normally match.
const textVec = new Float32Array(1536);
for (let i = 0; i < 1536; i++) textVec[i] = (i + 1) / 1536;
const textVec = new Float32Array(textDimensions);
for (let i = 0; i < textDimensions; i++) textVec[i] = (i + 1) / textDimensions;
await pg.putPage('notes/keyword', {
type: 'note', title: 'keyword', compiled_truth: 'sunset photo at the beach', timeline: '',
});

View File

@@ -17,7 +17,7 @@ import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:tes
import { mkdtempSync, writeFileSync, rmSync, mkdirSync, existsSync } from 'fs';
import { join } from 'path';
import { tmpdir } from 'os';
import { hasDatabase, setupDB, teardownDB, getEngine } from './helpers.ts';
import { hasDatabase, setupLegacyEmbeddingDB, teardownDB, getEngine } from './helpers.ts';
import { withEnv } from '../helpers/with-env.ts';
import { runExtractFacts } from '../../src/core/cycle/extract-facts.ts';
// v0.40: per-source lock id replaces the legacy bare SYNC_LOCK_ID constant.
@@ -28,7 +28,7 @@ const describeMaybe = SKIP ? describe.skip : describe;
beforeAll(async () => {
if (SKIP) return;
await setupDB();
await setupLegacyEmbeddingDB();
});
afterAll(async () => {

View File

@@ -2,7 +2,7 @@ import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, test } fr
import { mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { getEngine, hasDatabase, setupDB, teardownDB } from './helpers.ts';
import { getEngine, hasDatabase, setupLegacyEmbeddingDB, teardownDB } from './helpers.ts';
import { resetPgliteStateNarrow } from '../helpers/reset-pglite.ts';
import { configureGateway, resetGateway, __setEmbedTransportForTests } from '../../src/core/ai/gateway.ts';
import { dispatchToolCall } from '../../src/mcp/dispatch.ts';
@@ -25,7 +25,7 @@ async function put(body: string, pageSlug = slug) {
}
d('Postgres put_page persistence', () => {
beforeAll(async () => { await setupDB(); });
beforeAll(async () => { await setupLegacyEmbeddingDB(); });
afterAll(async () => { await teardownDB(); });
beforeEach(async () => {
await resetPgliteStateNarrow(getEngine(), ['pages', 'config', 'gbrain_cycle_locks', 'minion_jobs']);

View File

@@ -13,7 +13,7 @@
* - MCP dispatch with per-token allow-list (defense-in-depth Codex P0 #3)
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { setupDB, teardownDB, hasDatabase, getEngine } from './helpers.ts';
import { setupLegacyEmbeddingDB, teardownDB, hasDatabase, getEngine } from './helpers.ts';
import { extractTakesFromDb } from '../../src/core/cycle/extract-takes.ts';
import { dispatchToolCall } from '../../src/mcp/dispatch.ts';
import { TAKES_FENCE_BEGIN, TAKES_FENCE_END } from '../../src/core/takes-fence.ts';
@@ -26,7 +26,7 @@ let acmePageId: number;
beforeAll(async () => {
if (!RUN) return;
const engine = await setupDB();
const engine = await setupLegacyEmbeddingDB();
const alice = await engine.putPage('people/alice-example', {
title: 'Alice', type: 'person', compiled_truth: '## Takes\n',
});

View File

@@ -21,16 +21,28 @@
import { beforeAll, describe, expect, test } from 'bun:test';
import { cpSync, mkdirSync, mkdtempSync, readFileSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { join, resolve } from 'node:path';
const REPO = process.cwd();
const CLI = join(REPO, 'src', 'cli.ts');
let root: string;
let fixtures: string;
let gold: string;
let cliSnapshot = process.env.GBRAIN_TEST_DEFAULT_SNAPSHOT;
type RunResult = { exitCode: number; stdout: string; stderr: string };
function brainBenchEnv(): NodeJS.ProcessEnv {
const snapshot = process.env.GBRAIN_NO_SNAPSHOT === '1' ? undefined : cliSnapshot;
return {
...process.env,
GBRAIN_QUIET: '1',
// Bare CLI defaults are zembed/1280; its parent bun test uses legacy/1536.
// An absent default fixture falls back to cold init without a wrong-shape tar.
GBRAIN_PGLITE_SNAPSHOT: snapshot ? resolve(REPO, snapshot) : '',
};
}
function withDefaultCommittedBaseline(args: string[]): string[] {
// Foreign-corpus runs opt OUT of the repo's committed baseline: the
// poisoning defense requires any committed-vs-main divergence to match the
@@ -43,7 +55,7 @@ function withDefaultCommittedBaseline(args: string[]): string[] {
function run(args: string[], cwd = REPO): RunResult {
const proc = Bun.spawnSync(['bun', CLI, 'eval', 'brainbench', ...withDefaultCommittedBaseline(args)], {
cwd,
env: { ...process.env, GBRAIN_QUIET: '1' },
env: brainBenchEnv(),
stdout: 'pipe',
stderr: 'pipe',
});
@@ -57,7 +69,7 @@ function run(args: string[], cwd = REPO): RunResult {
async function runAsync(args: string[], cwd = REPO): Promise<RunResult> {
const proc = Bun.spawn(['bun', CLI, 'eval', 'brainbench', ...withDefaultCommittedBaseline(args)], {
cwd,
env: { ...process.env, GBRAIN_QUIET: '1' },
env: brainBenchEnv(),
stdout: 'pipe',
stderr: 'pipe',
});
@@ -103,6 +115,14 @@ async function runBatch(jobs: Array<[string, string[], string?]>, width = 2): Pr
}
beforeAll(async () => {
// Direct `bun test` invocations need the same profile as the slow-lane runner.
if (!cliSnapshot && process.env.GBRAIN_NO_SNAPSHOT !== '1') {
const build = Bun.spawnSync([
process.execPath, join(REPO, 'scripts/build-pglite-snapshot.ts'), '--profile', 'default',
], { cwd: REPO, env: { ...process.env }, stdout: 'pipe', stderr: 'pipe' });
if (build.exitCode === 0) cliSnapshot = join(REPO, 'test/fixtures/pglite-snapshot-default.tar');
else console.warn(`[brainbench] Default snapshot unavailable; using cold initialization. ${build.stderr.toString().trim()}`);
}
root = mkdtempSync(join(tmpdir(), 'bb-e2e-'));
fixtures = join(root, 'fixtures');
gold = join(root, 'gold');
@@ -183,6 +203,10 @@ describe('exit contract over a multi-brain run (PGLite exitCode-hijack guard)',
expect(doc.cells.length).toBeGreaterThan(0);
expect(doc.seed_failures).toEqual([]);
expect(r.stderr).not.toContain('not a git repository');
if (process.env.GBRAIN_TEST_DEFAULT_SNAPSHOT && process.env.GBRAIN_NO_SNAPSHOT !== '1') {
expect(r.stderr).not.toContain('embedding shape mismatch');
expect(r.stderr).not.toContain('migration(s) applied');
}
}, 60_000);
test('clean run: exit 0, --out is complete valid JSON with the glossary block', () => {
@@ -389,7 +413,7 @@ describe('run-all once-per-sweep semantics (decision 16)', () => {
const outDir = mkdtempSync(join(tmpdir(), 'bb-runall-'));
const proc = Bun.spawnSync(
['bun', 'src/cli.ts', 'eval', 'run-all', '--suites', 'brainbench', '--modes', 'conservative,balanced', '--output', outDir],
{ cwd: REPO, env: { ...process.env }, stdout: 'pipe', stderr: 'pipe' },
{ cwd: REPO, env: brainBenchEnv(), stdout: 'pipe', stderr: 'pipe' },
);
expect(proc.exitCode).toBe(0);
const lines = readFileSync(join(outDir, 'eval-results.jsonl'), 'utf-8').trim().split('\n');

View File

@@ -0,0 +1,17 @@
// Executed only in an owned subprocess by build-pglite-snapshot.test.ts.
// Kill the process after the real tar rename to exercise interrupted publication.
import { mock } from 'bun:test';
import * as fs from 'node:fs';
const realFs = { ...fs };
mock.module('node:fs', () => ({ ...realFs, renameSync(...args: Parameters<typeof fs.renameSync>) {
realFs.renameSync(...args);
if (args[1] === process.argv[3]) process.kill(process.pid, 'SIGKILL');
}}));
const { buildPgliteSnapshot } = await import('../../scripts/build-pglite-snapshot.ts');
await buildPgliteSnapshot('default', {
fixtureDir: process.argv[2],
log: () => {},
buildData: async () => new TextEncoder().encode('complete new tar'),
});

View File

@@ -43,6 +43,7 @@ const KEEP_EXACT = new Set([
'GBRAIN_DATABASE_URL', // e2e DB target; database-url-guard-preload (registered first) already vetoed un-opted runs
'GBRAIN_MODEL_DISCOVERY', // operator override provider-keys-preload deliberately respects
'GBRAIN_PGLITE_SNAPSHOT', // schema-snapshot fast path exported by every unit runner (scripts/lib/test-env.sh)
'GBRAIN_NO_SNAPSHOT', // cold-path opt-out must survive this preload and reach CLI children
'GBRAIN_PGBOUNCER_URL', // explicit pooled test target supplied by ci-local
'GBRAIN_PGBOUNCER_DIRECT_URL', // admin connection used to create the isolated pooler test DB
'GBRAIN_COMPILED_BIN', // heavy-lane compile-once binary (agent-harness.ts ensureCompiledGbrain)
@@ -79,3 +80,8 @@ if (process.env.GBRAIN_TEST_KEEP_AMBIENT_ENV !== '1') {
);
}
}
if (process.env.GBRAIN_NO_SNAPSHOT === '1') {
delete process.env.GBRAIN_PGLITE_SNAPSHOT;
delete process.env.GBRAIN_TEST_DEFAULT_SNAPSHOT;
}

View File

@@ -109,6 +109,8 @@ describe('operator-env-preload (#4023)', () => {
const probed = [
'GBRAIN_MODEL_DISCOVERY',
'GBRAIN_PGLITE_SNAPSHOT',
'GBRAIN_TEST_DEFAULT_SNAPSHOT',
'GBRAIN_NO_SNAPSHOT',
'GBRAIN_PGBOUNCER_URL',
'GBRAIN_PGBOUNCER_DIRECT_URL',
'GBRAIN_CI_REQUIRE_PGBOUNCER',
@@ -123,6 +125,8 @@ describe('operator-env-preload (#4023)', () => {
// preload's own default of 'off' when the var is absent.
GBRAIN_MODEL_DISCOVERY: '1',
GBRAIN_PGLITE_SNAPSHOT: 'probe-snapshot.tar',
GBRAIN_TEST_DEFAULT_SNAPSHOT: '/probe/default-snapshot.tar',
GBRAIN_NO_SNAPSHOT: '0',
GBRAIN_PGBOUNCER_URL: 'postgresql://pooler.example/gbrain_test',
GBRAIN_PGBOUNCER_DIRECT_URL: 'postgresql://direct.example/gbrain_test',
GBRAIN_CI_REQUIRE_PGBOUNCER: '1',
@@ -137,6 +141,16 @@ describe('operator-env-preload (#4023)', () => {
for (const name of probed) expect({ [name]: r.report[name] }).toEqual({ [name]: ambient[name] });
}, 30_000);
test('cold snapshot opt-out survives preload and clears both inherited snapshot paths', () => {
const r = runProbe({
GBRAIN_NO_SNAPSHOT: '1',
GBRAIN_PGLITE_SNAPSHOT: '/probe/legacy.tar',
GBRAIN_TEST_DEFAULT_SNAPSHOT: '/probe/default.tar',
}, ['GBRAIN_NO_SNAPSHOT', 'GBRAIN_PGLITE_SNAPSHOT', 'GBRAIN_TEST_DEFAULT_SNAPSHOT']);
expect(r.exitCode).toBe(0);
expect(r.report).toEqual({ GBRAIN_NO_SNAPSHOT: '1', GBRAIN_PGLITE_SNAPSHOT: null, GBRAIN_TEST_DEFAULT_SNAPSHOT: null });
}, 30_000);
test('keeps GBRAIN_DATABASE_URL in the opted-in e2e lane', () => {
// The e2e wrappers export GBRAIN_DATABASE_URL as the lane's target; a
// blanket GBRAIN_* delete here would sever it AFTER the guard preload

View File

@@ -0,0 +1,298 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join, resolve } from 'node:path';
import { buildPgliteSnapshot, parseSnapshotProfile, reclaimDeadSnapshotOwner, snapshotLockIdentity, snapshotProfile } from '../../scripts/build-pglite-snapshot.ts';
import { DEFAULT_EMBEDDING_DIMENSIONS, DEFAULT_EMBEDDING_MODEL } from '../../src/core/ai/defaults.ts';
import { LEGACY_EMBEDDING_CONFIG } from '../helpers/legacy-embedding-config.ts';
let dir: string;
beforeEach(() => { dir = mkdtempSync(join(tmpdir(), 'snapshot-builder-test-')); });
afterEach(() => { rmSync(dir, { recursive: true, force: true }); });
const bytes = (value: string) => new TextEncoder().encode(value);
const quiet = () => {};
const owner = (pid: number, token: string) => ({ ...snapshotLockIdentity(), pid, token, protocol: 1 });
async function departedPid() {
const child = Bun.spawn([process.execPath, '-e', ''], { stdout: 'ignore', stderr: 'ignore' });
await child.exited;
return child.pid;
}
describe('snapshot profiles and freshness', () => {
test('legacy stays the default; invalid arguments are rejected', () => {
expect(parseSnapshotProfile([])).toBe('legacy');
expect(parseSnapshotProfile(['--profile', 'default'])).toBe('default');
expect(parseSnapshotProfile(['--profile', 'legacy'])).toBe('legacy');
for (const args of [['--profile'], ['--profile', 'other'], ['--unknown'], ['--profile', 'default', 'extra']]) {
expect(() => parseSnapshotProfile(args)).toThrow('Usage:');
}
});
test('profiles use canonical shapes and independent artifacts, locks, and freshness', async () => {
const legacy = snapshotProfile('legacy', dir);
const shipped = snapshotProfile('default', dir);
expect(legacy.shape).toEqual(LEGACY_EMBEDDING_CONFIG);
expect(shipped.shape).toEqual({ embedding_model: DEFAULT_EMBEDDING_MODEL, embedding_dimensions: DEFAULT_EMBEDDING_DIMENSIONS });
expect(shipped.tar).not.toBe(legacy.tar);
expect(shipped.version).not.toBe(legacy.version);
expect(shipped.lock).not.toBe(legacy.lock);
let builds = 0;
const buildData = async () => bytes(`complete-${++builds}`);
const opts = { fixtureDir: dir, buildData, log: quiet };
expect(await buildPgliteSnapshot('legacy', opts)).toBe('built');
expect(await buildPgliteSnapshot('default', opts)).toBe('built');
expect(await buildPgliteSnapshot('legacy', opts)).toBe('fresh');
expect(await buildPgliteSnapshot('default', opts)).toBe('fresh');
expect(builds).toBe(2);
const version = readFileSync(shipped.version, 'utf8');
expect(version).toContain(`dims=${DEFAULT_EMBEDDING_DIMENSIONS}\nmodel=${DEFAULT_EMBEDDING_MODEL}\n`);
writeFileSync(shipped.version, version.replace(/^dims=\d+$/m, 'dims=99999'));
expect(await buildPgliteSnapshot('default', opts)).toBe('built');
expect(readFileSync(shipped.tar, 'utf8')).toBe('complete-3');
expect(readFileSync(legacy.tar, 'utf8')).toBe('complete-1');
expect(readdirSync(dir).some(name => name.endsWith('.tmp') || name.endsWith('.lock'))).toBe(false);
});
});
describe('snapshot lock ownership and atomic publication', () => {
test.each([false, true])('two subprocess builders publish one complete artifact (departed owner: %s)', async (departed) => {
const paths = snapshotProfile('default', dir);
if (departed) {
mkdirSync(paths.lock);
writeFileSync(join(paths.lock, 'owner.json'), JSON.stringify(owner(await departedPid(), 'departed-before-two-builders')));
}
const code = `
import { appendFileSync } from 'node:fs';
import { join } from 'node:path';
import { buildPgliteSnapshot } from './scripts/build-pglite-snapshot.ts';
const dir = process.argv[1];
const result = await buildPgliteSnapshot('default', { fixtureDir: dir, log: () => {}, buildData: async () => {
appendFileSync(join(dir, 'builds'), 'build\\n');
await Bun.sleep(100);
return new TextEncoder().encode('complete artifact');
}});
console.log(result);
`;
const children = Array.from({ length: 2 }, () => Bun.spawn([process.execPath, '-e', code, dir], {
cwd: resolve(import.meta.dir, '../..'), stdout: 'pipe', stderr: 'pipe',
}));
const results = await Promise.all(children.map(async child => {
const [out, err, code] = await Promise.all([new Response(child.stdout).text(), new Response(child.stderr).text(), child.exited]);
expect(err).toBe('');
expect(code).toBe(0);
return out.trim();
}));
expect(results.sort()).toEqual(['built', 'fresh']);
expect(readFileSync(join(dir, 'builds'), 'utf8')).toBe('build\n');
expect(readFileSync(snapshotProfile('default', dir).tar, 'utf8')).toBe('complete artifact');
expect(existsSync(snapshotProfile('default', dir).lock)).toBe(false);
});
test('a live owner times out without building, deleting its lock, or changing existing files', async () => {
const paths = snapshotProfile('default', dir);
mkdirSync(paths.lock);
const record = JSON.stringify(owner(process.pid, 'live-owner'));
writeFileSync(join(paths.lock, 'owner.json'), record);
writeFileSync(paths.tar, 'old tar');
writeFileSync(paths.version, 'old version');
let built = false;
await expect(buildPgliteSnapshot('default', {
fixtureDir: dir, lockTimeoutMs: 0, log: quiet,
buildData: async () => { built = true; return bytes('wrong'); },
})).rejects.toThrow('refusing to build without ownership');
expect(built).toBe(false);
expect(readFileSync(join(paths.lock, 'owner.json'), 'utf8')).toBe(record);
expect(readFileSync(paths.tar, 'utf8')).toBe('old tar');
expect(readFileSync(paths.version, 'utf8')).toBe('old version');
});
test('a departed owner is reclaimed, while an ownerless lock fails closed', async () => {
const paths = snapshotProfile('legacy', dir);
mkdirSync(paths.lock);
const record = JSON.stringify(owner(await departedPid(), '../../dead-owner/with-unsafe-path-characters'));
writeFileSync(join(paths.lock, 'owner.json'), record);
// An orphaned guard from the old implementation cannot wedge recovery.
mkdirSync(`${paths.lock}.reclaim`);
expect(await buildPgliteSnapshot('legacy', { fixtureDir: dir, log: quiet, buildData: async () => bytes('recovered') })).toBe('built');
expect(existsSync(paths.lock)).toBe(false);
const tombstones = readdirSync(dir).filter(name => /\.dead-[0-9a-f]{64}$/.test(name));
expect(tombstones).toHaveLength(2);
expect(tombstones.map(name => readFileSync(join(dir, name, 'owner.json'), 'utf8'))).toContain(record);
const other = snapshotProfile('default', dir);
mkdirSync(other.lock);
await expect(buildPgliteSnapshot('default', { fixtureDir: dir, lockTimeoutMs: 0, log: quiet })).rejects.toThrow('lock timeout');
expect(existsSync(other.lock)).toBe(true);
});
test('two stale reapers cannot rename a replacement live lock', async () => {
const paths = snapshotProfile('default', dir);
mkdirSync(paths.lock);
const departed = owner(await departedPid(), 'departed-observed-twice');
writeFileSync(join(paths.lock, 'owner.json'), JSON.stringify(departed));
// Both reapers read the departed owner before either acts. The first
// wins; a new live builder acquires before the second observer resumes.
const first = JSON.parse(readFileSync(join(paths.lock, 'owner.json'), 'utf8'));
const second = JSON.parse(readFileSync(join(paths.lock, 'owner.json'), 'utf8'));
expect(reclaimDeadSnapshotOwner(paths.lock, first)).toBe(true);
mkdirSync(paths.lock);
const replacement = JSON.stringify(owner(process.pid, 'replacement-live-owner'));
writeFileSync(join(paths.lock, 'owner.json'), replacement);
expect(reclaimDeadSnapshotOwner(paths.lock, second)).toBe(false);
expect(readFileSync(join(paths.lock, 'owner.json'), 'utf8')).toBe(replacement);
expect(readdirSync(dir).filter(name => name.includes('.dead-'))).toHaveLength(1);
});
test.each([false, true])('an observer delayed past normal release cannot reclaim the next owner (build failed: %s)', async (failed) => {
const paths = snapshotProfile('default', dir);
const observedPath = join(dir, 'observed.json');
const releasePath = join(dir, 'release');
const code = `
import { existsSync, readFileSync, writeFileSync } from 'node:fs';
import { join } from 'node:path';
import { buildPgliteSnapshot, snapshotProfile } from './scripts/build-pglite-snapshot.ts';
const dir = process.argv[1];
try {
await buildPgliteSnapshot('default', { fixtureDir: dir, log: () => {}, buildData: async () => {
writeFileSync(join(dir, 'observed.json'), readFileSync(join(snapshotProfile('default', dir).lock, 'owner.json')));
while (!existsSync(join(dir, 'release'))) await Bun.sleep(5);
if (process.argv[2] === 'true') throw new Error('expected build failure');
return new TextEncoder().encode('complete artifact');
}});
} catch { process.exit(1); }
`;
const child = Bun.spawn([process.execPath, '-e', code, dir, String(failed)], {
cwd: resolve(import.meta.dir, '../..'), stdout: 'ignore', stderr: 'ignore',
});
try {
const deadline = Date.now() + 5000;
while (!existsSync(observedPath) && Date.now() < deadline) await Bun.sleep(5);
const observed = JSON.parse(readFileSync(observedPath, 'utf8'));
writeFileSync(releasePath, 'continue');
expect(await child.exited).toBe(failed ? 1 : 0);
expect(existsSync(paths.lock)).toBe(false);
mkdirSync(paths.lock);
const replacement = JSON.stringify(owner(process.pid, 'live-after-normal-release'));
writeFileSync(join(paths.lock, 'owner.json'), replacement);
expect(reclaimDeadSnapshotOwner(paths.lock, observed)).toBe(false);
expect(readFileSync(join(paths.lock, 'owner.json'), 'utf8')).toBe(replacement);
} finally {
child.kill();
await child.exited;
}
});
test('foreign and missing process identities cannot be reclaimed even when their PID is absent locally', async () => {
const pid = await departedPid();
const local = owner(pid, 'foreign-owner');
const identities = [
{ ...local, platform: 'other-platform' },
{ ...local, hostname: `${local.hostname}-other-host` },
{ ...local, pidNamespace: `${local.pidNamespace}-other-namespace` },
{ ...local, pid: process.pid, pidNamespace: `${local.pidNamespace}-live-foreign-namespace` },
{ ...local, protocol: undefined },
{ pid, token: 'legacy-record-without-identity' },
];
for (let index = 0; index < identities.length; index++) {
const fixtureDir = join(dir, String(index));
const paths = snapshotProfile('default', fixtureDir);
mkdirSync(paths.lock, { recursive: true });
const record = JSON.stringify(identities[index]);
writeFileSync(join(paths.lock, 'owner.json'), record);
let builds = 0;
await expect(buildPgliteSnapshot('default', {
fixtureDir, lockTimeoutMs: 0, log: quiet,
buildData: async () => { builds++; return bytes('must not build'); },
})).rejects.toThrow('lock timeout');
expect(builds).toBe(0);
expect(readFileSync(join(paths.lock, 'owner.json'), 'utf8')).toBe(record);
expect(readdirSync(fixtureDir).some(name => name.includes('.dead-'))).toBe(false);
}
});
test('failed byte generation leaves existing artifacts unchanged and releases ownership', async () => {
const paths = snapshotProfile('legacy', dir);
writeFileSync(paths.tar, 'old tar');
writeFileSync(paths.version, 'old version');
await expect(buildPgliteSnapshot('legacy', {
fixtureDir: dir, log: quiet, buildData: async () => { throw new Error('simulated build failure'); },
})).rejects.toThrow('simulated build failure');
expect(readFileSync(paths.tar, 'utf8')).toBe('old tar');
expect(readFileSync(paths.version, 'utf8')).toBe('old version');
expect(existsSync(paths.lock)).toBe(false);
expect(readdirSync(dir).some(name => name.endsWith('.tmp'))).toBe(false);
});
test('a process killed between publication renames leaves a stale version and can be recovered', async () => {
const paths = snapshotProfile('default', dir);
writeFileSync(paths.tar, 'old tar');
writeFileSync(paths.version, 'old version');
// Fault injection lives only in this subprocess. The real first rename
// completes, then SIGKILL prevents the second rename and all finally code.
const fixture = resolve(import.meta.dir, '../fixtures/snapshot-publication-crash.ts');
const child = Bun.spawn([process.execPath, fixture, dir, paths.tar], {
cwd: resolve(import.meta.dir, '../..'), stdout: 'pipe', stderr: 'pipe',
});
const [exitCode, stderr] = await Promise.all([child.exited, new Response(child.stderr).text()]);
expect(stderr).toBe('');
expect(exitCode).not.toBe(0);
expect(child.signalCode).toBe('SIGKILL');
expect(readFileSync(paths.tar, 'utf8')).toBe('complete new tar');
expect(readFileSync(paths.version, 'utf8')).toBe('old version');
expect(existsSync(paths.lock)).toBe(true);
let builds = 0;
expect(await buildPgliteSnapshot('default', { fixtureDir: dir, log: quiet, buildData: async () => {
builds++;
return bytes('recovered tar');
}})).toBe('built');
expect(builds).toBe(1);
expect(readFileSync(paths.tar, 'utf8')).toBe('recovered tar');
expect(readFileSync(paths.version, 'utf8')).toContain(`dims=${DEFAULT_EMBEDDING_DIMENSIONS}\n`);
expect(existsSync(paths.lock)).toBe(false);
});
test('lost ownership refuses publication and leaves the replacement owner intact', async () => {
const paths = snapshotProfile('legacy', dir);
const replacement = JSON.stringify(owner(process.pid, 'replacement-owner'));
await expect(buildPgliteSnapshot('legacy', {
fixtureDir: dir, log: quiet, buildData: async () => {
writeFileSync(join(paths.lock, 'owner.json'), replacement);
return bytes('must not publish');
},
})).rejects.toThrow('ownership lost');
expect(existsSync(paths.tar)).toBe(false);
expect(existsSync(paths.version)).toBe(false);
expect(readFileSync(join(paths.lock, 'owner.json'), 'utf8')).toBe(replacement);
expect(readdirSync(dir).some(name => name.endsWith('.tmp'))).toBe(false);
});
});
describe('snapshot shell helper contracts', () => {
function run(body: string, env: NodeJS.ProcessEnv = {}) {
return Bun.spawnSync(['bash', '-c', '. "$1"\n' + body, 'snapshot-test', resolve(import.meta.dir, '../../scripts/lib/test-env.sh')], {
cwd: dir,
env: { ...process.env, GBRAIN_NO_SNAPSHOT: '', GBRAIN_TEST_DEFAULT_SNAPSHOT: '', GBRAIN_PGLITE_SNAPSHOT: 'legacy.tar', ...env },
stdout: 'pipe', stderr: 'pipe',
});
}
test('default build preserves the legacy parent and publishes an absolute child path', () => {
const result = run('bun() { [ "$*" = "run build:pglite-snapshot --profile default" ]; }\nensure_default_pglite_snapshot test\nprintf "%s\\n%s\\n" "$GBRAIN_PGLITE_SNAPSHOT" "$GBRAIN_TEST_DEFAULT_SNAPSHOT"');
expect(result.exitCode).toBe(0);
expect(result.stdout.toString().trim().split('\n')).toEqual(['legacy.tar', join(realpathSync(dir), 'test/fixtures/pglite-snapshot-default.tar')]);
});
test('build failure is nonfatal and leaves no default snapshot activated', () => {
const result = run('bun() { return 1; }\nensure_default_pglite_snapshot test\nprintf "%s\\n" "${GBRAIN_TEST_DEFAULT_SNAPSHOT-unset}"');
expect(result.exitCode).toBe(0);
expect(result.stdout.toString().trim()).toBe('unset');
expect(result.stderr.toString()).toContain('non-fatal');
});
test.each(['ensure_pglite_snapshot', 'ensure_default_pglite_snapshot'])('%s clears both inherited paths under the cold opt-out', (helper) => {
const result = run(`${helper} test\nprintf "%s\\n%s\\n" "\${GBRAIN_PGLITE_SNAPSHOT-unset}" "\${GBRAIN_TEST_DEFAULT_SNAPSHOT-unset}"`, {
GBRAIN_NO_SNAPSHOT: '1', GBRAIN_TEST_DEFAULT_SNAPSHOT: 'default.tar',
});
expect(result.exitCode).toBe(0);
expect(result.stdout.toString().trim().split('\n')).toEqual(['unset', 'unset']);
});
});

View File

@@ -0,0 +1,157 @@
import { afterEach, describe, expect, it } from 'bun:test';
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join, resolve } from 'node:path';
import { mineWeights } from '../../scripts/mine-shard-weights.ts';
import { captureTestLog } from '../../scripts/capture-test-log.ts';
const SCRIPT = resolve(import.meta.dir, '../../scripts/capture-test-log.ts');
const roots: string[] = [];
function fixture(source: string) {
const root = mkdtempSync(join(tmpdir(), 'gbrain-capture-log-'));
roots.push(root);
const script = join(root, 'fixture.ts');
const output = join(root, 'timings', 'execution.log');
writeFileSync(script, source);
return { root, script, output };
}
afterEach(() => {
for (const root of roots.splice(0)) rmSync(root, { recursive: true, force: true });
});
async function run(source: string, args: string[] = [], job = 'test (1)') {
const f = fixture(source);
const proc = Bun.spawn([process.execPath, SCRIPT, '--job', job, '--out', f.output, '--', process.execPath, f.script, ...args], {
stdout: 'pipe', stderr: 'pipe',
});
const [stdout, stderr, code] = await Promise.all([
new Response(proc.stdout).text(), new Response(proc.stderr).text(), proc.exited,
]);
return { ...f, stdout, stderr, code, artifact: readFileSync(f.output, 'utf8') };
}
describe('timestamped test log capture', () => {
it('preserves both live streams, group markers and final unterminated lines', async () => {
const r = await run(`
process.stdout.write('##[group]test/fixture.test.ts:\\n\\n');
process.stderr.write('diagnostic\\r\\n');
await Bun.sleep(15);
process.stdout.write('stdout tail');
process.stderr.write('stderr tail');
`);
expect(r.code).toBe(0);
expect(r.stdout).toBe('##[group]test/fixture.test.ts:\n\nstdout tail');
expect(r.stderr).toBe('diagnostic\r\nstderr tail');
const messages: string[] = [];
const timestamps: number[] = [];
for (const line of r.artifact.trimEnd().split('\n')) {
const match = /^test \(1\)\tcapture\t(\S+Z) (.*)$/.exec(line);
expect(match).not.toBeNull();
const timestamp = Date.parse(match![1]!);
expect(Number.isFinite(timestamp)).toBe(true);
timestamps.push(timestamp);
messages.push(match![2]!);
}
expect(messages[0]).toBe('##[gbrain-capture-start]');
expect(messages.at(-1)).toBe('##[gbrain-capture-complete] exit=0');
expect(messages.slice(1, -1).sort()).toEqual(['##[group]test/fixture.test.ts:', '', 'diagnostic', 'stdout tail', 'stderr tail'].sort());
expect(timestamps).toEqual([...timestamps].sort((a, b) => a - b));
});
it('can feed captured unit and E2E records directly to the weight miner', async () => {
const unit = await run(`
console.log('##[group]test/fixture.test.ts:');
await Bun.sleep(20);
console.log(' 0 fail');
console.log('Ran 1 test across 1 file. [20ms]');
`);
expect(mineWeights(unit.artifact, 'unit').get('test/fixture.test.ts')).toBeGreaterThan(0);
expect(() => mineWeights(unit.artifact.slice(0, unit.artifact.lastIndexOf('test (1)\tcapture')), 'unit')).toThrow();
const e2e = await run(`
console.log('=== fixture.e2e.test.ts ===');
console.log(' 0 fail');
console.log('Ran 1 test across 1 file. [125ms]');
console.log('Files: 1 total, 1 passed, 0 failed');
`, [], 'Selected E2E (diff-relevant) (2)');
expect([...mineWeights(e2e.artifact, 'e2e')]).toEqual([['test/e2e/fixture.e2e.test.ts', 125]]);
});
it('preserves exit codes and passes argv literally without shell interpretation', async () => {
const args = ['has spaces', '"quotes"', "'single'", '$(echo injected)', '`echo injected`', '; false', '*'];
const r = await run(`console.log(JSON.stringify(process.argv.slice(2))); process.exitCode = 37;`, args);
expect(r.code).toBe(37);
expect(JSON.parse(r.stdout)).toEqual(args);
expect(r.artifact).toContain(JSON.stringify(args));
expect(r.artifact).toContain('##[error]captured command exited 37');
expect(r.stderr).toBe('');
});
it('rejects invalid metadata before running a command and failed outer-runner artifacts before mining', async () => {
const f = fixture('');
await expect(captureTestLog('test\t(1)', f.output, [process.execPath, f.script])).rejects.toThrow('one TSV field');
expect(existsSync(f.output)).toBe(false);
const r = await run(`
console.log('##[group]test/fixture.test.ts:');
console.log(' 0 fail');
console.log('Ran 1 test across 1 file. [20ms]');
process.exitCode = 7;
`);
expect(r.code).toBe(7);
expect(() => mineWeights(r.artifact, 'unit')).toThrow('failed job log');
});
it('bounds long artifact lines while preserving all live bytes', async () => {
const length = 256 * 1024 + 17;
const r = await run(`process.stdout.write('x'.repeat(${length}));`);
expect(r.code).toBe(0);
expect(r.stdout).toBe('x'.repeat(length));
const messages = r.artifact.trimEnd().split('\n').slice(1, -1).map(line => line.split(/\t\S+Z /)[1]!);
expect(messages.join('')).toBe(r.stdout);
expect(messages.every(message => message.length <= 64 * 1024)).toBe(true);
});
it('reports a missing command and preserves a child signal exit code', async () => {
const f = fixture('');
const missing = Bun.spawn([process.execPath, SCRIPT, '--job', 'test (1)', '--out', f.output, '--', join(f.root, 'missing-command')], {
stdout: 'pipe', stderr: 'pipe',
});
expect(await missing.exited).toBe(2);
expect(await new Response(missing.stderr).text()).toContain('capture-test-log:');
const signalled = await run(`process.kill(process.pid, 'SIGTERM');`);
expect(signalled.code).toBe(143);
});
for (const signal of ['SIGTERM', 'SIGINT'] as const) {
it(`forwards ${signal} to its owned child and grandchild`, async () => {
const f = fixture(`
const child = Bun.spawn([process.execPath, '-e', 'setInterval(() => {}, 1000)'], { stdout: 'ignore', stderr: 'ignore' });
await Bun.write(process.argv[2], JSON.stringify([process.pid, child.pid]));
setInterval(() => {}, 1000);
`);
const pidsFile = join(f.root, 'pids.json');
const proc = Bun.spawn([process.execPath, SCRIPT, '--job', 'test (1)', '--out', f.output, '--', process.execPath, f.script, pidsFile], {
stdout: 'ignore', stderr: 'ignore',
});
let pids: number[] = [];
const alive = (pid: number) => {
try {
if (process.platform === 'linux' && /\) Z /.test(readFileSync(`/proc/${pid}/stat`, 'utf8'))) return false;
process.kill(pid, 0);
return true;
} catch { return false; }
};
try {
for (let i = 0; i < 100 && !existsSync(pidsFile); i++) await Bun.sleep(20);
expect(existsSync(pidsFile)).toBe(true);
pids = JSON.parse(readFileSync(pidsFile, 'utf8'));
expect(pids.every(alive)).toBe(true);
proc.kill(signal);
expect(await proc.exited).toBe(signal === 'SIGINT' ? 130 : 143);
for (let i = 0; i < 100 && pids.some(alive); i++) await Bun.sleep(20);
expect(pids.some(alive)).toBe(false);
} finally {
proc.kill('SIGKILL');
for (const pid of pids) if (alive(pid)) try { process.kill(pid, 'SIGKILL'); } catch { /* exited */ }
}
}, 10000);
}
});

View File

@@ -55,7 +55,7 @@ describe('CI execution evidence', () => {
expect(e2e.jobs['e2e-status'].if).toBe('always()');
const needs = e2e.jobs['e2e-status'].needs as string[];
const nightly = ['coverage-full-unit', 'coverage-full-serial', 'coverage-full-slow', 'coverage-full-e2e'];
expect(needs).toEqual(['jsonb-parity', 'tier1', 'tier2', 'selected-e2e', ...nightly]);
expect(needs).toEqual(['jsonb-parity', 'tier1', 'tier2', 'prepare-e2e', 'selected-e2e', ...nightly]);
for (const event of ['pull_request', 'push', 'workflow_dispatch']) {
expect(aggregate(e2e, 'e2e-status', event, Object.fromEntries(nightly.map(job => [job, 'skipped'])))).toBe(0);
}

View File

@@ -0,0 +1,98 @@
import { describe, test, expect } from 'bun:test';
import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync, symlinkSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { spawnSync } from 'node:child_process';
import { E2E_EXCLUSIONS, prepareMatrix, validateRow } from '../../scripts/e2e-matrix.ts';
const repo = join(import.meta.dir, '../..');
const paths = ['a', 'b', 'c', 'd', 'e'].map(n => `test/e2e/${n}.test.ts`);
const weights = new Map(paths.map((f, i) => [f, 100 - i * 10]));
function fixture(fn: (root: string) => void) {
const root = mkdtempSync(join(tmpdir(), 'gbrain-e2e-matrix-'));
try {
mkdirSync(join(root, 'test/e2e'), { recursive: true });
mkdirSync(join(root, 'scripts'));
for (const file of paths) writeFileSync(join(root, file), '// fixture');
fn(root);
} finally { rmSync(root, { recursive: true, force: true }); }
}
function worker(root: string, row: unknown) {
return spawnSync(process.execPath, [join(repo, 'scripts/e2e-matrix.ts'), 'run'], {
cwd: root, encoding: 'utf8', env: { ...process.env, SHARD: '2/4', E2E_MATRIX_ROW: JSON.stringify(row) },
});
}
describe('frozen E2E matrix', () => {
test('partitions the exact selection once, with deterministic weights and at most four workers', () => {
const matrix = prepareMatrix(paths, weights);
expect(matrix.include).toHaveLength(4);
expect(matrix.include.flatMap(r => r.files).sort()).toEqual(paths);
expect(prepareMatrix(paths, weights)).toEqual(matrix);
expect(prepareMatrix(paths.slice(0, 2), weights).include).toHaveLength(2);
});
test('filters only the existing named/live exclusions and emits an explicit empty sentinel', () => {
expect(prepareMatrix([...E2E_EXCLUSIONS, paths[0]], weights).include).toEqual([{ shard: 1, files: [paths[0]], empty: false }]);
expect(prepareMatrix([], weights).include).toEqual([{ shard: 1, files: [], empty: true }]);
expect(prepareMatrix([...E2E_EXCLUSIONS], weights).include[0].empty).toBe(true);
});
test('refuses invalid selections and duplicate paths', () => {
for (const files of [[paths[0], paths[0]], ['../outside.test.ts'], ['test/e2e/../escape.test.ts'], ['test/e2e/$(touch marker).test.ts']]) expect(() => prepareMatrix(files, weights)).toThrow();
});
test('empty sentinel launches no runner; missing/malformed input fails instead of selecting all', () => fixture(root => {
writeFileSync(join(root, 'scripts/run-e2e.sh'), 'exit 99\n');
expect(worker(root, { shard: 1, files: [], empty: true }).status).toBe(0);
for (const row of [null, {}, { shard: 1, files: [], empty: false }, { shard: 2, files: [], empty: true }, { shard: 5, files: [paths[0]], empty: false }]) expect(worker(root, row).status).not.toBe(0);
}));
test('worker executes the complete frozen argv and clears inherited SHARD', () => fixture(root => {
writeFileSync(join(root, 'scripts/run-e2e.sh'), 'test -z "${SHARD:-}" || exit 81\nprintf "FILE:%s\\n" "$@"\n');
const r = worker(root, { shard: 2, files: paths, empty: false });
expect(r.status, r.stderr).toBe(0);
expect(r.stdout.split('\n').filter(l => l.startsWith('FILE:'))).toEqual(paths.map(p => `FILE:${p}`));
}));
test('worker rejects missing, excluded, duplicate and symlink-escaping files', () => fixture(root => {
writeFileSync(join(root, 'outside.test.ts'), '// outside');
symlinkSync(join(root, 'outside.test.ts'), join(root, 'test/e2e/escape.test.ts'));
for (const files of [['test/e2e/missing.test.ts'], [paths[0], paths[0]], ['test/e2e/escape.test.ts'], ['test/e2e/mechanical.test.ts']]) expect(() => validateRow({ shard: 1, files, empty: false }, root)).toThrow();
}));
test('propagates runner failure without marking an executed partition successful', () => fixture(root => {
writeFileSync(join(root, 'scripts/run-e2e.sh'), 'exit 17\n');
expect(worker(root, { shard: 1, files: [paths[0]], empty: false }).status).toBe(17);
}));
test.each(['SIGTERM', 'SIGINT'] as const)('forwards %s cancellation to its owned runner and preserves its exit status', async (signal) => {
const root = mkdtempSync(join(tmpdir(), 'gbrain-e2e-matrix-cancel-'));
let child: ReturnType<typeof Bun.spawn> | undefined;
let runnerPid = 0;
try {
mkdirSync(join(root, 'scripts'));
mkdirSync(join(root, 'test/e2e'), { recursive: true });
writeFileSync(join(root, paths[0]), '// frozen fixture');
// Trap the actual signal in a portable shell child. Its completion marker
// proves the wrapper forwarded cancellation instead of merely dying.
writeFileSync(join(root, 'scripts/run-e2e.sh'), `trap 'printf "SIGTERM\\n" > received.txt; exit 143' TERM
trap 'printf "SIGINT\\n" > received.txt; exit 130' INT
printf '%s\\n' "$$" > runner.pid
while :; do sleep 0.05; done
`);
child = Bun.spawn([process.execPath, join(repo, 'scripts/e2e-matrix.ts'), 'run'], {
cwd: root,
env: { ...process.env, SHARD: '4/4', E2E_MATRIX_ROW: JSON.stringify({ shard: 1, files: [paths[0]], empty: false }) },
stdout: 'ignore', stderr: 'ignore',
});
const readyDeadline = Date.now() + 5000;
while (!existsSync(join(root, 'runner.pid')) && Date.now() < readyDeadline) await Bun.sleep(20);
expect(existsSync(join(root, 'runner.pid'))).toBe(true);
runnerPid = Number(readFileSync(join(root, 'runner.pid'), 'utf8'));
child.kill(signal);
const exitDeadline = Date.now() + 5000;
while (child.exitCode === null && Date.now() < exitDeadline) await Bun.sleep(20);
expect(child.exitCode).toBe(signal === 'SIGINT' ? 130 : 143);
expect(readFileSync(join(root, 'received.txt'), 'utf8').trim()).toBe(signal);
expect(() => process.kill(runnerPid, 0)).toThrow();
} finally {
child?.kill('SIGKILL');
if (runnerPid) { try { process.kill(runnerPid, 'SIGKILL'); } catch { /* already reaped */ } }
if (child) await child.exited;
rmSync(root, { recursive: true, force: true });
}
}, 15000);
});

View File

@@ -0,0 +1,69 @@
import { describe, test, expect } from 'bun:test';
import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync, copyFileSync, chmodSync, existsSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { spawnSync } from 'node:child_process';
const repo = join(import.meta.dir, '../..');
function setup() {
const root = mkdtempSync(join(tmpdir(), 'gbrain-e2e-runner-'));
for (const dir of ['scripts/lib', 'test/e2e', 'bin']) mkdirSync(join(root, dir), { recursive: true });
for (const file of ['run-e2e.sh', 'sharding.ts', 'lib/test-env.sh']) copyFileSync(join(repo, 'scripts', file), join(root, 'scripts', file));
for (const name of ['a', 'b']) writeFileSync(join(root, `test/e2e/${name}.test.ts`), `import {test,expect} from 'bun:test'; test('isolated routing',()=>{ expect(process.env.SHARD).toBeUndefined(); expect(process.env.COVERAGE_DIR ?? '').toBe(''); });`);
return root;
}
const env = { ...process.env, GBRAIN_NO_SNAPSHOT: '1', DATABASE_URL: '', GBRAIN_DATABASE_URL: '', SHARD: '', COVERAGE_DIR: '' };
function run(root: string, args: string[], shard = '') {
return spawnSync('bash', ['scripts/run-e2e.sh', ...args], { cwd: root, encoding: 'utf8', env: { ...env, SHARD: shard } });
}
describe('sequential E2E runner', () => {
test('weighted shards cover exactly the explicit input; empty shards launch nothing', () => {
const root = setup();
try {
const files = ['test/e2e/a.test.ts', 'test/e2e/b.test.ts'];
const selected = [1, 2, 3].flatMap(n => {
const r = run(root, ['--dry-run-list', ...files], `${n}/3`);
expect(r.status, r.stderr).toBe(0);
return r.stdout.trim().split('\n').filter(Boolean);
});
expect(selected.sort()).toEqual(files);
expect(run(root, files, '3/3').stdout).toContain('No files for shard 3/3');
for (const bad of ['2', '0/2', '3/2', '1/0', '1/2x', '1/2/3']) expect(run(root, files, bad).status).not.toBe(0);
} finally { rmSync(root, { recursive: true, force: true }); }
});
test('runs each file and preserves an assertion failure in the final status', () => {
const root = setup();
try {
const files = ['test/e2e/a.test.ts', 'test/e2e/b.test.ts'];
const good = run(root, files, '1/1');
expect(good.status, good.stderr + good.stdout).toBe(0);
expect(good.stdout).toContain('Files: 2 total, 2 passed, 0 failed');
writeFileSync(join(root, files[0]), "import {test,expect} from 'bun:test';test('failure',()=>expect(1).toBe(2));");
const bad = run(root, files);
expect(bad.status).toBe(1);
expect(bad.stdout).toContain('Files: 2 total, 1 passed, 1 failed');
} finally { rmSync(root, { recursive: true, force: true }); }
});
test('cancellation terminates its owned interrupt-resistant child', async () => {
const root = setup();
let child: ReturnType<typeof Bun.spawn> | undefined;
let pid: number | undefined;
try {
const fake = join(root, 'bin/bun');
writeFileSync(fake, `#!/usr/bin/env bash\necho $$ > '${join(root, 'child.pid')}'\ntrap '' INT TERM\nwhile true; do sleep 1; done\n`);
chmodSync(fake, 0o755);
child = Bun.spawn(['bash', 'scripts/run-e2e.sh', 'test/e2e/a.test.ts'], { cwd: root, env: { ...env, PATH: `${join(root, 'bin')}:${process.env.PATH}` }, stdout: 'ignore', stderr: 'ignore' });
const deadline = Date.now() + 5000;
while (!existsSync(join(root, 'child.pid')) && Date.now() < deadline) await Bun.sleep(20);
expect(existsSync(join(root, 'child.pid'))).toBe(true);
pid = Number(readFileSync(join(root, 'child.pid'), 'utf8').trim());
child.kill('SIGTERM');
expect(await child.exited).toBe(143);
expect(() => process.kill(pid!, 0)).toThrow();
} finally {
child?.kill();
if (pid) { try { process.kill(pid, 'SIGKILL'); } catch {} }
rmSync(root, { recursive: true, force: true });
}
}, 10000);
});

View File

@@ -27,6 +27,7 @@ import { existsSync, mkdtempSync, mkdirSync, readdirSync, readFileSync, rmSync,
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { spawnSync } from 'node:child_process';
import { E2E_EXCLUSIONS, prepareMatrix } from '../../scripts/e2e-matrix.ts';
import { E2E_TEST_MAP } from '../../scripts/e2e-test-map.ts';
const repoRoot = join(import.meta.dir, '..', '..');
@@ -50,10 +51,13 @@ const LIVE_KEY_FILES = new Set([
describe('selected-e2e job wiring', () => {
const job = jobBlock('selected-e2e');
const prep = jobBlock('prepare-e2e');
test('consumes select-e2e with nothing masking its exit code', () => {
expect(job).toContain('bun scripts/select-e2e.ts');
const selectorLine = job.split('\n').find(l => l.includes('bun scripts/select-e2e.ts'))!;
expect(prep).toContain('bun scripts/select-e2e.ts');
expect(job).not.toContain('bun scripts/select-e2e.ts');
expect(job).toContain('bun scripts/e2e-matrix.ts run');
const selectorLine = prep.split('\n').find(l => l.includes('bun scripts/select-e2e.ts'))!;
expect(selectorLine).not.toContain('|| true');
expect(selectorLine).not.toContain('|| echo');
expect(job).not.toContain('continue-on-error');
@@ -74,7 +78,7 @@ describe('selected-e2e job wiring', () => {
});
test('checkout fetches real history for the master diff', () => {
expect(job).toContain('fetch-depth: 0');
expect(prep).toContain('fetch-depth: 0');
});
test('selector emits separate lines so workflow exclusions retain other selected files', () => {
@@ -93,26 +97,15 @@ describe('selected-e2e job wiring', () => {
const selected = spawnSync(process.execPath, ['--no-env-file', join(repoRoot, 'scripts/select-e2e.ts')], { cwd: dir, encoding: 'utf8' });
expect(selected.status, selected.stderr).toBe(0);
expect(selected.stdout).toBe(files.join('\n') + '\n');
writeFileSync(join(dir, 'selected.txt'), selected.stdout);
// Execute the workflow's actual exclusion loop, with only its temporary
// paths relocated so concurrent tests never share /tmp/run.txt.
const start = job.indexOf("EXCLUDE='");
const end = job.indexOf('done < /tmp/selected.txt', start) + 'done < /tmp/selected.txt'.length;
const filter = job.slice(start, end)
.replaceAll('/tmp/selected.txt', `'${join(dir, 'selected.txt')}'`)
.replaceAll('/tmp/run.txt', `'${join(dir, 'run.txt')}'`);
const filtered = spawnSync('bash', ['-e', '-c', filter], { encoding: 'utf8' });
expect(filtered.status, filtered.stderr).toBe(0);
expect(readFileSync(join(dir, 'run.txt'), 'utf8')).toBe(files.slice(1).join('\n') + '\n');
const filtered = prepareMatrix(selected.stdout.trim().split('\n'), new Map());
expect(filtered.include.flatMap(row => row.files).sort()).toEqual(files.slice(1));
} finally {
rmSync(dir, { recursive: true, force: true });
}
});
test('every EXCLUDE entry is named by another job here or is a live-key spender', () => {
const m = job.match(/EXCLUDE='([^']+)'/);
expect(m).not.toBeNull();
const excluded = m![1].split(/\s+/).filter(Boolean);
const excluded = [...E2E_EXCLUSIONS];
expect(excluded.length).toBeGreaterThan(0);
const restOfWorkflow = yml.replace(job, '');
for (const f of excluded) {
@@ -140,7 +133,7 @@ describe('e2e file claim ratchet', () => {
.filter(f => f.endsWith('.test.ts'))
.map(f => `test/e2e/${f}`);
expect(files.length).toBeGreaterThan(150);
const orphans = files.filter(f => !mapped.has(f) && !yml.includes(f) && !baseline.has(f));
const orphans = files.filter(f => !mapped.has(f) && !yml.includes(f) && !E2E_EXCLUSIONS.has(f) && !baseline.has(f));
if (orphans.length > 0) {
throw new Error(
`new e2e file(s) with no PR-time claim — map them in scripts/e2e-test-map.ts ` +
@@ -167,7 +160,7 @@ describe('e2e file claim ratchet', () => {
});
test('no redundant baseline entries: a mapped or workflow-named file must leave the baseline', () => {
const redundant = baselineEntries.filter(f => mapped.has(f) || yml.includes(f));
const redundant = baselineEntries.filter(f => mapped.has(f) || yml.includes(f) || E2E_EXCLUSIONS.has(f));
if (redundant.length > 0) {
throw new Error(
`baseline row(s) already claimed by E2E_TEST_MAP or the workflow — ` +

View File

@@ -12,6 +12,7 @@ import { basename, join } from "node:path";
import { isMergedArtifactPath, normalizeSf, parseLcovText } from "../../scripts/merge-lcov.ts";
const REPO_ROOT = join(import.meta.dir, "..", "..");
const HEAD_SHA = Bun.spawnSync(['git', 'rev-parse', 'HEAD'], { cwd: REPO_ROOT }).stdout.toString().trim();
interface RunResult {
code: number;
@@ -218,7 +219,7 @@ describe("merge: malformed input handling", () => {
});
describe("merge: lane manifests", () => {
const GOOD_MANIFEST = { lane: "shard-1", sha: "deadbeef", lcovCount: 1, complete: true };
const GOOD_MANIFEST = { lane: "shard-1", sha: HEAD_SHA, lcovCount: 1, complete: true };
it("complete expected lanes → not degraded; lanes.complete lists them", () => {
const a = laneDir("man-ok", LANE_A, GOOD_MANIFEST);
@@ -264,11 +265,69 @@ describe("merge: lane manifests", () => {
});
it("non-shard lane may carry many lcov files without tripping the tripwire", () => {
const a = laneDir("man-serial", LANE_A, { lane: "serial", sha: "d", lcovCount: 12, complete: true });
const a = laneDir("man-serial", LANE_A, { lane: "serial", sha: HEAD_SHA, lcovCount: 2, complete: true });
mkdirSync(join(a, 'second'));
writeFileSync(join(a, 'second/lcov.info'), LANE_B);
const out = outPaths("man-serial");
runMerge(["--out-lcov", out.lcov, "--out-json", out.json, "--manifest-expect", "serial", a]);
expect(readSummary(out.json).degraded).toBe(false);
});
it('requires every serial shard and accepts their independent multi-file counts', () => {
const dirs = Array.from({ length: 4 }, (_, i) => {
const dir = laneDir(`serial-matrix-${i}`, LANE_A, { lane: `serial-${i + 1}`, sha: HEAD_SHA, lcovCount: 2, complete: true });
mkdirSync(join(dir, 'second'));
writeFileSync(join(dir, 'second/lcov.info'), LANE_B);
return dir;
});
const out = outPaths('serial-matrix');
const args = ['--out-lcov', out.lcov, '--out-json', out.json, '--manifest-expect', 'serial-1,serial-2,serial-3,serial-4'];
expect(runMerge([...args, ...dirs]).code).toBe(0);
expect(readSummary(out.json).degraded).toBe(false);
runMerge([...args, ...dirs.slice(0, 3)]);
expect(readSummary(out.json).degraded).toBe(true);
});
it.each([
['missing count', { lcovCount: undefined }],
['wrong count', { lcovCount: 2 }],
['negative count', { lcovCount: -1 }],
['fractional count', { lcovCount: 1.5 }],
['wrong SHA', { sha: 'other-commit' }],
['missing SHA', { sha: undefined }],
])('%s cannot report complete coverage', (name, changes) => {
const a = laneDir(`metadata-${name}`, LANE_A, { ...GOOD_MANIFEST, lane: 'serial-1', ...changes });
const out = outPaths(`metadata-${name}`);
const result = runMerge(['--out-lcov', out.lcov, '--out-json', out.json, '--manifest-expect', 'serial-1', a]);
expect(result.code).toBe(0); // advisory degradation, never a test-result verdict
expect(readSummary(out.json).degraded).toBe(true);
expect(readSummary(out.json).lanes.complete).toEqual([]);
});
it('a missing LCOV file invalidates the lane count', () => {
const a = laneDir('missing-lcov', LANE_A, { ...GOOD_MANIFEST, lane: 'serial-1' });
rmSync(join(a, 'lcov.info'));
const out = outPaths('missing-lcov');
runMerge(['--out-lcov', out.lcov, '--out-json', out.json, '--manifest-expect', 'serial-1', a]);
expect(readSummary(out.json).degraded).toBe(true);
});
it('duplicate lane identities cannot satisfy the expected lane', () => {
const a = laneDir('duplicate-a', LANE_A, GOOD_MANIFEST);
const b = laneDir('duplicate-b', LANE_B, GOOD_MANIFEST);
const out = outPaths('duplicate');
const result = runMerge(['--out-lcov', out.lcov, '--out-json', out.json, '--manifest-expect', 'shard-1', a, b]);
expect(result.stderr).toContain('duplicate manifests');
expect(readSummary(out.json).degraded).toBe(true);
expect(readSummary(out.json).lanes.complete).toEqual([]);
});
it('--sha validates historical artifacts independently of the checkout', () => {
const a = laneDir('historical', LANE_A, { ...GOOD_MANIFEST, sha: 'historical-commit' });
const out = outPaths('historical');
runMerge(['--out-lcov', out.lcov, '--out-json', out.json, '--sha', 'historical-commit', '--manifest-expect', 'shard-1', a]);
expect(readSummary(out.json).degraded).toBe(false);
});
});
describe("merge: JSON metrics scope + hand-computed totals", () => {

View File

@@ -3,8 +3,13 @@
// is integration-tested by actually running it once during T4.
import { describe, expect, it } from "bun:test";
import { chmodSync, mkdirSync, mkdtempSync, writeFileSync, readFileSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { spawnSync } from 'node:child_process';
import {
computeWeights,
mineWeights,
parseLog,
serializeWeights,
} from "../../scripts/mine-shard-weights.ts";
@@ -156,3 +161,139 @@ describe("serializeWeights", () => {
expect(parsed["test/beta.test.ts"]).toBe(250);
});
});
describe('complete CI timing extraction', () => {
const line = (job: string, seconds: number, message: string) => `${job}\tstep\t2026-05-25T11:26:${String(seconds).padStart(2, '0')}.000Z ${message}\n`;
const header = (file: string) => `##[group]${file}:`;
const unitLog = line('test (1)', 1, header('test/one.test.ts')) + line('test (1)', 3, header('evals/two.test.ts')) + line('test (1)', 4, ' 0 fail') + line('test (1)', 6, 'Ran 2 tests across 2 files. [5.00s]');
it('includes evals and the final file, while excluding buffered/dedicated jobs', () => {
const noise = line('serial-tests', 1, header('test/wrong.test.ts')) + line('slow-eval', 1, header('test/also-wrong.test.ts'));
expect(Object.fromEntries(mineWeights(unitLog + noise, 'unit'))).toEqual({ 'test/one.test.ts': 2000, 'evals/two.test.ts': 3000 });
});
it('handles a one-file job', () => {
const raw = line('test (2)', 1, header('test/one.test.ts')) + line('test (2)', 2, ' 0 fail') + line('test (2)', 4, 'Ran 1 test across 1 file. [3.00s]');
expect(mineWeights(raw, 'unit').get('test/one.test.ts')).toBe(3000);
});
it('rejects truncated, failed, mismatched and out-of-order unit input', () => {
for (const raw of [SAMPLE, unitLog.replace('0 fail', '1 fail'), unitLog.replace('across 2', 'across 3'), unitLog.replace('11:26:03', '11:26:00')]) expect(() => mineWeights(raw, 'unit')).toThrow();
});
it('refuses setup-only jobs and missing jobs promised by GitHub metadata', () => {
expect(() => mineWeights(unitLog + line('test (2)', 1, 'snapshot setup started'), 'unit')).toThrow('incomplete Bun summary');
expect(() => mineWeights(unitLog, 'unit', { expectedJobs: ['test (1)', 'test (2)'] })).toThrow('missing: test (2)');
expect(() => mineWeights(unitLog, 'unit', { expectedJobs: ['test (2)'] })).toThrow('unexpected: test (1)');
expect(mineWeights(unitLog, 'unit', { expectedJobs: ['test (1)'] }).size).toBe(2);
});
it('mines raw Bun group commands and requires the outer capture completion', () => {
// Retained capture logs contain the original workflow command. Only the
// rendered GitHub log rewrites ::group:: into ##[group].
const captured = unitLog.replaceAll('\tstep\t', '\tcapture\t').replaceAll('##[group]', '::group::');
const marker = (seconds: number, text: string) => line('test (1)', seconds, text).replace('\tstep\t', '\tcapture\t');
const start = marker(0, '##[gbrain-capture-start]');
const complete = marker(7, '##[gbrain-capture-complete] exit=0');
expect(Object.fromEntries(mineWeights(start + captured + complete, 'unit'))).toEqual({ 'test/one.test.ts': 2000, 'evals/two.test.ts': 3000 });
for (const raw of [captured, start + captured, captured + complete, start + captured + complete.replace('exit=0', 'exit=7'), start + captured + complete + start, start + start + captured + complete]) {
expect(() => mineWeights(raw, 'unit')).toThrow('capture completion');
}
// Historical GitHub step logs remain supported with authoritative --run metadata.
expect(mineWeights(unitLog, 'unit').size).toBe(2);
});
it('does not mix interleaved jobs', () => {
const raw = line('test (1)', 1, header('test/a.test.ts')) + line('test (2)', 2, header('test/b.test.ts')) + line('test (1)', 3, ' 0 fail') + line('test (2)', 4, ' 0 fail') + line('test (2)', 5, 'Ran 1 test across 1 file. [3s]') + line('test (1)', 6, 'Ran 1 test across 1 file. [5s]');
expect(Object.fromEntries(mineWeights(raw, 'unit'))).toEqual({ 'test/a.test.ts': 5000, 'test/b.test.ts': 3000 });
});
it('uses serial runner durations in seconds, not buffered timestamp deltas', () => {
const raw = line('serial-tests (1)', 50, '[serial-tests] PASS 19s test/a.serial.test.ts (2pass)') + line('serial-tests (1)', 50, '[serial-tests] PASS 2s test/b.serial.test.ts (1pass)') + line('serial-tests (1)', 51, '[serial-tests] all 2 file(s) passed in 19s (pool=4)');
expect(Object.fromEntries(mineWeights(raw, 'serial'))).toEqual({ 'test/a.serial.test.ts': 19, 'test/b.serial.test.ts': 2 });
expect(() => mineWeights(raw.replace('all 2', 'all 3'), 'serial')).toThrow();
expect(() => mineWeights(raw + line('serial-tests (2)', 1, 'snapshot setup started'), 'serial')).toThrow('incomplete serial execution');
for (const bad of ['1..2', '9'.repeat(310)]) expect(() => mineWeights(raw.replace('PASS 19s', `PASS ${bad}s`), 'serial')).toThrow('invalid duration');
});
it('reads E2E summaries as milliseconds and requires the complete lane summary', () => {
const job = 'Selected E2E (diff-relevant) 1';
const raw = line(job, 1, '=== alpha.test.ts ===') + line(job, 2, ' 0 fail') + line(job, 3, 'Ran 4 tests across 1 file. [1.51s]') + line(job, 4, 'Files: 1 total, 1 passed, 0 failed');
expect(mineWeights(raw, 'e2e').get('test/e2e/alpha.test.ts')).toBe(1510);
expect(() => mineWeights(raw.replace('0 failed', '1 failed'), 'e2e')).toThrow();
expect(() => mineWeights(raw.replace(' 0 fail', ' 1 fail'), 'e2e')).toThrow();
expect(() => mineWeights(raw + line('Selected E2E (diff-relevant) 2', 1, 'snapshot setup started'), 'e2e')).toThrow('incomplete e2e execution');
expect(mineWeights(raw + line('Selected E2E (diff-relevant) 2', 1, 'selected E2E: explicit empty selection; no tests launched'), 'e2e').size).toBe(1);
for (const bad of ['1..2', '9'.repeat(310)]) expect(() => mineWeights(raw.replace('[1.51s]', `[${bad}s]`), 'e2e')).toThrow('invalid duration');
});
it('never serializes non-finite or negative weights into a poisoned map', () => {
for (const value of [NaN, Infinity, -1]) expect(() => serializeWeights(new Map([['test/bad.test.ts', value]]))).toThrow('invalid weight');
});
});
it('a failed CLI refresh leaves the existing output untouched', () => {
const dir = mkdtempSync(join(tmpdir(), 'gbrain-mine-failure-'));
try {
const out = join(dir, 'weights');
const input = join(dir, 'truncated.log');
writeFileSync(out, '{"keep":123}\n');
writeFileSync(input, SAMPLE);
const failed = spawnSync(process.execPath, [join(import.meta.dir, '../../scripts/mine-shard-weights.ts'), '--from-file', input, '--out', out], { encoding: 'utf8' });
expect(failed.status).not.toBe(0);
expect(readFileSync(out, 'utf8')).toBe('{"keep":123}\n');
writeFileSync(input, SAMPLE + 'test (1)\tstep\t2026-05-25T11:26:56.000Z 0 fail\ntest (1)\tstep\t2026-05-25T11:26:57.000Z Ran 4 tests across 4 files. [17s]\ntest (2)\tstep\t2026-05-25T11:26:56.000Z 0 fail\ntest (2)\tstep\t2026-05-25T11:26:58.000Z Ran 2 tests across 2 files. [16s]\n');
const passed = spawnSync(process.execPath, [join(import.meta.dir, '../../scripts/mine-shard-weights.ts'), '--from-file', input, '--out', out], { encoding: 'utf8' });
expect(passed.status, passed.stderr).toBe(0);
const weights = JSON.parse(readFileSync(out, 'utf8'));
expect(weights['test/delta.test.ts']).toBe(7000);
expect(weights.keep).toBe(123);
expect(JSON.parse(readFileSync(out + '.metadata.json', 'utf8'))).toMatchObject({ measuredFiles: 6, totalFiles: 7, mergeExisting: true });
} finally { rmSync(dir, { recursive: true, force: true }); }
});
it('partial serial artifacts preserve unobserved files; malformed durations cannot replace either output', () => {
const dir = mkdtempSync(join(tmpdir(), 'gbrain-mine-serial-'));
try {
const out = join(dir, 'weights.json');
const meta = join(dir, 'weights.metadata.json');
const input = join(dir, 'serial.log');
const line = (text: string) => `serial-tests (1)\tstep\t2026-05-25T11:26:50.000Z ${text}\n`;
const raw = line('[serial-tests] PASS 7s test/a.serial.test.ts (1pass)') + line('[serial-tests] all 1 file(s) passed in 7s (pool=4)');
writeFileSync(out, '{"test/unobserved.serial.test.ts":42}\n');
writeFileSync(input, raw);
const args = [join(import.meta.dir, '../../scripts/mine-shard-weights.ts'), '--lane', 'serial', '--from-file', input, '--out', out];
expect(spawnSync(process.execPath, args, { encoding: 'utf8' }).status).toBe(0);
expect(JSON.parse(readFileSync(out, 'utf8'))).toEqual({ 'test/a.serial.test.ts': 7, 'test/unobserved.serial.test.ts': 42 });
expect(JSON.parse(readFileSync(meta, 'utf8'))).toMatchObject({ measuredFiles: 1, totalFiles: 2, mergeExisting: true });
const saved = [readFileSync(out, 'utf8'), readFileSync(meta, 'utf8')];
writeFileSync(input, raw.replace('PASS 7s', 'PASS 1..2s'));
expect(spawnSync(process.execPath, args, { encoding: 'utf8' }).status).not.toBe(0);
expect([readFileSync(out, 'utf8'), readFileSync(meta, 'utf8')]).toEqual(saved);
} finally { rmSync(dir, { recursive: true, force: true }); }
});
it('GitHub refresh verifies every eligible job before replacing the complete lane', () => {
const dir = mkdtempSync(join(tmpdir(), 'gbrain-mine-github-'));
try {
const bin = join(dir, 'bin');
mkdirSync(bin);
const fakeGh = join(bin, 'gh');
writeFileSync(fakeGh, '#!/bin/sh\ncase "$*" in\n "run view 123 --json conclusion,headSha,jobs") cat "$FIXTURE_GH_INFO" ;;\n "run view 123 --log") cat "$FIXTURE_GH_LOG" ;;\n *) exit 2 ;;\nesac\n');
chmodSync(fakeGh, 0o755);
const info = join(dir, 'run.json');
const input = join(dir, 'unit.log');
const out = join(dir, 'weights.json');
const meta = join(dir, 'weights.metadata.json');
const line = (job: string, text: string) => `${job}\tRun test shard\t2026-05-25T11:26:50.000Z ${text}\n`;
const complete = (job: string, file: string) => line(job, `##[group]${file}:`) + line(job, '0 fail') + line(job, 'Ran 1 test across 1 file. [1ms]');
const one = complete('test (1)', 'test/one.test.ts');
writeFileSync(info, JSON.stringify({ conclusion: 'success', headSha: 'fixture-commit', jobs: [{ name: 'test (1)', conclusion: 'success' }, { name: 'test (2)', conclusion: 'success' }, { name: 'verify', conclusion: 'success' }] }));
writeFileSync(out, '{"test/old.test.ts":42}\n');
writeFileSync(meta, '{"existing":true}\n');
const args = [join(import.meta.dir, '../../scripts/mine-shard-weights.ts'), '--run', '123', '--out', out];
const env = { ...process.env, PATH: `${bin}:${process.env.PATH}`, FIXTURE_GH_INFO: info, FIXTURE_GH_LOG: input };
for (const partial of [one, one + line('test (2)', 'snapshot setup started')]) {
writeFileSync(input, partial);
expect(spawnSync(process.execPath, args, { env, encoding: 'utf8' }).status).not.toBe(0);
expect(readFileSync(out, 'utf8')).toBe('{"test/old.test.ts":42}\n');
expect(readFileSync(meta, 'utf8')).toBe('{"existing":true}\n');
}
writeFileSync(input, one + complete('test (2)', 'test/two.test.ts'));
const result = spawnSync(process.execPath, args, { env, encoding: 'utf8' });
expect(result.status, result.stderr).toBe(0);
expect(JSON.parse(readFileSync(out, 'utf8'))).toEqual({ 'test/one.test.ts': 0, 'test/two.test.ts': 0 });
expect(JSON.parse(readFileSync(meta, 'utf8'))).toMatchObject({ commit: 'fixture-commit', measuredFiles: 2, totalFiles: 2, mergeExisting: false });
} finally { rmSync(dir, { recursive: true, force: true }); }
});

View File

@@ -16,7 +16,7 @@
import { describe, it, expect, beforeAll, afterAll } from 'bun:test';
import { execFileSync } from 'child_process';
import { mkdtempSync, mkdirSync, writeFileSync, rmSync, copyFileSync, chmodSync, symlinkSync } from 'fs';
import { existsSync, readFileSync, mkdtempSync, mkdirSync, writeFileSync, rmSync, copyFileSync, chmodSync, symlinkSync } from 'fs';
import { tmpdir } from 'os';
import { dirname, join, resolve } from 'path';
@@ -30,7 +30,7 @@ function stageSandbox(): string {
const root = mkdtempSync(join(tmpdir(), 'gbrain-serial-pool-'));
mkdirSync(join(root, 'scripts', 'lib'), { recursive: true });
mkdirSync(join(root, 'test'), { recursive: true });
for (const s of ['run-serial-tests.sh', 'lib/test-env.sh']) {
for (const s of ['run-serial-tests.sh', 'lib/test-env.sh', 'sharding.ts']) {
mkdirSync(dirname(join(root, 'scripts', s)), { recursive: true });
copyFileSync(resolve(REPO_ROOT, 'scripts', s), join(root, 'scripts', s));
}
@@ -41,7 +41,7 @@ function stageSandbox(): string {
const tools = [
'bash', 'sh', 'env', 'dirname', 'basename', 'mktemp', 'date', 'sleep',
'cat', 'tail', 'head', 'rm', 'mkdir', 'grep', 'sed', 'awk', 'wc', 'tr',
'find', 'sort', 'bun', 'timeout', 'gtimeout',
'find', 'sort', 'bun', 'timeout', 'gtimeout', 'pgrep',
];
for (const tool of tools) {
const p = Bun.which(tool);
@@ -104,6 +104,14 @@ describe('pooled serial runner', () => {
expect(r.out).toContain('test/b-ok.serial.test.ts');
expect(r.out).toContain('all 2 file(s) passed');
expect(r.out).toContain('pool=2');
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
expect(timing).toMatchObject({ version: 1, lane: 'serial', complete: true });
expect(timing.files).toHaveLength(2);
for (const file of timing.files) {
expect(file.status).toBe('pass');
expect(file.durationMs).toBeGreaterThan(0);
expect(file.attempts).toHaveLength(1);
}
rmSync(join(ROOT, 'test', 'a-ok.serial.test.ts'));
rmSync(join(ROOT, 'test', 'b-ok.serial.test.ts'));
});
@@ -111,8 +119,15 @@ describe('pooled serial runner', () => {
it('a failing file fails the run with its full log and a failed-files summary', () => {
writeFileSync(join(ROOT, 'test', 'a-ok.serial.test.ts'), PASSING);
writeFileSync(join(ROOT, 'test', 'z-bad.serial.test.ts'), FAILING);
const r = runScript();
const coverage = join(ROOT, 'coverage-failure');
mkdirSync(coverage);
writeFileSync(join(coverage, 'lane-manifest.json'), JSON.stringify({ complete: true }));
const r = runScript({ COVERAGE_DIR: coverage });
expect(r.code).toBe(1);
expect(existsSync(join(coverage, 'lane-manifest.json'))).toBe(false);
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
expect(timing.complete).toBe(false);
expect(timing.files.find((file: { file: string }) => file.file.endsWith('z-bad.serial.test.ts')).status).toBe('fail');
// Full bun log of the failing file is echoed (its assertion name shows).
expect(r.out).toContain('POOL_SENTINEL_ASSERTION');
expect(r.out).toContain('1 file(s) failed');
@@ -157,6 +172,10 @@ it('passes after one external SIGTERM', () => {
// marker + exit 0 are the contract.)
expect(r.out).toContain('rescued: external-kill phantom');
expect(r.code).toBe(0);
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
const killed = timing.files.find((file: { file: string }) => file.file.endsWith('k-killed.serial.test.ts'));
expect(killed.attempts.map((a: { status: string }) => a.status)).toEqual(['external-kill', 'pass']);
expect(timing.complete).toBe(true);
} finally {
rmSync(join(ROOT, 'test', 'k-killed.serial.test.ts'), { force: true });
rmSync(sentinel, { force: true });
@@ -236,4 +255,165 @@ it('passes after one external SIGTERM', () => {
rmSync(join(ROOT, 'scripts', 'serial-weights.json'), { force: true });
}
}, 60000);
it('shards weighted files exactly once, with exclusive work only on shard 1 and no SHARD in children', () => {
const names = ['a-heavy', 'b-mid', 'c-small', 'd-light', 'brain-repo-durability'];
const files = names.map(name => `test/${name}.serial.test.ts`);
writeFileSync(join(ROOT, 'fixture-module.ts'), 'export const answer = () => 42;\n');
const child = `import { it, expect } from 'bun:test'; import { answer } from '../fixture-module.ts'; it('routing is consumed', () => { expect(process.env.SHARD).toBeUndefined(); expect(answer()).toBe(42); });`;
for (const file of files) writeFileSync(join(ROOT, file), child);
writeFileSync(join(ROOT, 'scripts/serial-weights.json'), JSON.stringify(Object.fromEntries(files.map((file, i) => [file, 100 - i * 20]))));
try {
const collected: string[] = [];
for (let shard = 1; shard <= 4; shard++) {
const shardEnv = { ...ENV, SHARD: `${shard}/4` };
const list = execFileSync('bash', [join(ROOT, 'scripts/run-serial-tests.sh'), '--dry-run-list'], {
cwd: ROOT, encoding: 'utf8', env: shardEnv,
}).trim().split('\n').filter(Boolean);
collected.push(...list);
expect(list).toEqual([...list].sort());
expect(list.includes(files[4])).toBe(shard === 1);
const coverage = join(ROOT, `coverage-${shard}`);
const result = runScript({ SHARD: `${shard}/4`, COVERAGE_DIR: coverage });
expect(result.code, result.out).toBe(0);
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
expect(timing.lane).toBe(`serial-${shard}`);
expect(timing.files.map((file: { file: string }) => file.file).sort()).toEqual(list);
const manifest = JSON.parse(readFileSync(join(coverage, 'lane-manifest.json'), 'utf8'));
expect(manifest).toMatchObject({ lane: `serial-${shard}`, lcovCount: list.length, complete: true });
}
expect(collected.sort()).toEqual(files.sort());
} finally {
for (const file of files) rmSync(join(ROOT, file), { force: true });
rmSync(join(ROOT, 'fixture-module.ts'), { force: true });
rmSync(join(ROOT, 'scripts/serial-weights.json'), { force: true });
}
}, 60000);
it('empty shards produce complete empty evidence and malformed SHARD fails', () => {
writeFileSync(join(ROOT, 'test/a-only.serial.test.ts'), PASSING);
try {
const result = runScript({ SHARD: '4/4', COVERAGE_DIR: join(ROOT, 'coverage-empty') });
expect(result.code, result.out).toBe(0);
expect(result.out).toContain('all 0 file(s) passed');
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
expect(timing).toMatchObject({ lane: 'serial-4', complete: true, files: [] });
for (const SHARD of ['1', '0/4', '5/4', '1/0', '-1/4', '2/x']) {
expect(runScript({ SHARD }).code).toBe(2);
}
} finally {
rmSync(join(ROOT, 'test/a-only.serial.test.ts'), { force: true });
}
});
it('cancellation propagates to owned descendants and records incomplete timing', async () => {
const marker = join(ROOT, 'child.pid');
writeFileSync(join(ROOT, 'test/cancel.serial.test.ts'), `import { it } from 'bun:test';
import { writeFileSync } from 'fs';
it('keeps a child alive', async () => {
const child = Bun.spawn(['bun', '-e', 'setInterval(() => {}, 1000)']);
writeFileSync(${JSON.stringify(marker)}, String(child.pid));
await new Promise(() => {});
});`);
const runner = Bun.spawn(['bash', join(ROOT, 'scripts/run-serial-tests.sh')], {
cwd: ROOT, env: ENV, stdout: 'pipe', stderr: 'pipe',
});
let childPid = 0;
try {
const deadline = Date.now() + 10000;
while (!existsSync(marker) && Date.now() < deadline) await Bun.sleep(20);
expect(existsSync(marker)).toBe(true);
childPid = Number(readFileSync(marker, 'utf8'));
runner.kill('SIGTERM');
expect(await runner.exited).toBe(143);
// A dead child can briefly remain a zombie before init reaps it.
const until = Date.now() + 3000;
const alive = () => {
try {
const stat = readFileSync(`/proc/${childPid}/stat`, 'utf8');
return stat.split(') ')[1]?.[0] !== 'Z';
} catch { try { process.kill(childPid, 0); return true; } catch { return false; } }
};
while (alive() && Date.now() < until) await Bun.sleep(20);
expect(alive()).toBe(false);
const timing = JSON.parse(readFileSync(join(ROOT, '.context/serial-timings.json'), 'utf8'));
expect(timing.complete).toBe(false);
expect(timing.files[0].status).not.toBe('pass');
} finally {
try { runner.kill('SIGKILL'); } catch {}
if (childPid) { try { process.kill(childPid, 'SIGKILL'); } catch {} }
rmSync(join(ROOT, 'test/cancel.serial.test.ts'), { force: true });
rmSync(marker, { force: true });
}
}, 20000);
it.each([false, true])('exclusive cancellation allows graceful cleanup and records incomplete timing (rescue: %s)', async (rescue) => {
const root = stageSandbox();
const file = 'test/brain-repo-durability.serial.test.ts';
const marker = join(root, 'exclusive.pid');
const cleanup = join(root, 'cleanup-complete');
const coverage = join(root, 'coverage');
writeFileSync(join(root, file), `import { it } from 'bun:test';
import { existsSync, readFileSync, writeFileSync } from 'fs';
it('allows exclusive registration cleanup to finish', async () => {
const attemptsFile = ${JSON.stringify(join(root, 'attempts'))};
const attempt = existsSync(attemptsFile) ? Number(readFileSync(attemptsFile, 'utf8')) + 1 : 1;
writeFileSync(attemptsFile, String(attempt));
if (${rescue} && attempt === 1) { process.kill(process.pid, 'SIGTERM'); return; }
let cancelling = false;
process.on('SIGTERM', () => {
if (cancelling) return;
cancelling = true;
writeFileSync(${JSON.stringify(join(root, 'received-term'))}, 'TERM');
// Deliberately outlast the pooled runner's one-second escalation grace.
// Finishing this cleanup proves exclusive work was not force-killed.
setTimeout(() => {
writeFileSync(${JSON.stringify(cleanup)}, 'finished');
process.exit(0);
}, 1500);
});
writeFileSync(${JSON.stringify(marker)}, String(process.pid));
await new Promise(() => {});
});`);
const runner = Bun.spawn(['bash', join(root, 'scripts/run-serial-tests.sh')], {
cwd: root,
env: { ...ENV, PATH: join(root, 'bin'), SHARD: '1/4', COVERAGE_DIR: coverage },
stdout: 'ignore', stderr: 'ignore',
});
let exclusivePid = 0;
const alive = () => {
if (!exclusivePid) return false;
try {
if (process.platform === 'linux' && /\) Z /.test(readFileSync(`/proc/${exclusivePid}/stat`, 'utf8'))) return false;
process.kill(exclusivePid, 0);
return true;
} catch { return false; }
};
try {
const readyDeadline = Date.now() + 10000;
while (!existsSync(marker) && Date.now() < readyDeadline) await Bun.sleep(20);
expect(existsSync(marker)).toBe(true);
exclusivePid = Number(readFileSync(marker, 'utf8'));
runner.kill('SIGTERM');
const exitDeadline = Date.now() + 5000;
while ((runner.exitCode === null || !existsSync(cleanup) || alive()) && Date.now() < exitDeadline) await Bun.sleep(20);
expect(runner.exitCode).toBe(143);
expect(readFileSync(join(root, 'received-term'), 'utf8')).toBe('TERM');
expect(readFileSync(cleanup, 'utf8')).toBe('finished');
expect(alive()).toBe(false);
expect(existsSync(join(coverage, 'lane-manifest.json'))).toBe(false);
const timing = JSON.parse(readFileSync(join(root, '.context/serial-timings.json'), 'utf8'));
expect(timing).toMatchObject({ lane: 'serial-1', complete: false });
expect(timing.files).toHaveLength(1);
expect(timing.files[0].file).toBe(file);
expect(timing.files[0].status).not.toBe('pass');
expect(timing.files[0].attempts).toHaveLength(rescue ? 2 : 1);
if (rescue) expect(timing.files[0].attempts[0].status).toBe('external-kill');
} finally {
runner.kill('SIGKILL');
if (alive()) { try { process.kill(exclusivePid, 'SIGKILL'); } catch { /* already exited */ } }
await runner.exited;
rmSync(root, { recursive: true, force: true });
}
}, 20000);
});

View File

@@ -46,6 +46,7 @@ beforeAll(() => {
copyFileSync(PARALLEL_SH_SRC, join(TMPROOT, 'scripts', 'run-unit-parallel.sh'));
copyFileSync(SHARD_SH_SRC, join(TMPROOT, 'scripts', 'run-unit-shard.sh'));
copyFileSync(resolve(REPO_ROOT, 'scripts/sharding.ts'), join(TMPROOT, 'scripts/sharding.ts'));
copyFileSync(SERIAL_SH_SRC, join(TMPROOT, 'scripts', 'run-serial-tests.sh'));
chmodSync(join(TMPROOT, 'scripts', 'run-unit-parallel.sh'), 0o755);
chmodSync(join(TMPROOT, 'scripts', 'run-unit-shard.sh'), 0o755);
@@ -184,7 +185,7 @@ describe('run-unit-parallel.sh operator-interrupt cleanup', () => {
mkdirSync(join(root, 'scripts', 'lib'), { recursive: true });
mkdirSync(join(root, 'test'), { recursive: true });
mkdirSync(join(root, 'bin'), { recursive: true });
for (const s of ['run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh']) {
for (const s of ['sharding.ts', 'run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh']) {
copyFileSync(resolve(REPO_ROOT, 'scripts', s), join(root, 'scripts', s));
chmodSync(join(root, 'scripts', s), 0o755);
}
@@ -193,6 +194,7 @@ describe('run-unit-parallel.sh operator-interrupt cleanup', () => {
const fakeBun = join(root, 'bin', 'bun');
writeFileSync(fakeBun, `#!/usr/bin/env bash
[ "${'$'}{1:-}" = "scripts/sharding.ts" ] && exec ${JSON.stringify(process.execPath)} "${'$'}@"
[ "${'$'}{1:-}" = "test" ] || exit 0
echo "$$" > "${join(root, 'bun.pid')}"
trap '' INT TERM
@@ -268,7 +270,7 @@ describe('run-unit-parallel.sh no-timeout-binary fallback (rc from shard wait, n
FROOT = mkdtempSync(join(tmpdir(), 'gbrain-parallel-fallback-'));
mkdirSync(join(FROOT, 'scripts'), { recursive: true });
mkdirSync(join(FROOT, 'test'), { recursive: true });
for (const s of ['run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
for (const s of ['sharding.ts', 'run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
mkdirSync(dirname(join(FROOT, 'scripts', s)), { recursive: true });
copyFileSync(resolve(REPO_ROOT, 'scripts', s), join(FROOT, 'scripts', s));
chmodSync(join(FROOT, 'scripts', s), 0o755);
@@ -354,7 +356,7 @@ describe('run-unit-parallel.sh OOM rescue lane', () => {
OROOT = mkdtempSync(join(tmpdir(), 'gbrain-parallel-oom-'));
mkdirSync(join(OROOT, 'scripts'), { recursive: true });
mkdirSync(join(OROOT, 'test'), { recursive: true });
for (const s of ['run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
for (const s of ['sharding.ts', 'run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
mkdirSync(dirname(join(OROOT, 'scripts', s)), { recursive: true });
copyFileSync(resolve(REPO_ROOT, 'scripts', s), join(OROOT, 'scripts', s));
chmodSync(join(OROOT, 'scripts', s), 0o755);
@@ -531,7 +533,7 @@ describe('run-unit-parallel.sh no-timeout-binary wedge sentinel (rc 143 at cap
WROOT = mkdtempSync(join(tmpdir(), 'gbrain-parallel-wedge-'));
mkdirSync(join(WROOT, 'scripts'), { recursive: true });
mkdirSync(join(WROOT, 'test'), { recursive: true });
for (const s of ['run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
for (const s of ['sharding.ts', 'run-unit-parallel.sh', 'run-unit-shard.sh', 'run-serial-tests.sh', 'lib/test-env.sh']) {
mkdirSync(dirname(join(WROOT, 'scripts', s)), { recursive: true });
copyFileSync(resolve(REPO_ROOT, 'scripts', s), join(WROOT, 'scripts', s));
chmodSync(join(WROOT, 'scripts', s), 0o755);

View File

@@ -277,3 +277,8 @@ describe("loadWeights", () => {
}
});
});
it('distributes a fully zero-weight corpus instead of concentrating it in shard 1', () => {
const files = ['a', 'b', 'c', 'd'];
expect(partition(files, new Map(files.map(f => [f, 0])), 4)).toEqual([['a'], ['b'], ['c'], ['d']]);
});

View File

@@ -77,6 +77,17 @@ describe('test-shard.sh — exclusion contract', () => {
}
});
it('leaves the dedicated entity-card performance gate out of the unit matrix', () => {
const allFiles = [1, 2, 3, 4].flatMap(s => dryRunList(s, 4));
expect(allFiles).not.toContain('test/entity-card-perf.slow.test.ts');
const fs = require('fs');
const yaml = require('js-yaml');
const workflow = yaml.load(fs.readFileSync(resolve(REPO_ROOT, '.github/workflows/test.yml'), 'utf8'));
expect(workflow.jobs['slow-entity-resolve-perf'].steps.some((step: { run?: string }) =>
step.run?.includes('bun test test/entity-card-perf.slow.test.ts'))).toBe(true);
expect(workflow.jobs['test-status'].needs).toContain('slow-entity-resolve-perf');
});
it('INCLUDES *.slow.test.ts files (CI matrix is where slow files run)', () => {
const allFiles = [1, 2, 3, 4].flatMap(s => dryRunList(s, 4));
const slowFiles = allFiles.filter(f => /\.slow\.test\.ts$/.test(f));