Skip to content

Nightly corpus

Nightly corpus #37

# What the pull-request gate does not cover, run on a clock instead of on a
# diff: the PR-time specs on the two engines the gate never launches
# (playwright.browsers.config.ts), the interaction sweeps and throttled-network
# budgets, and the real-document matrix over a public corpus of real Office
# files (docs/superpowers/plans/2026-08-15-v9-test-coverage-strategy.md, section
# 2 tier 2 / section 6). Red here is a signal, not a gate.
#
# Two cadences, one nightly trigger (see the `gate` job):
# - cross-browser + sweeps run nightly. They are this repository's ONLY
# WebKit/Firefox coverage, and on 2026-08-25 that is what surfaced a
# shipped Safari defect the Chromium gate cannot see.
# - the corpus matrix runs on Sundays. It is 80 minutes against a corpus that
# never changes; what varies is this repository, and a week of it at a time
# is enough to keep the findings table current.
# Neither runs when nothing has been committed since the last time it did --
# `schedule` fires whether or not there is anything new to say.
name: Nightly corpus
on:
schedule:
- cron: "17 19 * * *" # 03:17 Asia/Shanghai
workflow_dispatch:
inputs:
limit:
description: "Max files to run (CORPUS_LIMIT)"
default: "300"
filter:
description: "Include regex on file paths (CORPUS_FILTER)"
default: ""
visual:
description: "Pixel-diff original vs re-opened save (CORPUS_VISUAL, empty to skip)"
default: "1"
jobs:
# Which of tonight's jobs have anything to say. Cheap (one checkout, no
# toolchain) and it runs every night, so the run itself is the record of why
# a quiet night was quiet -- a workflow that simply stops appearing is
# indistinguishable from one that broke.
gate:
name: What to run tonight
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
browsers: ${{ steps.decide.outputs.browsers }}
corpus: ${{ steps.decide.outputs.corpus }}
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Decide
id: decide
env:
MANUAL: ${{ github.event_name == 'workflow_dispatch' }}
run: |
age_hours=$(( ( $(date -u +%s) - $(git log -1 --format=%ct) ) / 3600 ))
# 1..7 with 7 = Sunday. The cron fires at 19:17 UTC, so this is the
# UTC weekday of the trigger, not of the Asia/Shanghai morning it
# lands in -- which is what the comment above the crons means.
weekday=$(date -u +%u)
browsers=false
corpus=false
# A cycle plus an hour of slack on each window: a commit landing just
# before the run it belongs to would otherwise wait a whole cycle.
# Written as `if` blocks rather than `a || b && c` -- that list
# evaluates to non-zero when both tests fail, and the step runs under
# `bash -e`, so a quiet night would end as a failed gate.
if [ "$MANUAL" = "true" ] || [ "$age_hours" -lt 25 ]; then
browsers=true
fi
if [ "$MANUAL" = "true" ] || { [ "$weekday" -eq 7 ] && [ "$age_hours" -lt 193 ]; }; then
corpus=true
fi
{
echo "HEAD is ${age_hours}h old, UTC weekday ${weekday}, manual=${MANUAL}"
echo ""
echo "- cross-browser + sweeps: ${browsers}"
echo "- corpus matrix: ${corpus}"
} >> "$GITHUB_STEP_SUMMARY"
echo "browsers=${browsers}" >> "$GITHUB_OUTPUT"
echo "corpus=${corpus}" >> "$GITHUB_OUTPUT"
corpus:
name: Real-document matrix (Apache POI test-data)
runs-on: ubuntu-latest
timeout-minutes: 300
needs: gate
if: needs.gate.outputs.corpus == 'true'
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Set up the toolchain
uses: ./.github/actions/setup
with:
browsers: chromium
# Apache POI's test-data tree (Apache-2.0) is the largest curated pile of
# real-world .doc/.docx/.xls/.xlsx/.ppt/.pptx around, including many
# regression files from real bug reports. Sparse + blobless clone keeps
# it to the three directories we need.
- name: Fetch public corpus (Apache POI test-data)
run: |
git clone --depth 1 --filter=blob:none --sparse https://github.com/apache/poi.git corpus-src
git -C corpus-src sparse-checkout set test-data/spreadsheet test-data/document test-data/slideshow
mkdir -p corpus
cp -r corpus-src/test-data/spreadsheet corpus/spreadsheet
cp -r corpus-src/test-data/document corpus/document
cp -r corpus-src/test-data/slideshow corpus/slideshow
echo "corpus files: $(find corpus -type f | wc -l)"
- name: Run corpus matrix
env:
CORPUS_DIR: ${{ github.workspace }}/corpus
CORPUS_LIMIT: ${{ github.event.inputs.limit || '300' }}
CORPUS_FILTER: ${{ github.event.inputs.filter || '' }}
CORPUS_VISUAL: ${{ github.event.inputs.visual || '1' }}
# Export to PDF and toggle readonly on each document. Legacy binary
# formats have no synthetic fixture anywhere in the suite, so this is
# the only run that exercises either action on a real .doc, .xls or
# .ppt. Both reuse the document already open; measured at a few
# hundred milliseconds per file.
CORPUS_DEEP: "1"
# Expected-to-fail inputs: encrypted/password files, deliberately
# truncated/corrupt regression samples, macro-only containers, and
# POI's fuzzer output (clusterfuzz-testcase-*, *Fuzzer*, Fuzzed.doc,
# poi-fuzz.xls, crash-<sha1>.*) -- byte soup kept precisely because
# it broke a parser, so "this editor will not open it" is not a
# finding. 15 of the 20 red rows on 2026-08-21 were these. The terms
# are narrow on purpose: a bare `crash` would also drop
# 51921-Word-Crash067.doc, which is a real document from a bug report.
CORPUS_EXCLUDE: "password|protect|encrypt|corrupt|truncat|broken|invalid|damaged|clusterfuzz|fuzzer|fuzzed|poi-fuzz|crash-[0-9a-f]{6}|\\.xlsm$|\\.xlsb$|\\.docm$|\\.pptm$"
run: pnpm exec playwright test test/e2e/corpus.spec.ts --workers=2 --reporter=list
- name: Summarize
if: always()
run: node bin/corpus-report.mjs test-results
- name: Upload corpus report
if: always()
uses: actions/upload-artifact@v7
with:
name: corpus-report
path: |
test-results/corpus-report.json
playwright-report/
retention-days: 14
browsers:
name: Cross-browser (WebKit + Firefox)
runs-on: ubuntu-latest
timeout-minutes: 90
needs: gate
if: needs.gate.outputs.browsers == 'true'
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Set up the toolchain
uses: ./.github/actions/setup
with:
browsers: webkit firefox
# The same PR-time suites (minus the opt-in sweeps) on the two engines
# the gate does not cover. Same-realm/L0 rules apply unchanged.
- name: Run E2E on WebKit and Firefox
run: pnpm exec playwright test -c playwright.browsers.config.ts --grep-invert "api surface|corpus|@serial" --reporter=list
# The other half of that `--grep-invert`, exactly as ci.yml's e2e job
# pairs them: an @serial case is one that cannot share the run. Without
# this split sw-silent-update.spec.ts rewrites the served sw.js while
# other specs are being served from it, replacing THEIR worker mid-test
# -- which is where sw-warm's "runtime-e2e-next" and cache-first's
# missing sentinel came from on 2026-08-25. Without the second pass
# nothing runs them at all and the job stays green regardless.
- name: Run the cases that cannot share the run
run: pnpm exec playwright test -c playwright.browsers.config.ts --grep @serial --workers=1 --reporter=list
- name: Upload report
if: always()
uses: actions/upload-artifact@v7
with:
name: playwright-report-browsers
path: playwright-report-browsers/
retention-days: 14
budgets:
name: Slow-network budgets
runs-on: ubuntu-latest
timeout-minutes: 30
needs: gate
if: needs.gate.outputs.browsers == 'true'
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Set up the toolchain
uses: ./.github/actions/setup
with:
browsers: chromium
# Roadmap direction 9 (L3): flag when the ecosystem packages on npm have
# moved past what this repo pins. Informational -- the bump itself is a
# PR (and for ranui usually a release from chaxus/ran first).
- name: ranui / ranuts upstream check
run: |
for pkg in ranui ranuts; do
PINNED=$(node -p "require('./package.json').dependencies['$pkg']")
LATEST=$(npm view "$pkg" version 2>/dev/null || echo unknown)
if [ "$PINNED" = "$LATEST" ]; then
echo "✅ $pkg pinned $PINNED == npm latest" | tee -a "$GITHUB_STEP_SUMMARY"
else
echo "⚠️ $pkg pinned $PINNED, npm latest $LATEST -- consider a bump PR (vendored IIFEs re-sync on build; ranui-vendor-sync.test.ts guards drift)" | tee -a "$GITHUB_STEP_SUMMARY"
fi
done
# Strategy section 9.1 layer 2: click every visible toolbar button of
# each editor once; L0 + "still saves" is the oracle. Found guard 8's
# comment/selection crash.
- name: UI crawl
env:
UI_CRAWL: "1"
run: pnpm exec playwright test test/e2e/ui-crawl.spec.ts --reporter=list
- name: Upload UI crawl reports
if: always()
uses: actions/upload-artifact@v7
with:
name: ui-crawl-reports
path: test-results/ui-crawl-*.json
retention-days: 14
# Cold open + first save under CDP network throttling (L4): the whole
# path must stay inside the save request's allowance.
- name: Throttled-network budget
env:
SLOW_NET: "1"
run: pnpm exec playwright test test/e2e/slow-network.spec.ts --reporter=list