diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fca79c39c2..596946ec09 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -43,15 +43,18 @@ jobs: if: github.event_name == 'pull_request' runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 with: # The gate diffs the PR range against its base — it needs history, # not just the merge commit. fetch-depth: 0 - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 + # setup-node v5 auto-enables pnpm caching from the root package.json + # `packageManager` field; this job never installs pnpm. + package-manager-cache: false - name: Test-presence gate (source change ⇒ test change) run: node runner/scripts/check-test-presence.mjs "${{ github.base_ref }}" @@ -62,13 +65,13 @@ jobs: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -86,15 +89,15 @@ jobs: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 # Read the pnpm version from runner/package.json's packageManager field # (avoids a conflict with the repo-root package.json). - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -110,7 +113,7 @@ jobs: run: pnpm typecheck - name: Upload the runtime build - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: runtime-dist path: runner/packages/runtime/dist/ @@ -124,13 +127,13 @@ jobs: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -139,7 +142,7 @@ jobs: - run: pnpm install --frozen-lockfile - name: Download the runtime build - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: runtime-dist path: runner/packages/runtime/dist/ @@ -155,7 +158,7 @@ jobs: run: node scripts/check-compiler-chunk.mjs - name: Upload the authoring build - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: authoring-dist path: runner/apps/authoring/dist/ @@ -176,13 +179,13 @@ jobs: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -191,7 +194,7 @@ jobs: - run: pnpm install --frozen-lockfile - name: Download the authoring build (the e2e target) - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: authoring-dist path: runner/apps/authoring/dist/ @@ -201,7 +204,7 @@ jobs: - name: Upload Playwright report on failure if: failure() - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: playwright-report # playwright-report/ holds the html report, test-results/ the traces. @@ -210,3 +213,111 @@ jobs: runner/test-results/ retention-days: 7 if-no-files-found: warn + + # The CI home for the local telemetry path (contract §10). Its own build, + # separate from the `authoring`/`e2e` jobs above — VITE_TELEMETRY_LOCAL=1 + # must never reach the real authoring-dist artifact those jobs share (the + # post-build leak check in master.yml exists to catch that), so this job + # builds its own dist and never uploads it. Both specs are self-contained + # (own preview server, `page.route` interception of `/telemetry/collect`, + # no o11y or API worker), unlike e2e/telemetry-metrics.spec.ts and + # e2e/o11y-local.spec.ts, which need a live worker/compose stack and run + # instead in `.github/workflows/e2e-o11y-local.yml`, off the per-PR gate. + e2e-telemetry: + needs: build + runs-on: ubuntu-latest + container: + image: mcr.microsoft.com/playwright:v1.61.1-noble + defaults: + run: + working-directory: runner + steps: + - uses: actions/checkout@v5 + + # The leak checks now run in PR CI as well as post-merge in master.yml, + # so a regression here (dead-code elimination of the local telemetry + # seam, or the dev-login bypass sentinel check itself) is caught before + # merge instead of surfacing as a blocked deploy. actionlint runs here + # too, so a workflow-YAML mistake in either fix is itself caught in PR + # CI, not just by reading the diff. + - name: actionlint (workflow YAML) + uses: reviewdog/action-actionlint@v1 + with: + fail_level: error + + - uses: pnpm/action-setup@v5 + with: + package_json_file: runner/package.json + + - uses: actions/setup-node@v5 + with: + node-version: 22 + cache: pnpm + cache-dependency-path: runner/pnpm-lock.yaml + + - run: pnpm install --frozen-lockfile + + - name: Download the runtime build + uses: actions/download-artifact@v7 + with: + name: runtime-dist + path: runner/packages/runtime/dist/ + + # Production-mode build (no VITE_TELEMETRY_LOCAL) — the same shape + # master.yml's `build` job ships, built here too so both leak checks + # run in PR CI, not only after merge. + - name: Build authoring in production mode + run: pnpm --filter @handsontable/demo-authoring build + + # Same check as master.yml's "Leak check — dev-login bypass must not + # reach the production bundle" step (AGENTS.md's own sanity check), + # now also gating the PR instead of only a post-merge deploy. + - name: Leak check — dev-login bypass must not reach the production bundle + run: | + if grep -rl "localhost:8787\|VITE_DEV_USER\|dev@handsontable.com" apps/authoring/dist; then + echo "::error::dev-login bypass leaked into the production bundle — rebuild with .env.local absent (AGENTS.md)" + exit 1 + fi + echo "ok: no dev-bypass sentinel found" + + # Same check as master.yml's "Leak check — local telemetry path must + # not reach the production bundle" step (contract §10). + - name: Leak check — local telemetry path must not reach the production bundle + run: pnpm check:telemetry-leak + + - name: Build authoring with the local telemetry path (contract §10) + run: pnpm --filter @handsontable/demo-authoring build + env: + VITE_TELEMETRY_LOCAL: '1' + + # Negative control: nothing in CI otherwise proves `check:telemetry-leak` + # can actually FAIL — only that it passes against builds where the + # sentinels are already absent. Assert it fails against a build made + # WITH the local telemetry path enabled; if the checker itself + # regresses (a + # dead-code-elimination change, a sentinel string edit) and stops + # detecting the leak, this step turns red instead of the check quietly + # passing everything forever. + - name: Leak check negative control — check:telemetry-leak must FAIL against a VITE_TELEMETRY_LOCAL=1 build + run: | + if pnpm check:telemetry-leak; then + echo "::error::check:telemetry-leak passed against a VITE_TELEMETRY_LOCAL=1 build — the leak check itself is broken" + exit 1 + fi + echo "ok: check:telemetry-leak correctly failed against the flag build" + + - name: E2E — Faro in the authoring app + example-analytics (T06, T12) + env: + E2E_TELEMETRY: '1' + run: pnpm e2e e2e/telemetry-faro.spec.ts e2e/example-analytics.spec.ts + + - name: Upload Playwright report on failure + if: failure() + uses: actions/upload-artifact@v6 + with: + name: playwright-report-telemetry + path: | + runner/playwright-report/ + runner/test-results/ + retention-days: 7 + if-no-files-found: warn diff --git a/.github/workflows/e2e-live.yml b/.github/workflows/e2e-live.yml index a2df0cd057..37df321986 100644 --- a/.github/workflows/e2e-live.yml +++ b/.github/workflows/e2e-live.yml @@ -220,6 +220,33 @@ jobs: E2E_BASE_URL: ${{ env.BASE_URL }} run: pnpm e2e e2e/ai-live.spec.ts + # A syntax error typed into a Tier-1 parcel example must reach + # /telemetry/collect as one `sandpack.compile_error` and no + # `preview.runtime_error`; a typed throwing line as one + # `preview.runtime_error`. Needs a VITE_TELEMETRY_LOCAL=1 build (the + # spec's own E2E_TELEMETRY gate, `page.route`-mocked collect) AND the + # hosted bundler (E2E_LIVE: the edit path needs a mounted Sandpack + # client), so it lives here rather than in ci.yml's deterministic + # e2e-telemetry job. Local mode only, and last: it rebuilds dist with + # the flag, then puts the plain build back. E2E_BASE_URL is a dummy + # that only disables the shared webServer (the spec serves its own + # dist); the guard proves the gated test ran instead of skipping to a + # green exit. + - name: E2E — typed syntax and runtime errors count once each (local telemetry build) + if: ${{ !inputs.smoke && env.BASE_URL == '' }} + env: + E2E_LIVE: '1' + E2E_TELEMETRY: '1' + E2E_BASE_URL: 'http://127.0.0.1:1' + PLAYWRIGHT_JSON_OUTPUT_NAME: telemetry-compile-error-report.json + run: | + VITE_TELEMETRY_LOCAL=1 pnpm --filter @handsontable/demo-authoring build + status=0 + pnpm e2e e2e/telemetry-faro.spec.ts -g "typed key by key reaches /telemetry/collect" --workers=1 --reporter=list,json || status=$? + pnpm --filter @handsontable/demo-authoring build + [ "$status" -eq 0 ] || exit $status + node scripts/ci/assert-e2e-ran.mjs telemetry-compile-error-report.json + - name: Upload Playwright report on failure if: failure() uses: actions/upload-artifact@v4 diff --git a/.github/workflows/e2e-o11y-local.yml b/.github/workflows/e2e-o11y-local.yml new file mode 100644 index 0000000000..10d9a74ef6 --- /dev/null +++ b/.github/workflows/e2e-o11y-local.yml @@ -0,0 +1,341 @@ +name: E2E — o11y local integration + +# CI home for the two gated specs docs/TESTING.md's "every gate must have a +# workflow home" rule left unhomed: +# +# e2e/telemetry-metrics.spec.ts (E2E_LIVE=1 + E2E_TELEMETRY=1) — a real +# local API worker (`wrangler dev`, Docker for the Tier-2 Sandbox +# container image) serving a VITE_TELEMETRY_LOCAL=1 build. +# e2e/o11y-local.spec.ts (E2E_O11Y_LOCAL=1) — a real API worker +# PLUS a real o11y worker (`wrangler dev`) and local ClickHouse/MinIO +# (Docker compose), proving telemetry actually reaches the real ingest +# pipeline and lands in Analytics Engine's local stand-in — not +# `page.route` mocks, which is what ci.yml's `e2e-telemetry` job already +# covers for telemetry-faro.spec.ts/example-analytics.spec.ts. Its own +# traffic is Tier-1 (Sandpack) only, so its API worker runs with +# `--enable-containers=false` — no Tier-2 container image. +# +# Both specs document (in their own file header) the exact local setup this +# workflow automates; see there for the "why" of each step. +# +# NOT the shared `mcr.microsoft.com/playwright` container image ci.yml and +# e2e-live.yml use: both specs spawn `wrangler dev` (telemetry-metrics's +# shells out to Docker to build/run the Tier-2 container image; o11y-local's +# doesn't, but still needs Docker for its own `docker compose` ClickHouse/ +# MinIO stack). Docker-in-Docker from inside that container image has no +# docker CLI and can't reach a sibling container's `localhost` port — this +# runs directly on the `ubuntu-latest` host, which ships Docker running +# (confirmed already relied on by master.yml's `deploy-api` job, which builds +# the same Tier-2 image via `wrangler deploy` on this exact runner image) and +# installs the Playwright browser itself instead. +# +# Kept OFF the per-PR merge gate for the common case (Docker + two `wrangler +# dev` boots + a compose stack costs several minutes no ordinary PR should +# pay), while still catching a regression before it ships: workflow_dispatch, +# a nightly run, and PRs that touch the ingest path these specs exist to +# prove (docs/run-and-deploy.md's "Tests (CI)" section, docs/TESTING.md's env +# gate taxonomy). + +on: + workflow_dispatch: {} + schedule: + # Nightly, clear of the other three nightly/weekly windows this repo + # already runs (01:00 starter matrix, 03:00 e2e-live prod canary, Monday + # 05:00 examples-build) so none of them ever contend for the same + # ubuntu-latest runner pool or Docker daemon. + - cron: '30 2 * * *' + pull_request: + paths: + - 'runner/workers/o11y/**' + - 'runner/containers/o11y/**' + - 'runner/apps/authoring/src/telemetry/**' + # e2e/telemetry-metrics.spec.ts boots a real local API worker + # (runner/workers/api) and asserts against it directly; both specs + # also exercise the shared @handsontable/demo-runtime telemetry code + # (runner/packages/**) that workers/api/src/analytics.ts and + # monitor-inject.ts import from — the same reasoning master.yml's own + # o11y deploy-gate comment gives for gating on the whole packages/ + # directory (e.g. scrub.ts imports redactPreviewHosts from + # monitor.ts, outside that subpath). Without these, a regression in + # either subtree is only caught by the nightly run or a manual + # workflow_dispatch, not by the PR that introduces it. + - 'runner/workers/api/**' + - 'runner/packages/**' + - 'runner/e2e/telemetry-metrics.spec.ts' + - 'runner/e2e/o11y-local.spec.ts' + - '.github/workflows/e2e-o11y-local.yml' + +permissions: + contents: read + +# Safe to cancel, unlike e2e-live.yml/e2e-starter-matrix.yml: both jobs here +# only ever touch throwaway state inside THIS runner's own ephemeral VM (its +# own Docker daemon, its own `wrangler dev` processes) — nothing prod-shaped +# to strand. +concurrency: + group: e2e-o11y-local-${{ github.ref }} + cancel-in-progress: true + +jobs: + # ---- e2e/telemetry-metrics.spec.ts ----------------------------------------- + telemetry-metrics: + runs-on: ubuntu-latest + timeout-minutes: 30 + defaults: + run: + working-directory: runner + steps: + - uses: actions/checkout@v5 + + # Job-scoped, never the machine-global default registry path — the same + # isolation rule COMMON.md gives every o11y task's own worktree, applied + # here so a stray leftover `wrangler dev` from a previous job on a + # self-hosted-style shared runner could never cross-wire service + # bindings with this run. `$RUNNER_TEMP` is wiped with the VM. Written + # to `$GITHUB_ENV` (not a job-level `env:`, which cannot see the + # `runner` context) so every later step, including the ones Playwright + # itself spawns `wrangler dev` from, inherits it. + - name: Isolate this job's wrangler dev registry + run: echo "WRANGLER_REGISTRY_PATH=${RUNNER_TEMP}/wrangler-registry" >> "$GITHUB_ENV" + + - uses: pnpm/action-setup@v5 + with: + package_json_file: runner/package.json + + - uses: actions/setup-node@v5 + with: + node-version: 22 + cache: pnpm + cache-dependency-path: runner/pnpm-lock.yaml + + - run: pnpm install --frozen-lockfile + + - name: Install the Playwright browser (no pre-built image on this runner) + run: pnpm exec playwright install --with-deps chromium + + - run: pnpm build + + # `apps/authoring/.env.local` must never be copied in (AGENTS.md — its + # VITE_DEV_USER poisons auth tests); this spec needs none of that, only + # workers/api/.dev.vars, generated fresh from the committed example + # exactly the way a developer's first `pnpm dev:live` bootstraps it + # (`scripts/dev-lib.mjs#bootstrapDevVars`), with PREVIEW_HOST patched to + # this spec's own API port (the file header's documented setup). + - name: Bootstrap workers/api/.dev.vars + run: | + cp workers/api/.dev.vars.example workers/api/.dev.vars + sed -i 's#^PREVIEW_HOST=.*#PREVIEW_HOST="localhost:4810"#' workers/api/.dev.vars + cat workers/api/.dev.vars + + # E2E_BASE_URL is set to a value this spec never reads (it hard-codes + # its own BASE_URL/API_BASE_URL and overrides Playwright's baseURL via + # `test.use()`) purely to make playwright.config.ts's shared `webServer` + # resolve to `undefined` — otherwise it would try to `vite preview` the + # default `dist/` this job never builds (the spec builds its own + # `dist-telemetry-metrics` instead), and fail before a single test runs. + # + # `--reporter=list,json` + `PLAYWRIGHT_JSON_OUTPUT_NAME`: the guard step + # right after this one reads `stats.expected`/`stats.skipped` from the + # json file — this is what catches a mistyped gate turning the whole + # step into a silent, all-skipped, exit-0 "pass" (docs/TESTING.md's + # "every gate must have a workflow home" only means something if this + # workflow provably runs the gated tests, not just exits clean). + - name: E2E — telemetry-metrics (E2E_LIVE + E2E_TELEMETRY) + env: + E2E_LIVE: '1' + E2E_TELEMETRY: '1' + E2E_BASE_URL: 'http://127.0.0.1:1' + PLAYWRIGHT_JSON_OUTPUT_NAME: telemetry-metrics-report.json + run: pnpm e2e e2e/telemetry-metrics.spec.ts --reporter=list,json + + - name: Guard — the gate actually ran tests, none silently skipped + run: node scripts/ci/assert-e2e-ran.mjs telemetry-metrics-report.json + + - name: Upload Playwright report on failure + if: failure() + uses: actions/upload-artifact@v6 + with: + name: playwright-report-telemetry-metrics + path: | + runner/playwright-report/ + runner/test-results/ + runner/telemetry-metrics-report.json + retention-days: 7 + if-no-files-found: warn + + # ---- e2e/o11y-local.spec.ts ------------------------------------------------- + o11y-local: + runs-on: ubuntu-latest + timeout-minutes: 30 + defaults: + run: + working-directory: runner + env: + COMPOSE_PROJECT_NAME: o11y-ci + # The exact port block e2e/o11y-local.spec.ts's own file header + # documents (its defaults for O11Y_DEV_PORT/O11Y_LOCAL_CLICKHOUSE_PORT + # already match these) — kept identical rather than reassigned, since a + # single-job ubuntu-latest runner has nothing else to collide with. + O11Y_MINIO_PORT: '5210' + O11Y_MINIO_CONSOLE_PORT: '5211' + O11Y_CLICKHOUSE_PORT: '5212' + O11Y_CLICKHOUSE_NATIVE_PORT: '5213' + AE_SQL_TOKEN: local-dev-token + steps: + - uses: actions/checkout@v5 + + # See the telemetry-metrics job's identical step for why this is + # written to $GITHUB_ENV instead of a job-level `env:` entry. + - name: Isolate this job's wrangler dev registry + run: echo "WRANGLER_REGISTRY_PATH=${RUNNER_TEMP}/wrangler-registry" >> "$GITHUB_ENV" + + - uses: pnpm/action-setup@v5 + with: + package_json_file: runner/package.json + + - uses: actions/setup-node@v5 + with: + node-version: 22 + cache: pnpm + cache-dependency-path: runner/pnpm-lock.yaml + + - run: pnpm install --frozen-lockfile + + - name: Install the Playwright browser (no pre-built image on this runner) + run: pnpm exec playwright install --with-deps chromium + + - run: pnpm build + + # No separate `minio-init` one-shot container: quay.io/minio/mc is not + # pullable. `minio` creates its own `loki` bucket via + # MINIO_DEFAULT_BUCKETS before its healthcheck goes green, so `--wait` + # (block until every named service is healthy/running) is sufficient. + - name: Start local ClickHouse + MinIO (Analytics Engine stand-in) + run: docker compose -f containers/o11y/compose.yml up -d --wait minio clickhouse + + - name: Wait for local ClickHouse + run: | + for _ in $(seq 1 60); do + if curl -sf "http://localhost:${O11Y_CLICKHOUSE_PORT}/ping" >/dev/null; then + echo "ClickHouse is up"; exit 0 + fi + sleep 2 + done + echo "::error::ClickHouse never answered on :${O11Y_CLICKHOUSE_PORT}" + docker compose -f containers/o11y/compose.yml logs clickhouse + exit 1 + + # Same non-secret local-only bypass values the file header names — + # DEV_ADMIN can be any non-empty string (workers/o11y/src/gates/session.ts + # honours it only when O11Y_ENV=local, contract §2, K1); the two secret + # fields the export/Sentry-hook routes check are exercised with fixed + # CI-only values, never real ones. + - name: Bootstrap workers/o11y/.dev.vars + run: | + cp workers/o11y/.dev.vars.example workers/o11y/.dev.vars + sed -i \ + -e 's#^DEV_ADMIN=.*#DEV_ADMIN=ci-e2e@handsontable.com#' \ + -e 's#^O11Y_EXPORT_SECRET=.*#O11Y_EXPORT_SECRET=ci-e2e-export-secret#' \ + -e 's#^SENTRY_HOOK_SECRET=.*#SENTRY_HOOK_SECRET=ci-e2e-sentry-hook-secret#' \ + -e "s#^AE_SQL_TOKEN=.*#AE_SQL_TOKEN=${AE_SQL_TOKEN}#" \ + -e 's#^O11Y_ENV=.*#O11Y_ENV=local#' \ + workers/o11y/.dev.vars + { + echo "O11Y_LOCAL_MINIO_PORT=${O11Y_MINIO_PORT}" + echo "O11Y_LOCAL_CLICKHOUSE_PORT=${O11Y_CLICKHOUSE_PORT}" + echo "RUNNER_EVENTS_CLICKHOUSE_URL=http://localhost:${O11Y_CLICKHOUSE_PORT}" + } >> workers/o11y/.dev.vars + + # PREVIEW_HOST must be a non-production host (the file header) or the + # API worker starts reporting to the real Sentry project — matched to + # this spec's own API_PORT (5280) the same way telemetry-metrics.spec.ts's + # job matches its API port above. + - name: Bootstrap workers/api/.dev.vars + run: | + cp workers/api/.dev.vars.example workers/api/.dev.vars + sed -i 's#^PREVIEW_HOST=.*#PREVIEW_HOST="localhost:5280"#' workers/api/.dev.vars + { + echo "RUNNER_EVENTS_CLICKHOUSE_URL=http://localhost:${O11Y_CLICKHOUSE_PORT}" + echo "AE_SQL_TOKEN=${AE_SQL_TOKEN}" + } >> workers/api/.dev.vars + + - name: Apply local D1 migrations (workers/api) + working-directory: runner/workers/api + run: pnpm exec wrangler d1 migrations apply handsontable-demos --local + + # o11y-local.spec.ts starts its OWN api worker + authoring preview + # internally, but expects the o11y worker already answering (its + # `beforeAll` throws a clear "start it first" error otherwise) — this + # is that "first". Backgrounded + polled here rather than left to the + # spec, because the spec's own error message assumes a human did this + # step by hand. + # + # `--enable-containers=false` (hidden flag, `wrangler dev --help` does + # not list it — confirmed present in this pinned wrangler 4.136.3's + # CLI parser): without it, `wrangler dev` runs an "⎔ Preparing + # container image(s)…" step first to build the GrafanaBox image (Loki + # 3.3.2 + Grafana 11.4.0 base pulls, plugins) before it ever opens + # :5220, which alone can outlast this poll's 120s budget on a cold + # runner. e2e/o11y-local.spec.ts never touches Grafana or Loki — it + # asserts against local ClickHouse rows — so nothing here needs the + # box, and building it would only cost minutes for no coverage. The + # box is otherwise only woken by the (unfired-in-dev) backlog cron and + # the `/grafana/*` visit route, neither of which this spec exercises, + # so `/telemetry/*` ingest still works unaffected. + - name: Start the o11y worker (wrangler dev) + working-directory: runner/workers/o11y + run: | + pnpm exec wrangler dev --port 5220 --inspector-port 5221 --enable-containers=false \ + > "${RUNNER_TEMP}/o11y-dev.log" 2>&1 & + echo $! > "${RUNNER_TEMP}/o11y-dev.pid" + # No `-f`: this worker registers no route at "/", so a healthy + # worker answers with a plain 404 — matching `scripts/dev.mjs`'s + # own `waitForServer`, which polls with a bare `fetch()` and + # treats any response (2xx or not) as "up". `-f` would turn that + # 404 into a curl failure and the loop would never succeed, no + # matter how fast the worker started. `--max-time 5` caps each + # attempt so a hung/misbundled `wrangler dev` still fails within + # the 60x2s budget instead of stalling on a single slow request. + for _ in $(seq 1 60); do + if curl -s --max-time 5 -o /dev/null "http://localhost:5220"; then + echo "o11y worker is up"; exit 0 + fi + sleep 2 + done + echo "::error::o11y worker never answered on :5220" + tail -n 100 "${RUNNER_TEMP}/o11y-dev.log" + exit 1 + + # Same E2E_BASE_URL trick as the telemetry-metrics job — see that job's + # comment for why. + - name: E2E — o11y-local (E2E_O11Y_LOCAL) + env: + E2E_O11Y_LOCAL: '1' + E2E_BASE_URL: 'http://127.0.0.1:1' + PLAYWRIGHT_JSON_OUTPUT_NAME: o11y-local-report.json + run: pnpm e2e e2e/o11y-local.spec.ts --reporter=list,json + + - name: Guard — the gate actually ran tests, none silently skipped + run: node scripts/ci/assert-e2e-ran.mjs o11y-local-report.json + + - name: Stop the o11y worker + if: always() + run: | + pid="$(cat "${RUNNER_TEMP}/o11y-dev.pid" 2>/dev/null || true)" + [ -n "$pid" ] && kill "$pid" 2>/dev/null || true + + - name: Tear down ClickHouse + MinIO + if: always() + run: docker compose -f containers/o11y/compose.yml down -v + + - name: Upload Playwright report on failure + if: failure() + uses: actions/upload-artifact@v6 + with: + name: playwright-report-o11y-local + path: | + runner/playwright-report/ + runner/test-results/ + runner/o11y-local-report.json + retention-days: 7 + if-no-files-found: warn diff --git a/.github/workflows/master.yml b/.github/workflows/master.yml index d4c75c2f8d..5b5d2c73b0 100644 --- a/.github/workflows/master.yml +++ b/.github/workflows/master.yml @@ -37,6 +37,10 @@ on: description: 'Deploy the API worker + Tier-2 container image' type: boolean default: false + deploy_o11y: + description: 'Deploy the observability worker + Grafana box image (T10)' + type: boolean + default: false permissions: contents: read @@ -53,8 +57,9 @@ jobs: outputs: authoring: ${{ steps.detect.outputs.authoring }} api: ${{ steps.detect.outputs.api }} + o11y: ${{ steps.detect.outputs.o11y }} steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 with: # Full history: the diff below spans the whole push range, and a # multi-commit push (rebase-and-merge, a direct push of several @@ -65,8 +70,11 @@ jobs: name: Detect what this push touches run: | if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then - echo "authoring=${{ inputs.deploy_authoring }}" >> "$GITHUB_OUTPUT" - echo "api=${{ inputs.deploy_api }}" >> "$GITHUB_OUTPUT" + { + echo "authoring=${{ inputs.deploy_authoring }}" + echo "api=${{ inputs.deploy_api }}" + echo "o11y=${{ inputs.deploy_o11y }}" + } >> "$GITHUB_OUTPUT" exit 0 fi # The push event's `before` bounds the range: HEAD^ only covers the @@ -84,28 +92,44 @@ jobs: # The same path sets the two deploy workflows used to declare under # `on.push.paths` (catalog.json is authoring-only: the app bundles it # at build time; scripts/ and containers/ are api-only). - authoring=false; api=false + authoring=false; api=false; o11y=false grep -qE '^runner/(apps/authoring/|packages/|config/|catalog\.json$|pnpm-lock\.yaml$)|^\.github/workflows/(master|ci)\.yml$' /tmp/changed.txt && authoring=true - grep -qE '^runner/(workers/api/|containers/|scripts/|config/|packages/|pnpm-lock\.yaml$)|^\.github/workflows/(master|ci)\.yml$' /tmp/changed.txt && api=true - echo "authoring=$authoring" >> "$GITHUB_OUTPUT" - echo "api=$api" >> "$GITHUB_OUTPUT" + # Minor triage item 4 (C-M5): `containers/` used to match ALL of + # it, including a Grafana-box image edit under containers/o11y/ — + # any such edit also redeployed the API worker and rebuilt/pushed + # the unrelated Tier-2 image. Narrowed to the two containers/ + # subtrees the API worker's own deploy actually builds/pushes + # (containers/live/, containers/builder/); containers/o11y/ stays + # gated on the o11y line below only. + grep -qE '^runner/(workers/api/|containers/(live|builder)/|scripts/|config/|packages/|pnpm-lock\.yaml$)|^\.github/workflows/(master|ci)\.yml$' /tmp/changed.txt && api=true + # T10: workers/o11y (the worker itself), containers/o11y (the Grafana + # box image) and packages/ (the shared telemetry module — gated on the + # whole directory, not just packages/runtime/src/telemetry/, since + # e.g. scrub.ts imports redactPreviewHosts from monitor.ts, outside + # that subpath) all redeploy the o11y worker. + grep -qE '^runner/(workers/o11y/|containers/o11y/|packages/|pnpm-lock\.yaml$)|^\.github/workflows/(master|ci)\.yml$' /tmp/changed.txt && o11y=true + { + echo "authoring=$authoring" + echo "api=$api" + echo "o11y=$o11y" + } >> "$GITHUB_OUTPUT" - # One install, one workspace build — shared by both deploys via artifacts. + # One install, one workspace build — shared by all three deploys via artifacts. build: needs: [changes] - if: needs.changes.outputs.authoring == 'true' || needs.changes.outputs.api == 'true' + if: needs.changes.outputs.authoring == 'true' || needs.changes.outputs.api == 'true' || needs.changes.outputs.o11y == 'true' runs-on: ubuntu-latest defaults: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -116,7 +140,7 @@ jobs: - run: pnpm build - name: Upload the runtime build - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: runtime-dist path: runner/packages/runtime/dist/ @@ -127,8 +151,10 @@ jobs: # VITE_API_BASE comes from apps/authoring/.env.production (committed) so # it targets prod. SENTRY_* are only set on this deploying build; PR CI # (ci.yml) gets no token, so PR builds neither emit source maps nor - # upload a release — see apps/authoring/vite.config.ts. - - name: Build authoring (skipped when only the api deploys) + # upload a release — see apps/authoring/vite.config.ts. VITE_SENTRY_SCOPE + # (contract §11) is pinned to "full" explicitly here, rather than relying + # on the code's own absent-means-full default. + - name: Build authoring (skipped when only the api/o11y deploys) if: needs.changes.outputs.authoring == 'true' run: pnpm --filter @handsontable/demo-authoring build env: @@ -136,6 +162,77 @@ jobs: SENTRY_ORG: ${{ vars.SENTRY_ORG }} SENTRY_PROJECT: ${{ vars.SENTRY_PROJECT }} GITHUB_SHA: ${{ github.sha }} + VITE_SENTRY_SCOPE: full + + # ADR §C.3: the Sentry plugin above already uploaded these maps to + # Sentry for stack-trace symbolication there; here they're removed from + # dist/ only after also landing in R2 for the o11y worker's own + # symbolicator (workers/o11y/src/drain/symbolicate.ts#mapKeyFor: + # `sourcemaps//.map`). No maps + # means SENTRY_* wasn't configured (a PR build, or an api/o11y-only + # push) — the loop is then a no-op. + # + # Dedicated R2 S3 credential (Object Read & Write, scoped to + # `handsontable-demos-o11y-maps` — one-time setup step 2 in + # docs/run-and-deploy.md), via `aws s3 cp` (the AWS CLI ships + # preinstalled on GitHub-hosted Ubuntu runners). + - name: Upload source maps to R2, then remove them from dist + if: needs.changes.outputs.authoring == 'true' + env: + AWS_ACCESS_KEY_ID: ${{ secrets.R2_MAPS_ACCESS_KEY_ID }} + AWS_SECRET_ACCESS_KEY: ${{ secrets.R2_MAPS_SECRET_ACCESS_KEY }} + # Required by the SDK/CLI's request signing; R2 does not use it + # (contract: the region for an R2 bucket's S3 API is always "auto"). + AWS_DEFAULT_REGION: auto + # EU jurisdiction (contract §2 — same value workers/o11y/wrangler.jsonc's + # own account_id + box.ts's Loki-bucket endpoint use), not a secret. + R2_MAPS_ENDPOINT: https://15111272c53ed0aaf84a908f0c9c7f8b.eu.r2.cloudflarestorage.com + # A modern AWS CLI defaults to a CRC32 trailing checksum on every + # request and validates one on every response; R2's S3 API doesn't + # implement that trailer, so the upload can be rejected outright. + # Both must be "when_required" (only send/validate when the + # operation explicitly requires it) so `aws s3 cp` against R2 works + # with whatever CLI version the runner image ships. + AWS_REQUEST_CHECKSUM_CALCULATION: when_required + AWS_RESPONSE_CHECKSUM_VALIDATION: when_required + run: | + set -o pipefail + mapfile -t maps < <(find apps/authoring/dist -name '*.map') + if [ ${#maps[@]} -eq 0 ]; then + echo "no source maps in this build — nothing to upload to R2 (expected unless SENTRY_* secrets are set)" + else + for f in "${maps[@]}"; do + rel="${f#apps/authoring/dist/}" + key="sourcemaps/${GITHUB_SHA}/${rel}" + echo "uploading ${rel} -> ${key}" + aws s3 cp "$f" "s3://handsontable-demos-o11y-maps/${key}" \ + --endpoint-url "$R2_MAPS_ENDPOINT" --content-type application/json + rm -f "$f" + done + fi + + # AGENTS.md's own dev-bypass sanity check ("Always build prod with + # .env.local absent"). Must run AFTER the maps are gone + # above — a source map's sourcesContent literally embeds + # "localhost:8787" (App.tsx's API-base fallback) and "VITE_DEV_USER" + # (auth.ts), which would false-fire this exact grep against a map file + # that was never actually leaked into the served JS. + - name: Leak check — dev-login bypass must not reach the production bundle + if: needs.changes.outputs.authoring == 'true' + run: | + if grep -rl "localhost:8787\|VITE_DEV_USER\|dev@handsontable.com" apps/authoring/dist; then + echo "::error::dev-login bypass leaked into the production bundle — rebuild with .env.local absent (AGENTS.md)" + exit 1 + fi + echo "ok: no dev-bypass sentinel found" + + # Contract §10's own leak check: the local-only telemetry path + # (VITE_TELEMETRY_LOCAL) must never survive dead-code elimination into a + # production bundle. scripts/check-telemetry-leak.mjs greps + # apps/authoring/dist/assets/*.js for its sentinels. + - name: Leak check — local telemetry path must not reach the production bundle + if: needs.changes.outputs.authoring == 'true' + run: pnpm check:telemetry-leak # The one regression that only shows up in a deploy: the compiler chunk's content hash # coming back, which strands every tab opened before the next deploy (DEV-2569). ci.yml @@ -147,7 +244,7 @@ jobs: - name: Upload the authoring build if: needs.changes.outputs.authoring == 'true' - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v6 with: name: authoring-dist path: runner/apps/authoring/dist/ @@ -158,14 +255,20 @@ jobs: needs: [changes, build] if: needs.changes.outputs.authoring == 'true' runs-on: ubuntu-latest + # id-token: write — the deploy-event step below mints a GitHub OIDC token + # for POST /telemetry/deploy. Job-level permissions REPLACE the + # workflow-level block, so contents: read is repeated here. + permissions: + contents: read + id-token: write defaults: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - name: Download the authoring build - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: authoring-dist path: runner/apps/authoring/dist/ @@ -173,10 +276,55 @@ jobs: # No pnpm install: wrangler is pinned through npx and ships ./dist as # Workers Assets — the checkout only supplies wrangler.jsonc. - name: Deploy authoring worker + id: deploy working-directory: runner/apps/authoring env: CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} - run: npx -y wrangler@4.108.0 deploy + run: | + set -o pipefail + npx -y wrangler@4.108.0 deploy | tee /tmp/deploy-authoring.log + # Minor triage item 3 (C-M3): `|| true` so a wording change in + # wrangler's own "Current Version ID:" line (grep finds nothing, + # exits 1) can't fail this step — under `set -o pipefail` that + # would fail the whole job AFTER the deploy already shipped, + # skipping the deploy-event report and the smoke test below it. + version_id=$(grep -oE 'Current Version ID:.*' /tmp/deploy-authoring.log | awk '{print $NF}') || true + # B-I1: the `|| true` above trades "job goes red" for "job stays + # green" on a wrangler wording change — this warning is what makes + # that trade safe: an empty version_id would otherwise ship a + # deploy-event row with cf_version_id:"" (silently corrupting the + # ADR §C.2 deploy-correlation record) with nothing pointing at the + # actual root cause (the version-id parse, not the deploy itself). + [ -n "$version_id" ] || echo "::warning::could not parse Current Version ID from wrangler deploy output (handsontable-demos-authoring) — the deploy-event report below will ship with an empty cf_version_id" + echo "version_id=${version_id}" >> "$GITHUB_OUTPUT" + + - name: Report the deploy to the o11y worker (T10, ADR §C.2) + # Never fails the job — see "Deploy events" in docs/run-and-deploy.md: + # a deploy that shipped must not go red because reporting it hiccuped, + # and on the very first merge of this feature the o11y route may not + # be reachable yet. + if: always() && steps.deploy.outcome == 'success' + working-directory: . + # VERSION_ID passed through env, the same way GITHUB_SHA already is — + # never spliced with ${{ }} straight into the script, which would let + # an unexpectedly-shaped value (e.g. one containing a quote) break out + # of the JSON string it's built into. + env: + VERSION_ID: ${{ steps.deploy.outputs.version_id }} + run: | + oidc_token=$(curl -sf -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=https://demos.handsontable.com/telemetry/deploy" \ + | jq -r '.value') || true + if [ -z "$oidc_token" ]; then + echo "::warning::could not mint a GitHub OIDC token — skipping the deploy event" + exit 0 + fi + code=$(curl -sf -o /dev/null -w '%{http_code}' -X POST https://demos.handsontable.com/telemetry/deploy \ + -H "Authorization: Bearer ${oidc_token}" -H 'Content-Type: application/json' \ + --data "{\"event\":\"deploy\",\"service\":\"handsontable-demos-authoring\",\"sha\":\"${GITHUB_SHA}\",\"cf_version_id\":\"${VERSION_ID}\"}") \ + || true + echo "deploy-event response: ${code:-}" + [ "$code" = "200" ] || echo "::warning::deploy event for handsontable-demos-authoring did not return 200 (${code:-})" - name: Smoke test — prod frontend serves current bundle working-directory: . @@ -216,20 +364,31 @@ jobs: smoke: true deploy-api: - needs: [changes, build] - if: needs.changes.outputs.api == 'true' + # Also needs deploy-o11y (mutual service bindings — API's own O11Y + # binding and o11y's own API#O11yUsage binding). `if` accepts a skipped + # deploy-o11y (unrelated push, nothing to wait for) but not a failed one — + # see "First deploy, in order" in docs/run-and-deploy.md. + needs: [changes, build, deploy-o11y] + if: >- + !cancelled() && + needs.changes.outputs.api == 'true' && + needs.build.result == 'success' && + needs.deploy-o11y.result != 'failure' runs-on: ubuntu-latest + permissions: + contents: read + id-token: write defaults: run: working-directory: runner steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v5 - - uses: pnpm/action-setup@v4 + - uses: pnpm/action-setup@v5 with: package_json_file: runner/package.json - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v5 with: node-version: 22 cache: pnpm @@ -241,7 +400,7 @@ jobs: - run: pnpm install --frozen-lockfile - name: Download the runtime build - uses: actions/download-artifact@v4 + uses: actions/download-artifact@v7 with: name: runtime-dist path: runner/packages/runtime/dist/ @@ -259,10 +418,39 @@ jobs: # which attaches the demos.handsontable.com routes via --routes (routes are # intentionally NOT in wrangler.jsonc — see ADR-0020 / run-and-deploy.md). - name: Deploy API worker (builds + pushes Tier-2 image, attaches routes) + id: deploy working-directory: runner/workers/api env: CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} - run: pnpm run deploy + run: | + set -o pipefail + pnpm run deploy | tee /tmp/deploy-api.log + # Minor triage item 3 (C-M3): see the matching comment in + # deploy-authoring — same reasoning, same fix. + version_id=$(grep -oE 'Current Version ID:.*' /tmp/deploy-api.log | awk '{print $NF}') || true + # B-I1: see the matching comment in deploy-authoring — same reasoning, same fix. + [ -n "$version_id" ] || echo "::warning::could not parse Current Version ID from wrangler deploy output (handsontable-demos-api) — the deploy-event report below will ship with an empty cf_version_id" + echo "version_id=${version_id}" >> "$GITHUB_OUTPUT" + + - name: Report the deploy to the o11y worker (T10, ADR §C.2) + if: always() && steps.deploy.outcome == 'success' + working-directory: . + env: + VERSION_ID: ${{ steps.deploy.outputs.version_id }} + run: | + oidc_token=$(curl -sf -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=https://demos.handsontable.com/telemetry/deploy" \ + | jq -r '.value') || true + if [ -z "$oidc_token" ]; then + echo "::warning::could not mint a GitHub OIDC token — skipping the deploy event" + exit 0 + fi + code=$(curl -sf -o /dev/null -w '%{http_code}' -X POST https://demos.handsontable.com/telemetry/deploy \ + -H "Authorization: Bearer ${oidc_token}" -H 'Content-Type: application/json' \ + --data "{\"event\":\"deploy\",\"service\":\"handsontable-demos-api\",\"sha\":\"${GITHUB_SHA}\",\"cf_version_id\":\"${VERSION_ID}\"}") \ + || true + echo "deploy-event response: ${code:-}" + [ "$code" = "200" ] || echo "::warning::deploy event for handsontable-demos-api did not return 200 (${code:-})" - name: Smoke test — prod API health working-directory: . @@ -275,3 +463,79 @@ jobs: done echo "::error::prod /api/health did not return 200 after deploy" exit 1 + + # The observability worker (InboxWriter + GrafanaBox). Deployed before + # deploy-api (see that job's `needs` and docs/run-and-deploy.md's "First + # deploy, in order") because of their mutual service bindings. + deploy-o11y: + needs: [changes, build] + if: needs.changes.outputs.o11y == 'true' + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write + defaults: + run: + working-directory: runner + steps: + - uses: actions/checkout@v5 + + - uses: pnpm/action-setup@v5 + with: + package_json_file: runner/package.json + + - uses: actions/setup-node@v5 + with: + node-version: 22 + cache: pnpm + cache-dependency-path: runner/pnpm-lock.yaml + + # Install stays: `pnpm run deploy` runs the workspace's own wrangler and + # bundles the worker, which resolves @handsontable/demo-runtime/telemetry + # from the artifact downloaded below instead of rebuilding it. + - run: pnpm install --frozen-lockfile + + - name: Download the runtime build + uses: actions/download-artifact@v7 + with: + name: runtime-dist + path: runner/packages/runtime/dist/ + + # Builds + pushes the Grafana box image (Docker required, same as + # deploy-api) and attaches the /telemetry/* and /grafana/* routes via + # --routes (never in wrangler.jsonc — ADR-0020), plus + # --var SERVICE_VERSION:$GITHUB_SHA (contract §2). + - name: Deploy o11y worker (builds + pushes the Grafana box image, attaches routes) + id: deploy + working-directory: runner/workers/o11y + env: + CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} + run: | + set -o pipefail + pnpm run deploy | tee /tmp/deploy-o11y.log + # Minor triage item 3 (C-M3): see the matching comment in + # deploy-authoring — same reasoning, same fix. + version_id=$(grep -oE 'Current Version ID:.*' /tmp/deploy-o11y.log | awk '{print $NF}') || true + # B-I1: see the matching comment in deploy-authoring — same reasoning, same fix. + [ -n "$version_id" ] || echo "::warning::could not parse Current Version ID from wrangler deploy output (handsontable-demos-o11y) — the deploy-event report below will ship with an empty cf_version_id" + echo "version_id=${version_id}" >> "$GITHUB_OUTPUT" + + - name: Report the deploy to the o11y worker (T10, ADR §C.2) + if: always() && steps.deploy.outcome == 'success' + working-directory: . + env: + VERSION_ID: ${{ steps.deploy.outputs.version_id }} + run: | + oidc_token=$(curl -sf -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=https://demos.handsontable.com/telemetry/deploy" \ + | jq -r '.value') || true + if [ -z "$oidc_token" ]; then + echo "::warning::could not mint a GitHub OIDC token — skipping the deploy event" + exit 0 + fi + code=$(curl -sf -o /dev/null -w '%{http_code}' -X POST https://demos.handsontable.com/telemetry/deploy \ + -H "Authorization: Bearer ${oidc_token}" -H 'Content-Type: application/json' \ + --data "{\"event\":\"deploy\",\"service\":\"handsontable-demos-o11y\",\"sha\":\"${GITHUB_SHA}\",\"cf_version_id\":\"${VERSION_ID}\"}") \ + || true + echo "deploy-event response: ${code:-}" + [ "$code" = "200" ] || echo "::warning::deploy event for handsontable-demos-o11y did not return 200 (${code:-})" diff --git a/runner/.gitignore b/runner/.gitignore index 0f054a1455..682a7c6e1e 100644 --- a/runner/.gitignore +++ b/runner/.gitignore @@ -21,6 +21,19 @@ build/ test-results/ playwright-report/ .playwright/ +# telemetry e2e specs' own private preview builds (telemetry-faro.spec.ts's +# dist-uncaught-scope, example-analytics.spec.ts's dist-example-analytics) — +# `dist/` above only matches the literal name, not these (D-I1 fix round). +apps/authoring/dist-*/ # generated starter-matrix run reports — paste into the ticket/PR instead docs/reports/ + +# Minor triage item 8 (C-M15): node:test scratch dirs copied out of +# workers/api/src (so a bare @sentry/cloudflare / @handsontable/demo-runtime +# specifier still resolves through workers/api/node_modules — see +# pipeline/api-telemetry-diagnostic.test.mjs and +# pipeline/api-telemetry-cron-step.test.mjs's own header comments). Removed +# by the test itself on the success path; this covers the failure path too +# (a failing import used to leave the directory behind for `git add -A`). +workers/api/.hot-* diff --git a/runner/AGENTS.md b/runner/AGENTS.md index 959b94abb1..d2079ae9c0 100644 --- a/runner/AGENTS.md +++ b/runner/AGENTS.md @@ -58,24 +58,13 @@ build step. Edit shell files and the app picks them up on hot-reload. ## Run locally -Two processes. **API worker** (Tier-2 live containers need Docker running): - -```bash -cd workers/api -# .dev.vars (gitignored) holds the local login stand-in — see src/auth.ts. -npx wrangler dev # http://localhost:8787 -``` - -**Authoring app**: - -```bash -cd apps/authoring -# .env.local (gitignored): VITE_API_BASE=http://localhost:5173 # this dev server, not :8787 -# VITE_DEV_USER=you@handsontable.com # bypasses broker login locally -pnpm --filter @handsontable/demo-authoring dev # http://localhost:5173 -``` - -Two traps in that setup: +`pnpm dev` (Tier 1), `pnpm dev:live` (+ the API worker, Docker), `pnpm dev:full` +(+ o11y, compose, the Slack capture server) — one orchestrator +(`scripts/dev.mjs --tier=1|2|full`) behind all three; `pnpm o11y:dev` is a +separate standalone o11y-only command. Full details, every port/env var, and +the `.dev.vars` bootstrap rules are in `docs/run-and-deploy.md`'s "Run +locally" section — this is just the two traps worth knowing up front if you +ever run a piece by hand instead: - **`?mode=full` needs the SPA and the worker on one origin.** `serveDemoAsset` sends `frame-ancestors 'self'` + `X-Frame-Options: SAMEORIGIN` for `/d/:id` and is not wrapped in @@ -84,8 +73,10 @@ Two traps in that setup: which renders as `● error` on a demo that is fine. `vite.config.ts` proxies `/api`, `/d` and `/embed` to the worker for exactly this reason — keep `VITE_API_BASE` on the dev server's own origin. An **empty** value does not work: `App.tsx` falls back to `:8787` on any falsy value. + (`pnpm dev:live`/`dev:full` set this correctly for you, as process env, not a file.) - **`VITE_DEV_USER` short-circuits `currentUser()` before the fetch.** Any test of the real broker - path has to override it (`VITE_DEV_USER= pnpm dev`) or it silently exercises the bypass instead. + path has to override it (`VITE_DEV_USER= pnpm --filter @handsontable/demo-authoring dev`) or it + silently exercises the bypass instead. ## Verify before pushing @@ -167,6 +158,9 @@ grep -rl "localhost:8787\|VITE_DEV_USER\|dev@handsontable.com" apps/authoring/di is what catches a leaked `.env.local`; `localhost:8787` catches a missing `.env.production` (the `|| "http://localhost:8787"` fallback in `App.tsx` surviving into the bundle). Don't widen that term to a bare `localhost:` — catalog README text mentions dev-server ports and it false-fires. +This grep does not, and does not need to, cover `VITE_TELEMETRY_LOCAL` (a separate local-only +build-time flag, `.env.example`) — `scripts/check-telemetry-leak.mjs` is the dedicated check for +that one, run right after the production build in CI. ## CI/CD @@ -185,6 +179,29 @@ The two deploy workflows authenticate with a repository secret (`CLOUDFLARE_API_ no credential is committed. CI reads the pnpm version from `runner/package.json` so it does not pick up the repo-root manifest. +## Code comments + +Comments state what the code cannot say for itself, once. Review and fix history belongs in the commit +message and the PR, not in the source. This applies to people and agents alike, and reviewers treat a +violation as a finding. + +- **No review ids.** No finding, round, task or review ids (`F32`, `A-C1`, `R9`, `T06`, "fix round", + "per the reviewer") in comments or test titles. Keep contract and ADR pointers (`contract §5`, + `ADR-0041 §L.9`) and ClickUp ids (`DEV-NNNN`) of work that is still open. +- **Present tense.** Describe the code as it is: no "used to", "before this fix", "the old …", + "was changed", "fails without the fix", and no revert-check evidence. +- **No tombstones.** When you delete code, delete the comments that talk about it. +- **One sentence of why.** Say why a guard, constant or ordering exists, with the measurement when it + sets a limit, and stop. Leave out the alternatives you considered, the failure story, a restatement + of what the code plainly does, and rationale that is already written down elsewhere (point to it). +- **Length caps.** A file header is at most 6 lines, a constant or field doc at most 3, a function doc + at most 8. Only an ordering or concurrency proof may run longer, by up to about 6 lines. +- **Density.** In `apps/authoring` and `scripts`, stay near the existing code's density, about 0.4 + comment lines per code line. +- **Tests.** A test title states the behaviour, with no id. A block comment is worth it only for a + non-obvious setup or oracle, under the same caps. +- **Docs** (contract, ADRs, runbook) record rules and decisions, not the story of how an issue was found. + ## Conventions / guardrails - **No secrets in git.** Auth is the Handsontable Google login broker (per-user token, sessionStorage). Dev bypasses live only in gitignored `.env.local` / `.dev.vars`. @@ -214,7 +231,8 @@ does not pick up the repo-root manifest. imported under the global's own identifier, and Handsontable's CDN CSS becomes an npm import so the demo follows the version picker. Copying verbatim produced demos that could not run. - Tier-2 containers stay warm while a tab is open (client keepalive + `sleepAfter=5m`); disk is ephemeral, so a slept container cold-boots on return. -- **Cost guardrails** (DEV-2030, ADR-0022): `max_instances` 10/5 is the container cap **and** the live-preview concurrency limit (no queue sits in front of the pool; the overflow caller gets the `at_capacity` 503) — raised from 5/3 in DEV-2909, and not to be changed again without redoing the arithmetic in `docs/cost-guardrails.md` in the same commit. Spend degrades live sessions in stages while static shares keep serving. The dollar thresholds and the enforcement switch are **editable at runtime in `/admin`** (stored in `runner_settings`); the `BUDGET_*` vars are only defaults. Pool pressure — `at_capacity` refusals, hourly awake-seconds and sampled peak concurrency — is instrumented per **ADR-0040**, because daily totals cannot yield peak concurrency and that is what sizing the pool actually depends on. +- **Cost guardrails** (DEV-2030, ADR-0022): `max_instances` 10/5 is the container cap **and** the live-preview concurrency limit (no queue sits in front of the pool; the overflow caller gets the `at_capacity` 503) — raised from 5/3 in DEV-2909, and not to be changed again without redoing the arithmetic in `docs/cost-guardrails.md` in the same commit. Spend degrades live sessions in stages while static shares keep serving. The dollar thresholds and the enforcement switch are **editable at runtime in `/admin`** (stored in `runner_settings`); the `BUDGET_*` vars are only defaults. Pool pressure — `at_capacity` refusals, awake-seconds and sampled peak concurrency — was specified by **ADR-0040** and is delivered under **ADR-0041** (built, on this branch), as Analytics Engine points rather than the hourly-bucketed D1 rows ADR-0040 originally proposed (`docs/cost-guardrails.md`): the `at_capacity` counter in `usage_daily` as ADR-0040 C.1 says (`recordUsageEvent`, `workers/api/src/index.ts`); awake-seconds as `session.end`'s `value` (contract §5, per session, bucketable into an hourly view by the Grafana query rather than a dedicated hourly metric); and peak concurrency as the `pool.gauge` Analytics Engine point sampled every 5 minutes (`telemetry/cron.ts#emitPoolGauge`) — because daily totals cannot yield peak concurrency and that is what sizing the pool actually depends on. +- **Observability** (ADR-0041, **built, verified locally and on the sandbox platform, not yet deployed**): a self-hosted, sleeping Loki + Grafana box (`workers/o11y/`, `containers/o11y/`) fed over OTLP, with metrics in Workers Analytics Engine, plus Sentry for uncaught errors (errors escaping a handler, ADR-0041 §E.1) and spend alerts. `SENTRY_SCOPE`/`VITE_SENTRY_SCOPE` still default to `full` in every committed config, so Sentry keeps receiving everything it does today until the launch plan (`docs/run-and-deploy.md`) flips the switch — nothing about today's Sentry behaviour changes until then. New telemetry follows the ADR's `hot.*` attribute set and its "never a label" rule for demo id, session id, cf-ray and the user pseudonym, checked by `pipeline/telemetry-contract.test.mjs`. ADR-0042 (example analytics) ships with it. ADR-0043 (`/admin` cutover) follows after launch, once ADR-0041 itself is Accepted (see its own status line — two items are pending: a real-Workers-isolate symbolication measurement, and an R2 retention-lifecycle clock still running). Shared names, Analytics Engine slots and payload shapes are frozen in [`docs/observability-contract.md`](docs/observability-contract.md); the deploy runbook, one-time setup and post-deploy checklist are in [`docs/run-and-deploy.md`](docs/run-and-deploy.md)'s Observability and Launch plan sections. - **Ask AI** (DEV-2047, `docs/example-chat.md`): chat panel scoped to the open example; docs chunks are retrieved **in the browser** (Cloudflare blocks Worker→workers.dev, error 1042), the Worker adds Algolia page links and calls LiteLLM. Model edits are proposed, never auto-applied, and every answer is metered into the cost ledger. - **Style panel** (DEV-2047, `docs/style-panel.md`): Theme Builder's controls applied to the open example via Handsontable's **JS theme API** (`registerTheme`/`params`), written into the demo as a real module and wired into the grid; `POST /api/theme` does natural-language styling. Controls show the demo's own Handsontable version's preset defaults — fetched at runtime, override state stays empty until you edit — and the panel is gated off below core 17, which has no theme API (DEV-2560). - **Analytics are anonymous by construction** — no cookies, no IPs, no user agents, no query strings, no per-request rows; unique visitors use a daily-rotating salted hash. Keep it that way when touching `workers/api/src/analytics.ts`. diff --git a/runner/apps/authoring/.env.example b/runner/apps/authoring/.env.example new file mode 100644 index 0000000000..a456c2d830 --- /dev/null +++ b/runner/apps/authoring/.env.example @@ -0,0 +1,79 @@ +# Local dev config for the authoring app. Copy the lines you need into +# your gitignored `.env.local` — never commit that file. `.env.production` +# (committed, no secrets) covers the deployed build; this file documents the +# local-dev-only variables and their defaults. + +# API worker origin. `App.tsx` falls back to http://localhost:8787 on any +# falsy value, and `auth.ts`/`catalog.ts` have the same fallback — this line +# is only needed to point at something else (a deployed API, or this dev +# server's own origin for the `/api`/`/d`/`/embed` proxy — see vite.config.ts). +# `pnpm dev:live`/`pnpm dev:full` (docs/run-and-deploy.md) set this for you +# as process env, not a file — you only need this line for a manual, +# outside-the-orchestrator `vite` invocation. +# VITE_API_BASE=http://localhost:8787 + +# Local login bypass — an internal @handsontable.com address. NEVER set this +# in a committed file or a production build; it short-circuits currentUser() +# before the broker round trip (AGENTS.md). Same note as VITE_API_BASE above: +# `pnpm dev:live`/`dev:full` inject this as process env automatically. +# VITE_DEV_USER=you@handsontable.com + +# Sentry DSN. Reporting is gated on the deployed hostname regardless of this +# value (`reportingGate.ts`) — set it only to verify the Sentry wiring against +# a separate, non-production project (docs/run-and-deploy.md). +# VITE_SENTRY_DSN= + +# Contract §11 / ADR §E.3. "full" (default, unset) sends explicit diagnostic +# reports (reportError, the Tier-1/Tier-2 branches of reportRuntimeError) to +# both Sentry and the facade; "uncaught" sends them to the facade only — +# Sentry then keeps only what escapes a handler (window.onerror, +# unhandledrejection, Sentry.ErrorBoundary) plus the budget-alert +# captureMessage. +# VITE_SENTRY_SCOPE=full + +# Demo-runtime monitoring (DEV-2527), temporary — see docs/run-and-deploy.md. +# VITE_MONITOR_DEMOS=1 + +# Contract §10. Build-time flag that opens Faro's LOCAL telemetry path — also +# requires the runtime host to be localhost/127.0.0.1 (`telemetry/gate.ts`). +# Never set in a production build. Unlike when this note was first written, +# the post-build leak check DOES catch this one now: `scripts/check-telemetry-leak.mjs` +# greps for VITE_TELEMETRY_LOCAL's own name and the test seams it gates +# (CrashProbe, the T06 e2e Sentry-capture hooks) — CI runs it right after the +# production build. Still, double-check `.env.local` is absent before +# building for real; the leak check is a backstop, not a reason to be sloppy. +# Faro then posts to same-origin `/telemetry/collect`, proxied in +# vite.config.ts to the o11y worker's local `wrangler dev` — read from the +# `O11Y_DEV_PORT` env var (default 4200), not a hardcoded port; see +# docs/run-and-deploy.md's "Run locally" section (`pnpm dev:full` sets this +# automatically). +# VITE_TELEMETRY_LOCAL=1 + +# /admin's "Open Grafana" link target. Defaults to `/grafana/`, which only +# resolves on the deployed zone (the authoring app and the o11y worker share +# one origin there) — NOT proxied through vite.config.ts locally, because +# Grafana's own GF_SERVER_ROOT_URL is the o11y worker's origin and a dev +# proxy would just make its redirects bounce off it. `pnpm dev:full` sets +# this for you to the o11y worker's own local origin +# (`http://localhost:/grafana/`), injected as process env for +# that run only, never written to a file; see docs/run-and-deploy.md's +# "Browsing logs" section. Leave unset here: unlike VITE_TELEMETRY_LOCAL, +# `pnpm check:telemetry-leak` does NOT scan for this one, so a value pasted +# into a real .env.local would silently bake into a "production" build. +# VITE_GRAFANA_URL= + +# The Handsontable docs-assistant search endpoint (DEV-2047). Defaults to the +# hosted one; override to point at a local instance. +# VITE_DOCS_SEARCH_URL= + +# The Google login broker. Defaults to the hosted Render service. +# VITE_LOGIN_BROKER_URL= + +# The CodeSandbox-hosted Sandpack bundler (Tier-1 preview). Self-hosting it +# stack-overflows on HOT v18 (AGENTS.md) — do not point this at a local build. +# VITE_SANDPACK_BUNDLER_URL= + +# Set by CI/the deploy workflow, not a developer — the full git SHA, matched +# against the Sentry release the source-map upload targets. Leave unset +# locally. +# VITE_SENTRY_RELEASE= diff --git a/runner/apps/authoring/package.json b/runner/apps/authoring/package.json index bbf5a015bd..80c83184bc 100644 --- a/runner/apps/authoring/package.json +++ b/runner/apps/authoring/package.json @@ -11,6 +11,7 @@ }, "dependencies": { "@fontsource/fira-code": "^5.3.0", + "@grafana/faro-web-sdk": "2.12.1", "@handsontable/demo-editor-shell": "workspace:*", "@handsontable/demo-runtime": "workspace:*", "@sentry/react": "^10.68.0", diff --git a/runner/apps/authoring/src/Admin.tsx b/runner/apps/authoring/src/Admin.tsx index f3bd907220..868b501ce9 100644 --- a/runner/apps/authoring/src/Admin.tsx +++ b/runner/apps/authoring/src/Admin.tsx @@ -19,6 +19,7 @@ import { useCallback, useEffect, useState } from "react"; import { theme, logoUrl } from "@handsontable/demo-editor-shell"; import { assertApiOk, readApiJson } from "./api.js"; import { reportError } from "./sentry.js"; +import { apiHeaders } from "./telemetry/index.js"; interface LedgerRow { day: string; sku: string; source: string; units: number; usd: number } interface UsageRow { day: string; metric: string; dimension: string; count: number } @@ -62,6 +63,10 @@ export interface BudgetSettings { closedUsd: number; enforce: boolean; alertsUsd: number[]; + /** ADR-0041 §G: the o11y stack's own monthly ceiling (default $15), + * separate from `limitUsd`. Optional: an older API response won't + * carry it, so the panel must render without crashing. */ + o11yBudgetUsd?: number; source?: "defaults" | "override"; updatedAt?: string | null; updatedBy?: string | null; @@ -92,6 +97,15 @@ interface UsageReport { reconciled: boolean; enforced: boolean; }; + /** ADR-0041 §G: "`/admin` shows app, observability and total." Optional: + * an older-deployed API worker won't send it; the panel renders + * without this line rather than crashing. */ + o11y?: { + spendUsd: number; + capUsd: number; + appSpendUsd: number; + totalSpendUsd: number; + }; settings: BudgetSettings; audience: Audience; spendBySku: Record; @@ -111,6 +125,12 @@ interface UsageReport { const WINDOWS = [7, 30, 90]; +/** `import.meta.env.VITE_GRAFANA_URL` — on the deployed zone a plain + * `/grafana/` reaches the o11y worker (same origin); locally + * `scripts/dev-lib.mjs` sets this to the worker's own dev port. Not + * covered by `check:telemetry-leak` — don't set it outside dev-lib.mjs. */ +const GRAFANA_URL = import.meta.env.VITE_GRAFANA_URL || "/grafana/"; + const usd = (n: number): string => (n >= 100 ? `$${n.toFixed(0)}` : n >= 1 ? `$${n.toFixed(2)}` : `$${n.toFixed(3)}`); const int = (n: number): string => n.toLocaleString("en-US"); const duration = (seconds: number): string => { @@ -133,6 +153,8 @@ const SKU_LABEL: Record = { workers: "Workers requests", r2: "R2 storage", llm: "AI assistant", + o11y_container: "Observability container", + o11y_workers: "Observability workers", }; const METRIC_LABEL: Record = { @@ -167,7 +189,7 @@ export function AdminPanel({ apiBase, token }: AdminPanelProps) { (window: number) => { setError(null); fetch(`${apiBase}/api/admin/usage?days=${window}`, { - headers: token ? { Authorization: `Bearer ${token}` } : {}, + headers: apiHeaders(token ? { Authorization: `Bearer ${token}` } : undefined), }) .then(async (r) => { if (!r.ok) throw new Error(`usage request failed (${r.status})`); @@ -209,6 +231,17 @@ export function AdminPanel({ apiBase, token }: AdminPanelProps) { ))} + {/* ADR-0043: dashboards live in Grafana behind the o11y worker's own + * broker login, and nothing else in the app links there — a plain + * same-tab-avoiding anchor is enough, no client-side auth needed. */} + + Open Grafana ↗ + ← Editor @@ -365,8 +398,9 @@ export function AdminPanel({ apiBase, token }: AdminPanelProps) { /** Budget headline: where spend sits against the ceiling, and what each * threshold will do when it is crossed. */ function BudgetCard({ report }: { report: UsageReport }) { - const { budget, settings } = report; + const { budget, settings, o11y } = report; const tier = TIERS[budget.tier] ?? { label: budget.tier, color: theme.color.text }; + const o11yOverCap = o11y ? o11y.spendUsd >= o11y.capUsd : false; const pct = Math.max(0, Math.min(1, budget.pct)); const limit = settings.limitUsd || 1; const marks: [string, number][] = [ @@ -408,6 +442,26 @@ function BudgetCard({ report }: { report: UsageReport }) { : "Observe-only: tiers are computed and logged but nothing is refused. Turn enforcement on below " + "once these figures track the Cloudflare Billable Usage dashboard."}

+ + {/* ADR-0041 §G: app, observability and total. Absent entirely against + an older API response that predates this line (see the `o11y?` + doc comment on UsageReport) — nothing to show, so nothing renders. */} + {o11y && ( +
+ App: {usd(o11y.appSpendUsd)} + + Observability: + {usd(o11y.spendUsd)} + of {usd(o11y.capUsd)} cap + {o11yOverCap && ( + + backlog drains paused + + )} + + Total: {usd(o11y.totalSpendUsd)} +
+ )} ); } @@ -464,10 +518,10 @@ function SettingsForm({ try { const res = await fetch(`${apiBase}/api/admin/settings`, { method, - headers: { + headers: apiHeaders({ "Content-Type": "application/json", ...(token ? { Authorization: `Bearer ${token}` } : {}), - }, + }), body: method === "PUT" ? JSON.stringify({ ...draft, @@ -517,6 +571,11 @@ function SettingsForm({
{field("closedUsd", "Close live editing ($)", `${pctOf(draft.closedUsd)} — running sessions torn down.`)} + {field( + "o11yBudgetUsd", + "Observability cap ($)", + "ADR-0041 §G: crossing this pauses backlog drains (visit wakes still work). Counts toward the ceiling above too.", + )}
} + // ADR §E.2: a render crash never reaches `window.onerror` on its own + // (React swallows it into `componentDidCatch`), so this is the Faro + // half of the tee — Sentry already captures it via the boundary. + // `reportUncaughtError`, not `telemetry.error()`, which is handled-only. + onError={(error) => reportUncaughtError(error)} > + diff --git a/runner/apps/authoring/src/profile.ts b/runner/apps/authoring/src/profile.ts index 34d25fb7b5..8a8b2da43e 100644 --- a/runner/apps/authoring/src/profile.ts +++ b/runner/apps/authoring/src/profile.ts @@ -18,6 +18,7 @@ import { readApiJson } from "./api.js"; import { getToken, PROFILE_CACHE_KEY } from "./auth.js"; import { reportError } from "./sentry.js"; +import { apiHeaders } from "./telemetry/index.js"; /** Mirrors `ProfileView` in `workers/api/src/profile-store.ts`. */ export interface Profile { @@ -36,9 +37,12 @@ export interface Profile { const CACHE_KEY = PROFILE_CACHE_KEY; +/** Every one of this file's `fetch` calls is an API call, so this is also + * where `x-hot-session` (T06, `apiHeaders`) rides along — one call site to + * touch instead of four. */ function authHeaders(): Record { const token = getToken(); - return token ? { Authorization: `Bearer ${token}` } : {}; + return Object.fromEntries(apiHeaders(token ? { Authorization: `Bearer ${token}` } : undefined).entries()); } /** The cached profile, but only if it belongs to `email`. A stale row from a diff --git a/runner/apps/authoring/src/sentry.ts b/runner/apps/authoring/src/sentry.ts index 15365f600c..554fbe57c7 100644 --- a/runner/apps/authoring/src/sentry.ts +++ b/runner/apps/authoring/src/sentry.ts @@ -14,10 +14,26 @@ import { sanitizeMonitorPayload, type MonitorPayload, } from "@handsontable/demo-runtime/monitor"; +import { + fingerprint as contractFingerprint, + fingerprintShape, + type HotAttrs, + type HtMajor, +} from "@handsontable/demo-runtime/telemetry"; import { ApiError } from "./apiError.js"; import { resolveReporting } from "./reportingGate.js"; -import { isEdgelessForeignSessionStart, isOfficeScannerRejection } from "./eventGate.js"; +import { + applyFaroTee, + isEdgelessForeignSessionStart, + isForeignUnhandled, + isOfficeScannerRejection, + isUnhandledNoise, +} from "./eventGate.js"; +import { resolveSentryScope, reportsDiagnosticToSentry } from "./sentryScope.js"; +import { demoEventReport, type DemoMonitorKind } from "./demoEventReport.js"; +import { createDemoEventCollapse, type PushOutcome } from "./demoEventCollapse.js"; import { tier2StderrReport } from "./tier2Report.js"; +import { telemetry } from "./telemetry/index.js"; const DSN = import.meta.env.VITE_SENTRY_DSN as string | undefined; @@ -51,233 +67,287 @@ const reporting = resolveReporting({ export const reportingEnabled = reporting.enabled; -/** - * Demo-runtime monitoring (DEV-2527). Temporary and deliberately build-time: off is - * a one-line commit plus a deploy, which is why the in-page caps in - * `packages/runtime/src/monitor.ts` are the brake that acts immediately. Removal - * path in docs/run-and-deploy.md. - * - * No second host gate: `reportingEnabled` already pins reporting to production, so - * local runs and PR CI stay silent whatever this is set to. It inherits the - * automation gate through the same flag, also deliberately (DEV-2540) — an - * e2e-driven page load is not real demo usage, so the ~36-minute `e2e:matrix` run - * against production no longer relays preview events either. - */ -export const monitorDemos = - reportingEnabled && (import.meta.env.VITE_MONITOR_DEMOS as string | undefined) === "1"; +/** The same two build-time+host conditions as Faro's local path (contract + * §10) — a pure function of `VITE_TELEMETRY_LOCAL`, so a production build + * folds this branch to dead code (`check:telemetry-leak` greps for the + * flag's name as proof). Declared before `diagnosticsGoToSentry`, which + * calls it. */ +function localTestSentryEnabled(): boolean { + return ( + (import.meta.env.VITE_TELEMETRY_LOCAL as string | undefined) === "1" && + typeof window !== "undefined" && + (window.location.hostname === "localhost" || window.location.hostname === "127.0.0.1") + ); +} -/** The `environment` (and tag) demo-side events are filed under, so a flood of them - * can be rate-limited or muted in the Sentry UI without touching the app — the only - * brake that works without a build. */ -const DEMO_SURFACE = "demo-runtime"; +/** Whether SOME Sentry client is initialised — production or the local + * test-capture path. Distinct from `reportingEnabled`: `monitorDemos` + * gates real preview instrumentation and must stay production-only. */ +const sentryActive = reportingEnabled || localTestSentryEnabled(); -/** - * Browser noise that is never actionable: a benign layout-loop warning browsers - * surface as an error, plus the shapes an in-flight request takes when the user - * navigates away mid-fetch (`Failed to fetch` in Chrome, `Load failed` in Safari). - * - * These are matched ONLY against unhandled errors — see `isUnhandledNoise`. They - * must not go in `ignoreErrors`: that runs in the event-filters integration, which - * processes every event including explicit `captureException` calls, so - * `/Failed to fetch/` there would silently discard the offline broker and - * `/api/versions` failures that `reportError` exists to surface. - * - * The two other NOT-OURS populations this project has classified — the Office - * scanner rejection (DEMOS-5F) and the edgeless-foreign session-start facet - * (DEMOS-9) — are NOT regexes here. They live in `eventGate.ts`, gated in - * `beforeSend` below, and are pinned by `pipeline/sentry-gating.test.mjs`. Adding - * another regex to this array for either would lose that test coverage. - */ -const UNHANDLED_NOISE = [ - /^ResizeObserver loop/i, - /^AbortError/i, - /Failed to fetch/i, - /Load failed/i, -]; +/** Contract §11 / ADR §E.3: whether an explicit diagnostic report also + * reaches Sentry, besides the facade (which always receives it). Gated + * on `sentryActive`, not `reportingEnabled` directly. */ +const SENTRY_SCOPE = resolveSentryScope(import.meta.env.VITE_SENTRY_SCOPE as string | undefined); +export const diagnosticsGoToSentry = reportsDiagnosticToSentry(sentryActive, SENTRY_SCOPE); -/** - * True for a global `onerror` / `onunhandledrejection` event whose message is - * known noise. `mechanism.handled === false` is what distinguishes those from - * anything we reported on purpose (`captureException` sets `handled: true`), and - * it is populated before `beforeSend` runs. - */ -function isUnhandledNoise(event: Sentry.ErrorEvent): boolean { - const values = event.exception?.values ?? []; - return values.some( - (v) => - v.mechanism?.handled === false && - UNHANDLED_NOISE.some((re) => re.test(v.value ?? "") || re.test(v.type ?? "")), - ); -} +/** Demo-runtime monitoring (DEV-2527), deliberately build-time: off is a + * one-line commit + deploy. Inherits the production/automation gate + * through `reportingEnabled`, so local runs and PR CI stay silent. */ +export const monitorDemos = + reportingEnabled && (import.meta.env.VITE_MONITOR_DEMOS as string | undefined) === "1"; -/** - * True for an *unhandled* event whose stack points outside this app's origin. - * - * The preview iframe runs arbitrary authored and imported example code, so a typo - * there is product output, not an application fault — see `reportRuntimeError` in - * App.tsx. Being cross-origin, the iframe cannot reach this window's error handlers - * at all; this is the backstop for whatever does arrive that way (the Sandpack - * bundler, a container preview host, an injected extension script). - * - * Scoped to `mechanism.handled === false` — the same discriminator - * `isUnhandledNoise` uses, and for the same reason. Applied to every event, as it - * was, it silently discarded explicit `reportError` and ErrorBoundary reports whose - * stack merely *passed through* a foreign frame: precisely the failure the - * `UNHANDLED_NOISE` note above avoids by keeping those regexes out of - * `ignoreErrors`. `reportDemoEvent`'s relays are exempted at the callsite too — - * they carry preview-origin frames by definition, so a future change to how they - * are captured must not be able to re-break ingest through this path. - */ -function isForeignUnhandled(event: Sentry.ErrorEvent): boolean { - const values = event.exception?.values ?? []; - return values.some( - (v) => - v.mechanism?.handled === false && - (v.stacktrace?.frames ?? []).some( - (f) => f.filename?.startsWith("http") && !f.filename.startsWith(window.location.origin), - ), - ); -} +/** Widens `monitorDemos`' "preview reporter injected" gate with the local + * leg, so Faro/facade reach the local stack under `dev:full`; Sentry + * calls stay gated on the real `monitorDemos` (see call sites below). */ +export const previewMonitoring = monitorDemos || localTestSentryEnabled(); -if (reportingEnabled) { - Sentry.init({ - dsn: DSN, - environment: reporting.environment, - // `|| undefined` matters: the define below substitutes "" when GITHUB_SHA is - // absent, and a release of "" would not match the SHA-named artifact bundle - // the plugin uploads — source maps would silently stop resolving. +/** The `environment`/tag demo-side events are filed under, so they can be + * rate-limited in the Sentry UI without a build. Under `full` scope, + * `reportDemoEvent` re-homes into it (see `beforeSend` below). */ +const DEMO_SURFACE = "demo-runtime"; + +/** Everything `Sentry.init()` needs beyond `dsn`/`transport` — shared by + * the production and local test-capture inits so they can't drift. */ +function sharedSentryOptions(environment: string): Sentry.BrowserOptions { + return { + environment, + // `|| undefined`: a "" release would not match the SHA-named source-map bundle. release: (import.meta.env.VITE_SENTRY_RELEASE as string | undefined) || undefined, - // Errors only. Spans would triple the event volume for signal we don't act on. - tracesSampleRate: 0, - // Stated rather than inherited, because `reportDemoEvent` now writes into this - // buffer (DEV-2539). The SDK default is 100 and it keeps the *most recent* N, so - // with the default a demo spending its whole `MONITOR_BREADCRUMB_CEILING` (50) - // would evict half the authoring app's own trail — a save failure would file an - // issue whose breadcrumbs are demo warnings instead of the user's clicks and - // fetches. At 200 the demo's ceiling can never take more than a quarter. Raise - // this alongside that ceiling, never one without the other. + tracesSampleRate: 0, // errors only — spans would triple volume for signal we don't act on + // SDK default of 100 would evict half the app's own trail (`reportDemoEvent` + // also writes here, DEV-2539) — raise alongside MONITOR_BREADCRUMB_CEILING. maxBreadcrumbs: 200, + // Contract §11 / ADR §E.3: `"uncaught"` narrows to global handlers + dedupe only. + ...(SENTRY_SCOPE === "uncaught" + ? { + defaultIntegrations: false, + integrations: [Sentry.globalHandlersIntegration(), Sentry.dedupeIntegration()], + } + : {}), beforeSend(event) { if (isUnhandledNoise(event)) return null; - // DEMOS-5F, Office/Outlook safelink scanner (DEV-2858). Sits ahead of the - // DEMO_SURFACE branch, unlike isForeignUnhandled below: it requires - // `mechanism.handled === false`, and every relay arrives via - // `captureException`, which sets `handled: true` — so it cannot fire on a - // relayed event and needs no re-homing protection. + // DEMOS-5F, Office/Outlook safelink scanner (DEV-2858): requires + // `mechanism.handled === false`, which no relayed event carries. if (isOfficeScannerRejection(event)) return null; - // DEMOS-9, edgeless-foreign session-start facet (DEV-2858). Also sits ahead - // of the DEMO_SURFACE branch: it requires the `tier2-session-start` / - // `session_response_origin` tags that only `App.tsx`'s own - // `Sentry.captureException` call sets — `reportDemoEvent` (:254-260) never - // sets them, so this gate cannot fire on a relayed event either. + // DEMOS-9, edgeless-foreign session-start facet (DEV-2858): requires + // tags only App.tsx's own capture call sets, never a relay. if (isEdgelessForeignSessionStart(event)) return null; - // A client carries one `environment` from init, so a relayed demo event is - // re-homed per event here. See `reportDemoEvent`. + // A client carries one `environment` from init, so a relayed demo + // event is re-homed per event here; return before the tee below. if (event.tags?.surface === DEMO_SURFACE) { event.environment = DEMO_SURFACE; return event; } - return isForeignUnhandled(event) ? null : event; + if (isForeignUnhandled(event, window.location.origin)) return null; + // ADR §E.2 tee — best-effort, wrapped so its own failure never costs the event. + return applyFaroTee(event, telemetry); }, + }; +} + +if (reportingEnabled) { + Sentry.init({ dsn: DSN, ...sharedSentryOptions(reporting.environment) }); +} else if (localTestSentryEnabled()) { + // e2e-only: a second, mutually-exclusive `Sentry.init()` gated like Faro's + // local path. Envelopes are captured to `window.__t06SentryCapture` instead + // of sent, for `e2e/telemetry-faro.spec.ts` to read back. + Sentry.init({ + dsn: "https://t06e2e@o0.ingest.sentry.io/0", + ...sharedSentryOptions("local-test"), + transport: () => ({ + send(envelope) { + const w = window as unknown as { __t06SentryCapture?: unknown[] }; + (w.__t06SentryCapture ??= []).push(envelope); + return Promise.resolve({}); + }, + flush: () => Promise.resolve(true), + }), }); } -/** Report a caught error that would otherwise be swallowed. No-op when gated off. */ +/** + * Report a caught error that would otherwise be swallowed. Always reaches + * the facade; reaches Sentry only when `diagnosticsGoToSentry` (contract + * §11 / ADR §E.3). + */ export function reportError(error: unknown, context: string): void { - if (!reportingEnabled) return; - // A described failure the user is already being told about, and that says - // nothing about this app's health, stops here (DEV-2534). One gate, rather - // than an `if` at each of the callsites, is what retires the expired-session - // half of DEMOS-3/-6/-7/-B/-W without touching a single `catch`. Note this is - // deliberately narrow: an ownership 403 is still `reportable`, because the UI - // only offers Save and Delete on a demo it believes is the user's. + // A described failure the user is already told about says nothing about + // app health (DEV-2534): skip it for both destinations. Narrow — an + // ownership 403 is still `reportable` (Save/Delete only offered on own demos). if (error instanceof ApiError && !error.reportable) return; - Sentry.captureException(error, { tags: { context } }); + telemetry.error(error, context); + if (diagnosticsGoToSentry) { + Sentry.captureException(error, { tags: { context } }); + } } -/** - * The relay's budget, module-scoped so it lasts the page load rather than the mount — - * switching examples must not hand out a fresh allowance. The reporter's in-page copy - * of this cap is advisory: the demo it lives beside can bypass it by posting straight - * at this window (see `createMonitorBudget`). This is the enforceable one. - */ +/** The relay's budget, module-scoped so it lasts the page load, not the + * mount. The in-page copy is advisory; this is the enforceable one. */ const demoRelayBudget = createMonitorBudget(MONITOR_EVENT_CEILING); +/** A second, looser budget for warnings-as-breadcrumbs (DEV-2539) — a + * chatty demo must not spend the relay budget on warnings alone. */ +const demoBreadcrumbBudget = createMonitorBudget(MONITOR_BREADCRUMB_CEILING); + +/** One collapsed demo-runtime report, ready for the facade. */ +interface CollapsedDemoEvent { + attrs: HotAttrs; + reason: string; + fingerprint: string; + recordName: string; + shape: string; +} + /** - * A second, separate budget for the warnings that become breadcrumbs (DEV-2539). - * - * Separate in both directions. A breadcrumb files no issue, so it can be looser than - * the relay ceiling; and a demo that warns on every render must not be able to spend - * the relay budget before the `console.error` explaining the breakage arrives. Its - * `admit` also dedupes, so the repeated "Theme is already registered" notice occupies - * one breadcrumb rather than the whole buffer. + * What survives the edit-burst collapse becomes two facade calls: the + * `preview.runtime_error` count (§5) and one handled Faro exception (the + * Loki line), whose message is the §7 fingerprint shape, never the relayed + * text. */ -const demoBreadcrumbBudget = createMonitorBudget(MONITOR_BREADCRUMB_CEILING); +function emitCollapsedDemoEvent(event: CollapsedDemoEvent): void { + telemetry.metric( + "preview.runtime_error", + { count: 1 }, + { ...event.attrs, reason: event.reason, fingerprint: event.fingerprint }, + ); + const record = new Error(event.shape); + record.name = event.recordName; + record.stack = ""; + telemetry.error(record, DEMO_SURFACE, event.attrs); +} + +/** Facade demo-runtime reports go through the edit-burst collapse — one + * report per fingerprint per burst. Tier-1 compile errors share this + * instance since a non-compiling burst has no run of its own. */ +const demoEventCollapse = createDemoEventCollapse<() => void>({ + emit: (emitItem) => emitItem(), + setTimer: (fn, ms) => setTimeout(fn, ms), + clearTimer: (handle) => clearTimeout(handle as ReturnType), +}); + +/** Collapse key for a compile error — by kind, not message; prefix + * `compile:` cannot collide with a §7 `context:hash` fingerprint. */ +const COMPILE_ERROR_KEY = "compile:sandpack.compile_error"; + +/** Routes one Tier-1 compile error through the collapse with `replacesRun`, + * so a typed syntax error counts as one `sandpack.compile_error` and no + * `preview.runtime_error`. Not behind `previewMonitoring` — wired for every + * preview like `sandpack.compile_ms`. */ +export function collapseCompileError(emit: () => void, origin: "transpile" | "bundler"): void { + demoEventCollapse.report(COMPILE_ERROR_KEY, emit, { replacesRun: true, fromBundler: origin === "bundler" }); +} + +/** An edit that re-runs the preview — opens/extends the burst. Not behind + * `previewMonitoring`: with monitoring off nothing else enters the + * collapse, so this only arms a timer. */ +export function noteDemoEdit(): void { + demoEventCollapse.noteEdit(); +} + +/** The Tier-1 runtime's push outcome for the newest edit (`onPushOutcome`). */ +export function noteDemoPushOutcome(outcome: PushOutcome): void { + demoEventCollapse.pushOutcome(outcome); +} + +/** A preview is being torn down (example/version switch, remount) — + * count its last run, then let the next preview's first load count afresh. + * Ungated for the same reason as `noteDemoEdit`. */ +export function resetDemoEventCollapse(): void { + demoEventCollapse.reset(); +} + +if (typeof window !== "undefined") { + // A burst still open when the tab goes away: its last run is real. + window.addEventListener("pagehide", () => demoEventCollapse.flush()); +} /** Where a relayed event came from. `tier` distinguishes the two engines; `demoId` * is present only for a saved demo. */ export interface DemoEventContext { tier: 1 | 2; framework: string; + htMajor: HtMajor; demoId?: string | null; } /** - * File an event the preview reported through the monitor bridge (DEV-2527). - * - * Everything here crossed an origin boundary, so nothing in the payload is trusted: - * the message is re-truncated (the reporter's own cap could have been bypassed by - * anything else on the page posting the same shape) and only the fields the payload - * type declares are read. - * - * Fingerprinted by kind plus a normalised message. Without it one demo stuck in a - * throwing render shards into an issue per distinct row index — the same reasoning - * `ContainerBootFailure` already applies to boot logs in App.tsx. + * Files an event the preview reported through the monitor bridge + * (DEV-2527). Enters the edit-burst collapse unless it is a console + * warning (not a runtime error, `demoEventReport.ts`); under `full` scope + * (default) ALSO reaches Sentry; under `uncaught` scope, facade only. */ export function reportDemoEvent(payload: MonitorPayload, context: DemoEventContext): void { - if (!monitorDemos) return; - // Bound and redacted before anything else touches it — including the dedupe key - // below, which hashes the stack. An unbounded `stack` from a crafted postMessage is - // free client-side resource pressure, and a Tier-2 preview host inside it is a live - // session token. + if (!previewMonitoring) return; + reportDemoEventUnguarded(payload, context, { sentry: monitorDemos }); +} + +/** + * Body of `reportDemoEvent` without the `previewMonitoring` gate, split + * out so the e2e-only hook below can drive it directly. `opts.sentry` + * (default `true`) gates this relay's Sentry calls independently of + * `diagnosticsGoToSentry`. + */ +function reportDemoEventUnguarded( + payload: MonitorPayload, + context: DemoEventContext, + opts: { sentry: boolean } = { sentry: true }, +): void { + // Bound and redacted before anything touches it (including the dedupe + // key below) — an unbounded stack could leak a live session token. const clean = sanitizeMonitorPayload(payload); const message = clean.message; - // A warning is context, not a fault (DEV-2539). Handsontable's own "Theme is already - // registered" notice is emitted by normal re-renders, and every warning used to open - // a Sentry issue — a message event at `warning` level is still an issue. Filed as a - // breadcrumb instead, so it survives as the context attached to the next real error - // from the preview without being one itself. - // - // Before `demoRelayBudget.admit`, so a warning never consumes a relay slot, and after - // `sanitizeMonitorPayload`, so the breadcrumb is bounded and host-redacted like - // everything else that crossed the origin boundary. - // - // Breadcrumbs live on the Sentry scope, which outlives a preview: one recorded while - // example A was mounted can still be attached to an error from example B. `data` - // carries the tier, framework and demo id so a stale one is identifiable — cheaper - // and less fragile than trying to clear the buffer on every mount. + const report = demoEventReport({ + kind: clean.kind as DemoMonitorKind, + message, + tier: context.tier, + framework: context.framework, + htMajor: context.htMajor, + demoId: context.demoId, + }); + + // Into the edit-burst collapse, not straight to the facade, and before + // either Sentry budget — a keystroke ladder must not drain the relay + // budget before a later real error of the page load. + if (report.reason !== null && report.recordName !== null) { + const fp = contractFingerprint(report.fingerprintContext, report.fingerprintMessage); + const collapsed: CollapsedDemoEvent = { + attrs: report.attrs, + reason: report.reason, + fingerprint: fp, + recordName: report.recordName, + shape: fingerprintShape(report.fingerprintMessage), + }; + demoEventCollapse.report(fp, () => emitCollapsedDemoEvent(collapsed)); + } + + // A warning is context, not a fault (DEV-2539): filed as a breadcrumb, + // not an issue, before the breadcrumb budget so it never spends a relay + // slot. if (clean.kind === "console-warn") { if (!demoBreadcrumbBudget.admit(clean.kind, message)) return; - Sentry.addBreadcrumb({ - category: `${DEMO_SURFACE}.console`, - level: "warning", - message, - data: { - tier: context.tier, - framework: context.framework, - ...(context.demoId ? { demo_id: context.demoId } : {}), - }, - }); + if (opts.sentry && diagnosticsGoToSentry) { + // Breadcrumbs live on the Sentry scope, which outlives a preview. + // `data` carries the tier/framework/demo id so a stale one is identifiable. + Sentry.addBreadcrumb({ + category: `${DEMO_SURFACE}.console`, + level: "warning", + message, + data: { + tier: context.tier, + framework: context.framework, + ...(context.demoId ? { demo_id: context.demoId } : {}), + }, + }); + } return; } if (!demoRelayBudget.admit(clean.kind, message, clean.stack)) return; - // DEV-2854 / DEV-2876: a recognised Tier-2 compiler diagnostic, or a recognised Tier-2 - // build-failure envelope, collapses into its own flat, constant-titled bucket instead of - // the per-message fingerprint below. Never fed into `demoRelayBudget.admit` above — that - // stays keyed on the raw message, so 20 distinct diagnostics in one bad editing session - // still consume 20 of `MONITOR_EVENT_CEILING` rather than collapsing and losing their - // `extra` after the first. See `tier2Report.ts` for why, and for why the two shapes get - // two fingerprints rather than one. + if (!(opts.sentry && diagnosticsGoToSentry)) return; + + // DEV-2854/DEV-2876: a recognised Tier-2 diagnostic/build-failure + // envelope collapses into its own bucket instead of the per-message + // fingerprint below; see `tier2Report.ts`. const tier2 = tier2StderrReport(clean.kind, message); const tags: Record = { surface: DEMO_SURFACE, @@ -305,9 +375,8 @@ export function reportDemoEvent(payload: MonitorPayload, context: DemoEventConte : {}), }; - // An exception (with the preview's own stack) for a throw; a message for the - // kinds that never had one. A synthesised Error is how the relayed stack reaches - // Sentry's parser at all — captureMessage would drop it. + // An exception (with the preview's own stack) for a throw; a message for + // the kinds that never had one — captureMessage would drop the stack. if (clean.kind === "error" || clean.kind === "rejection") { const error = new Error(message); error.name = clean.kind === "rejection" ? "DemoUnhandledRejection" : "DemoError"; @@ -315,18 +384,9 @@ export function reportDemoEvent(payload: MonitorPayload, context: DemoEventConte Sentry.captureException(error, captureContext); return; } - // Display only, and only for network events (DEV-2539/DEMOS-12). "resource failed to - // load" as an issue title says nothing; the URL is the whole diagnosis, and `extra` - // is not visible from the issue list. Deliberately NOT used for - // `demoRelayBudget.admit` or the fingerprint above, both of which stay on the bare - // `message` — so a demo with a dozen broken assets still collapses into one issue and - // still costs one relay slot, while the title becomes actionable. - // - // This is the first path that puts an untrusted `url` into an issue *title*, and the - // bare-message fingerprint is what bounds it: the url is already capped at - // MONITOR_URL_MAX and host-redacted by `sanitizeMonitorPayload`, and because it never - // enters the fingerprint, a crafted payload posting a thousand distinct urls still - // produces one issue, titled with whichever arrived first. + // Display only, for network events (DEV-2539/DEMOS-12): the URL is the + // whole diagnosis. Kept off the fingerprint/budget key so a dozen broken + // assets still collapse into one issue. const display = tier2 ? tier2.display : clean.kind === "network" && clean.url @@ -335,4 +395,23 @@ export function reportDemoEvent(payload: MonitorPayload, context: DemoEventConte Sentry.captureMessage(display, captureContext); } +// e2e-only hooks under the `__t06ReportDemoEvent` prefix `check:telemetry-leak` +// covers: unguarded entry, guarded entry (proves the two-gate behaviour +// without a real preview mount), and the edit signal for a keystroke ladder. +if (localTestSentryEnabled()) { + ( + window as unknown as { + __t06ReportDemoEvent?: (payload: MonitorPayload, context: DemoEventContext) => void; + } + ).__t06ReportDemoEvent = reportDemoEventUnguarded; + ( + window as unknown as { + __t06ReportDemoEventGuarded?: (payload: MonitorPayload, context: DemoEventContext) => void; + } + ).__t06ReportDemoEventGuarded = reportDemoEvent; + ( + window as unknown as { __t06ReportDemoEventNoteEdit?: () => void } + ).__t06ReportDemoEventNoteEdit = noteDemoEdit; +} + export { Sentry }; diff --git a/runner/apps/authoring/src/sentryScope.ts b/runner/apps/authoring/src/sentryScope.ts new file mode 100644 index 0000000000..b82bfb24f3 --- /dev/null +++ b/runner/apps/authoring/src/sentryScope.ts @@ -0,0 +1,18 @@ +// Contract §11 / ADR §E.3 — the `VITE_SENTRY_SCOPE` switch. Import-free +// like `reportingGate.ts` so `node --test` can import it. + +export type SentryScope = "full" | "uncaught"; + +/** `import.meta.env.VITE_SENTRY_SCOPE`, resolved to the closed set. + * Anything other than `"uncaught"` stays `"full"`, the safer default. */ +export function resolveSentryScope(raw: string | undefined): SentryScope { + return raw === "uncaught" ? "uncaught" : "full"; +} + +/** Whether an explicit diagnostic report (a HANDLED condition) also reaches + * Sentry, beside the facade. Only narrows `reportingEnabled`, never widens + * it. Uncaught errors bypass this — they stay in Sentry in both scopes + * (ADR §E.1), via Sentry's own global handlers / `componentDidCatch`. */ +export function reportsDiagnosticToSentry(reportingEnabled: boolean, scope: SentryScope): boolean { + return reportingEnabled && scope === "full"; +} diff --git a/runner/apps/authoring/src/telemetry/faro.ts b/runner/apps/authoring/src/telemetry/faro.ts new file mode 100644 index 0000000000..d1b06703b8 --- /dev/null +++ b/runner/apps/authoring/src/telemetry/faro.ts @@ -0,0 +1,258 @@ +// Faro init + the contract §6 `Telemetry` facade backed by it (ADR §E.4). +// Not import-free — pulls in `@grafana/faro-web-sdk`, so no test imports +// this file directly. +// +// Errors + web-vitals instrumentations only (ADR §E.4); session tracking +// off (the facade sets `session.id` itself via `metas.add`); no tracing. +import { + ErrorsInstrumentation, + FetchTransport, + WebVitalsInstrumentation, + initializeFaro, + type BeforeSendHook, + type Faro, + type TransportItem, +} from "@grafana/faro-web-sdk"; +import { + scrubTelemetry, + fingerprint as contractFingerprint, + ATTR_HOT_SURFACE, + ATTR_HOT_TIER, + ATTR_HOT_FRAMEWORK, + ATTR_HOT_HT_MAJOR, + ATTR_HOT_OUTCOME, + ATTR_HOT_DEMO_ID, + ATTR_HOT_METRIC_KIND, + ATTR_HOT_REF, + ATTR_HOT_AREA, + ATTR_HOT_BUCKET, + ATTR_HOT_REASON, + ATTR_HOT_FINGERPRINT, + type EventName, + type HotAttrs, + type MetricName, + type MetricValues, + type ScrubbableFaroItem, + type Telemetry, +} from "@handsontable/demo-runtime/telemetry"; +import { resolveTelemetryEnabled, telemetryEnvironment } from "./gate.js"; +import { FARO_BATCHING, FARO_BUFFER_SIZE, FARO_RETRY } from "./faroConfig.js"; +import { + isForeignUnhandled, + isOfficeScannerRejection, + isUnhandledNoise, + withoutMessageEchoFrames, +} from "../eventGate.js"; + +/** + * Contract §6: `beforeSend` runs `scrubTelemetry` then the shared noise + * gates, so Faro drops the same browser noise Sentry always has. Those + * gates are Sentry-shaped (`{ exception: { values: [...] } }`); a Faro + * `ExceptionEvent` is a single flat shape, so it's adapted here rather + * than duplicated. `context.handled` mirrors Sentry's + * `mechanism.handled`: only `buildFacade().error()` sets `"true"`. + */ +function faroExceptionToExceptionShape(payload: ScrubbableFaroItem["payload"]) { + return { + exception: { + values: [ + { + value: payload.value, + type: payload.type, + mechanism: { handled: payload.context?.handled === "true" }, + stacktrace: payload.stacktrace, + }, + ], + }, + }; +} + +/** Every item passes through the contract scrubber (ADR §E.4) before an + * `exception` item runs through the shared noise gates, AFTER scrubbing. + * BEFORE scrubbing, stack frames are run through `withoutMessageEchoFrames` + * on the RAW message — `scrubTelemetry` strips the URL query/fragment, + * which would break the substring match against Faro's fake + * message-echo frame (see `eventGate.ts`) before Gate 0b sees it. */ +const beforeSend: BeforeSendHook = (item) => { + const raw = item as unknown as ScrubbableFaroItem; + const candidate: ScrubbableFaroItem = + raw.type === "exception" && raw.payload.stacktrace + ? { + ...raw, + payload: { + ...raw.payload, + stacktrace: { + ...raw.payload.stacktrace, + frames: withoutMessageEchoFrames(raw.payload.value, raw.payload.stacktrace.frames), + }, + }, + } + : raw; + const scrubbed = scrubTelemetry(candidate) as ScrubbableFaroItem | null; + if (!scrubbed) return null; + if (scrubbed.type === "exception") { + const shape = faroExceptionToExceptionShape(scrubbed.payload); + if ( + isUnhandledNoise(shape) || + isOfficeScannerRejection(shape) || + isForeignUnhandled(shape, window.location.origin) + ) { + return null; + } + } + return scrubbed as unknown as TransportItem; +}; + +/** `scrub.ts#allowlistAttributes` drops any `HotAttrs` key with no dotted + * mapping here — six map 1:1 to `hot.*` resource attrs/Loki labels; the + * rest map to AE-only `hot.*` columns, never a resource attr or Loki label + * (`attrs.ts#AE_ONLY_ATTRIBUTE_KEYS`). */ +const DOTTED_ATTR_KEY: Partial> = { + surface: ATTR_HOT_SURFACE, + tier: ATTR_HOT_TIER, + framework: ATTR_HOT_FRAMEWORK, + ht_major: ATTR_HOT_HT_MAJOR, + outcome: ATTR_HOT_OUTCOME, + demo_id: ATTR_HOT_DEMO_ID, + kind: ATTR_HOT_METRIC_KIND, + ref: ATTR_HOT_REF, + area: ATTR_HOT_AREA, + bucket: ATTR_HOT_BUCKET, + reason: ATTR_HOT_REASON, + fingerprint: ATTR_HOT_FINGERPRINT, +}; + +/** Stringify a `HotAttrs` bag for Faro's `Record` context, + * dropping `undefined` fields and remapping the six keys above to their + * dotted equivalent. `String(...)` is defensive, not a real conversion — + * every field is already a string at the type level. Parameter typed as + * `object`, not `Record`: `HotAttrs` has no + * index signature, and TS requires a source type to declare one too. */ +function attrsToContext(attrs?: object): Record | undefined { + if (!attrs) return undefined; + const out: Record = {}; + for (const [key, value] of Object.entries(attrs)) { + if (value === undefined) continue; + out[DOTTED_ATTR_KEY[key] ?? key] = String(value); + } + return Object.keys(out).length > 0 ? out : undefined; +} + +function buildFacade(faro: Faro, pageLoadId: string): Telemetry { + return { + // `skipDedupe: true` on both calls: faro-core's default GLOBAL dedupe + // keeps exactly one `lastPayload` per API and skips a push that + // deep-equals the previous one, with no time window — two identical + // `example.downloaded`/`example.open` calls back-to-back would otherwise + // silently drop the second before it leaves the browser. Server-side + // redelivery hashing already includes the client timestamp, so this + // client-side collapse buys no dedupe value here, only data loss. + metric(name: MetricName, values: MetricValues, attrs: HotAttrs) { + faro.api.pushMeasurement( + { type: name, values: { ...values } as unknown as Record }, + { context: attrsToContext(attrs), skipDedupe: true }, + ); + }, + event(name: EventName, attrs: HotAttrs & Record) { + faro.api.pushEvent(name, attrsToContext(attrs), undefined, { skipDedupe: true }); + }, + // A HANDLED error only (contract §6) — tagged `context.handled = + // "true"` so the o11y worker's ingest-time split files it as + // `error.handled`. An uncaught render crash goes through + // `reportUncaughtError` below instead (not part of `Telemetry`). + error(err: unknown, context: string, attrs?: HotAttrs) { + const error = err instanceof Error ? err : new Error(String(err)); + faro.api.pushError(error, { + context: { handled: "true", context, ...(attrsToContext(attrs) ?? {}) }, + fingerprint: contractFingerprint(context, error.message), + }); + }, + pageLoadId: () => pageLoadId, + }; +} + +export interface InitFaroOptions { + /** The page-load id already minted by `noopTelemetry` (contract facade.ts) — + * reused, not re-minted, so a call to `apiHeaders()` before `initTelemetry()` + * runs and one after both carry the identical id for the life of the page. */ + pageLoadId: string; + /** `resolveReporting(...).enabled` (`reportingGate.ts`) — the SAME + * production/automation gate Sentry uses. */ + productionReportingEnabled: boolean; + /** `import.meta.env.VITE_SENTRY_RELEASE` — the same full git SHA Sentry + * tags its events with, so a Faro `app.version` and a Sentry `release` name + * the same deploy. */ + release: string | undefined; +} + +let faroInstance: Faro | null = null; + +/** + * Initialise Faro and return the facade backed by it, or `null` when the gate + * is closed (production automation, or no local-flag/host match) — the caller + * keeps `noopTelemetry` in that case and this module stays fully inert (no + * `initializeFaro` call, no global patched, nothing to tear down). + */ +export function initFaroTelemetry(options: InitFaroOptions): Telemetry | null { + const hostname = typeof window !== "undefined" ? window.location.hostname : undefined; + const localFlag = import.meta.env.VITE_TELEMETRY_LOCAL as string | undefined; + const enabled = resolveTelemetryEnabled({ + productionReportingEnabled: options.productionReportingEnabled, + localFlag, + hostname, + }); + if (!enabled) return null; + + const environment = telemetryEnvironment(options.productionReportingEnabled); + + const faro = initializeFaro({ + transports: [ + new FetchTransport({ url: "/telemetry/collect", bufferSize: FARO_BUFFER_SIZE, retry: { ...FARO_RETRY } }), + ], + batching: { ...FARO_BATCHING }, + app: { + name: "demos-authoring", + version: options.release, + environment, + }, + sessionTracking: { enabled: false }, + instrumentations: [new ErrorsInstrumentation(), new WebVitalsInstrumentation()], + beforeSend, + }); + + // §3/§6: `session.id` = the page-load id, on every item, via a + // `metas.add` getter so it also covers `reportUncaughtError`'s direct + // `pushError` call and Faro's own instrumentation pushes. + faro.metas.add(() => ({ session: { id: options.pageLoadId } })); + + faroInstance = faro; + const impl = buildFacade(faro, options.pageLoadId); + + // An e2e-only hook, same dead-code-elimination guarantee as + // `sentry.ts`'s local-test hooks and `main.tsx`'s CrashProbe: gated on + // `VITE_TELEMETRY_LOCAL === "1"` + localhost, a literal Vite folds away + // in production (`check:telemetry-leak` greps `__t06Telemetry` as + // proof). `e2e/telemetry-faro.spec.ts` calls `event`/`metric` directly + // to prove two identical repeat pushes are NOT collapsed. + if (localFlag === "1" && typeof window !== "undefined" && (hostname === "localhost" || hostname === "127.0.0.1")) { + (window as unknown as { __t06Telemetry?: Pick }).__t06Telemetry = { + event: impl.event, + metric: impl.metric, + }; + } + + return impl; +} + +/** + * A render crash caught by `Sentry.ErrorBoundary` (ADR §E.2) — not routed + * through the facade's `Telemetry.error()`, which is contractually + * handled-only (§6). React catches this before `window.onerror` sees it, + * but it's still classified `error.uncaught` at ingest. No-op when Faro + * never initialised. + */ +export function reportUncaughtError(err: unknown): void { + if (!faroInstance) return; + const error = err instanceof Error ? err : new Error(String(err)); + faroInstance.api.pushError(error, { context: { context: "render-crash" } }); +} diff --git a/runner/apps/authoring/src/telemetry/faroConfig.ts b/runner/apps/authoring/src/telemetry/faroConfig.ts new file mode 100644 index 0000000000..6c984f3863 --- /dev/null +++ b/runner/apps/authoring/src/telemetry/faroConfig.ts @@ -0,0 +1,15 @@ +// Faro batching and delivery settings, sized against the ingest rate limit +// (`workers/o11y/wrangler.jsonc` `ratelimits`, runbook "Ingest rate limit"). +// Import-free so `pipeline/faro-config.test.mjs` can pin them. + +/** One POST per tab at most every 5 s (12/min). Faro also flushes the buffer + * on `visibilitychange` → hidden, and redelivers queued retries on `pagehide`. */ +export const FARO_BATCHING = { enabled: true, sendTimeout: 5_000, itemLimit: 50 } as const; + +/** Above the 429's `Retry-After: 60` plus Faro's 0–20 % jitter, so a 429'd batch + * waits out the limiter window instead of being dropped as "retry-after-too-long". */ +export const FARO_RETRY = { maxAttempts: 3, initialBackoffMs: 1_000, maxBackoffMs: 75_000, backoffMultiplier: 2 } as const; + +/** Batches admitted to the delivery queue, including those waiting to retry; + * Faro drops a new batch while all are taken. Bounds memory at 30 × `itemLimit`. */ +export const FARO_BUFFER_SIZE = 30; diff --git a/runner/apps/authoring/src/telemetry/gate.ts b/runner/apps/authoring/src/telemetry/gate.ts new file mode 100644 index 0000000000..2986d9b865 --- /dev/null +++ b/runner/apps/authoring/src/telemetry/gate.ts @@ -0,0 +1,38 @@ +// Contract §10 + ADR §E.4's last bullet: Faro runs when production reuses +// `resolveReporting(...).enabled` (same gate Sentry uses), or when +// `VITE_TELEMETRY_LOCAL=1` (BUILD time) AND a localhost/127.0.0.1 RUNTIME +// host — never DEV/webdriver, since Playwright serves a production build +// under automation. Import-free like `reportingGate.ts`/`eventGate.ts`. + +const LOCAL_HOSTS = new Set(["localhost", "127.0.0.1"]); + +export interface TelemetryGateInputs { + /** `resolveReporting({ dsn, hostname, webdriver }).enabled` — computed by the + * caller (`reportingGate.ts`), not recomputed here, so this module stays + * free of the `dsn`/`webdriver` inputs that decision needs. */ + productionReportingEnabled: boolean; + /** `import.meta.env.VITE_TELEMETRY_LOCAL` — a BUILD-time flag. Only the exact + * string `"1"` opens the local path; absent, empty or any other value keeps + * it closed. */ + localFlag?: string; + /** `window.location.hostname`, or undefined outside a browser. */ + hostname?: string; +} + +/** Whether Faro initialises at all. `true` on either leg — production is not + * "more true" than local; a caller that needs to know which one fired reads + * `telemetryEnvironment` instead. */ +export function resolveTelemetryEnabled({ + productionReportingEnabled, + localFlag, + hostname, +}: TelemetryGateInputs): boolean { + if (productionReportingEnabled) return true; + return localFlag === "1" && hostname !== undefined && LOCAL_HOSTS.has(hostname); +} + +/** `deployment.environment.name` (contract §3/§10): `"production"` when + * the production leg opened the gate, `"local"` when only local did. */ +export function telemetryEnvironment(productionReportingEnabled: boolean): "production" | "local" { + return productionReportingEnabled ? "production" : "local"; +} diff --git a/runner/apps/authoring/src/telemetry/index.ts b/runner/apps/authoring/src/telemetry/index.ts new file mode 100644 index 0000000000..d29bcdd392 --- /dev/null +++ b/runner/apps/authoring/src/telemetry/index.ts @@ -0,0 +1,47 @@ +// COMMON.md interface 4 — the browser facade path; callers import ONLY +// from here, never `./faro.js` directly. `telemetry` starts as +// `noopTelemetry` (mints a real page-load id) and becomes the Faro-backed +// impl once `initTelemetry()` runs, reusing the SAME id. Does NOT import +// `../sentry.js` — `resolveReporting` is called here directly on the same +// pure inputs, so the two modules have no cycle and always agree. + +import { noopTelemetry, type Telemetry } from "@handsontable/demo-runtime/telemetry"; +import { resolveReporting } from "../reportingGate.js"; +import { initFaroTelemetry, reportUncaughtError } from "./faro.js"; + +export let telemetry: Telemetry = noopTelemetry; + +/** Call once, from `main.tsx`, after `Sentry.init()` (order doesn't matter + * for correctness — this module computes its own gate). */ +export function initTelemetry(): void { + const reporting = resolveReporting({ + dsn: import.meta.env.VITE_SENTRY_DSN as string | undefined, + hostname: typeof window !== "undefined" ? window.location.hostname : undefined, + webdriver: typeof navigator !== "undefined" ? navigator.webdriver : undefined, + }); + const impl = initFaroTelemetry({ + pageLoadId: noopTelemetry.pageLoadId(), + productionReportingEnabled: reporting.enabled, + release: (import.meta.env.VITE_SENTRY_RELEASE as string | undefined) || undefined, + }); + if (impl) telemetry = impl; +} + +/** + * COMMON.md interface 4. Every API `fetch` merges these headers in, so + * `x-hot-session` rides along regardless of whether telemetry is enabled. + * `/d`/`/embed` asset requests don't call this — static fetches, not API calls. + */ +export function apiHeaders(init?: HeadersInit): Headers { + const headers = new Headers(init); + headers.set("x-hot-session", telemetry.pageLoadId()); + return headers; +} + +/** Whether the Faro-backed facade is live — the gate a browser-side event obeys, + * exposed for a count the API worker writes on the browser's behalf. */ +export function telemetryEnabled(): boolean { + return telemetry !== noopTelemetry; +} + +export { reportUncaughtError }; diff --git a/runner/apps/authoring/src/telemetry/metrics.ts b/runner/apps/authoring/src/telemetry/metrics.ts new file mode 100644 index 0000000000..a3e8567306 --- /dev/null +++ b/runner/apps/authoring/src/telemetry/metrics.ts @@ -0,0 +1,296 @@ +// Observability contract §5 browser metric catalogue (ADR-0041 §F.2). +// +// Every emission function takes an INJECTED `Telemetry` parameter (not +// imported directly) so `App.tsx` can pass its live, reassignable binding. +// Erasable TS only: `pipeline/browser-metrics.test.mjs` imports this file +// directly under `node --experimental-strip-types`. + +import type { DemoRuntime, SandpackCompileErrorEvent, SandpackCompileTimingEvent } from "@handsontable/demo-runtime"; +import { isNextPrereleaseVersion, selectedReleaseMajor } from "@handsontable/demo-runtime"; +import { fingerprint } from "@handsontable/demo-runtime/telemetry"; +import { HT_MAJORS, type HotAttrs, type HtMajor, type Surface, type Telemetry } from "@handsontable/demo-runtime/telemetry"; + +// ---- ht_major ------------------------------------------------------------------- + +/** + * Contract §3 `hot.ht_major` from a Handsontable version ref. + * + * A pkg.pr.new build ref maps to `"next"`, same as an actual `next` + * prerelease. `HT_MAJORS` (the closed set `toAePoint` enforces) has no slot for a + * build id, and both channels are equally "not a stable release" from a metrics + * point of view — inventing a value outside the closed set would throw inside + * `toAePoint` the first time anyone opened a demo pinned to a PR build. + */ +export function htMajorOf(ref: string | null | undefined): HtMajor { + if (!ref) return "none"; + if (isNextPrereleaseVersion(ref)) return "next"; + const major = selectedReleaseMajor(ref); + if (major === null) return "next"; // pkg.pr.new build ref, or unparsed + const asString = String(major); + return (HT_MAJORS as readonly string[]).includes(asString) ? (asString as HtMajor) : "none"; +} + +// ---- preview.ready_ms ------------------------------------------------------------- + +export interface PreviewResolveContext { + surface: Surface; + /** Derived from `entry.engine === "container" ? 2 : 1` (App.tsx), never + * from catalog `entry.tier` — the two disagree for UI-library starters, + * which would otherwise get the wrong ready-timeout bucket. */ + tier: 1 | 2; + framework: string; + /** The version ref the preview is being resolved against — converted to the + * closed `ht_major` set internally, so callers pass the raw ref they already + * have (`HandsontableVersionRef.ref` / `v.value.ref` in App.tsx). */ + versionRef: string; + bucket?: string; +} + +export type PreviewReadyOutcome = "ready" | "error" | "timeout" | "abandoned"; + +export interface PreviewReadyTracker { + /** + * Observe the SAME promise the caller already awaits from `mount()` — + * never call `mount()` again. Required because both engines can reject + * `mount()` without ever calling `onError` (DEV-2130); a tracker that + * only listened to ready/error would record those as `abandoned` + * instead of `error`. + */ + observe(mountPromise: Promise): void; + /** The caller is switching away before this preview settled (a version switch, + * an example switch, an unmount). No-op once ready/error/timeout already fired. */ + abandon(): void; +} + +/** Generous, tier-specific defaults: Tier-2 cold boots can take minutes + * (install + start after the create POST); Tier-1's hosted bundler has no + * such step. Neither number is measured against production traffic yet. */ +const DEFAULT_PREVIEW_TIMEOUT_MS: Record<1 | 2, number> = { + 1: 30_000, + 2: 180_000, +}; + +/** + * §5 `preview.ready_ms`, from resolve to `data-preview-status = ready`. + * Call at resolve time (before `mount()`), pass `mountPromise` to + * `observe()`, call `abandon()` from the effect's cleanup. + * + * Emits exactly once: the first of `onReady`, an observed rejection, the + * timeout, or `abandon()` wins; every later signal (including a second + * `onReady` on a clean Sandpack recompile) is a no-op. + */ +export function trackPreviewReady( + runtime: DemoRuntime, + ctx: PreviewResolveContext, + telemetry: Telemetry, + opts: { timeoutMs?: number; now?: () => number } = {}, +): PreviewReadyTracker { + const now = opts.now ?? (() => performance.now()); + const startedAt = now(); + const timeoutMs = opts.timeoutMs ?? DEFAULT_PREVIEW_TIMEOUT_MS[ctx.tier]; + const htMajor = htMajorOf(ctx.versionRef); + let settled = false; + let timer: ReturnType | null = null; + + const finish = (outcome: PreviewReadyOutcome) => { + if (settled) return; + settled = true; + if (timer !== null) clearTimeout(timer); + const attrs: HotAttrs = { + surface: ctx.surface, + tier: String(ctx.tier) as HotAttrs["tier"], + framework: ctx.framework, + ht_major: htMajor, + outcome, + }; + if (ctx.bucket !== undefined) attrs.bucket = ctx.bucket; + telemetry.metric("preview.ready_ms", { duration_ms: Math.round(now() - startedAt) }, attrs); + }; + + runtime.onReady(() => finish("ready")); + runtime.onError(() => finish("error")); + timer = setTimeout(() => finish("timeout"), timeoutMs); + + return { + observe(mountPromise) { + mountPromise.catch(() => finish("error")); + }, + abandon() { + finish("abandoned"); + }, + }; +} + +// ---- sandpack.compile_ms/compile_error/bundler_unreachable, session.start_ms, hmr.roundtrip_ms -- + +/** Quiet time after a compile before its held `sandpack.compile_ms` is sent; + * the same settle window as the edit-burst collapse (`DEMO_EDIT_SETTLE_MS`). */ +export const COMPILE_TIMING_SETTLE_MS = 2000; + +/** Senders of every runtime's held point, for the page-hide flush. */ +const heldCompileTimings = new Set<() => void>(); + +/** Sends every held `sandpack.compile_ms` now. */ +export function flushCompileTimings(): void { + for (const send of [...heldCompileTimings]) send(); +} + +if (typeof window !== "undefined") { + // Capture phase at `window` runs before Faro's own hidden-flush listener on `document`. + window.addEventListener( + "visibilitychange", + () => { + if (document.visibilityState === "hidden") flushCompileTimings(); + }, + true, + ); +} + +export interface WireRuntimeMetricsOptions { + collapseCompileError?: (emit: () => void, origin: SandpackCompileErrorEvent["origin"]) => void; + setTimer?: (fn: () => void, ms: number) => unknown; + clearTimer?: (handle: unknown) => void; +} + +/** + * Wires whichever §5 timing hooks `runtime` implements to their contract + * points, through optional chains (`runtime.onX?.(cb)`) so no engine + * branch is needed at the call site. + * + * `sandpack.compile_ms` is the settled compile of an edit burst (§5): the + * mount's compile is not sent (`preview.ready_ms` covers first load); after + * it, each compile replaces the held one and the last of a burst is sent once + * no compile or compile error follows for `COMPILE_TIMING_SETTLE_MS`. + * + * A compile error is deduped by fingerprint for the life of `runtime`, + * unless `opts.collapseCompileError` (the edit-burst collapse) is given, + * in which case it counts once per burst instead. `session.start_ms`'s + * `reason` (cold/warm) is intentionally never set — no client-observable + * signal exists to attach it from. + */ +export function wireRuntimeMetrics( + runtime: DemoRuntime, + ctx: { framework: string; versionRef: string }, + telemetry: Telemetry, + opts: WireRuntimeMetricsOptions = {}, +): void { + const htMajor = htMajorOf(ctx.versionRef); + const seenFingerprints = new Set(); + const setTimer = opts.setTimer ?? ((fn, ms) => setTimeout(fn, ms)); + const clearTimer = opts.clearTimer ?? ((handle) => clearTimeout(handle as ReturnType)); + + const sendCompileTiming = (event: SandpackCompileTimingEvent) => + telemetry.metric( + "sandpack.compile_ms", + { duration_ms: event.durationMs }, + { tier: "1", framework: ctx.framework, ht_major: htMajor, outcome: event.outcome }, + ); + /** The mount's compile has resolved, or failed before one was dispatched. */ + let mounted = false; + let held: { event: SandpackCompileTimingEvent; timer: unknown } | null = null; + const sendHeld = () => { + heldCompileTimings.delete(sendHeld); + if (!held) return; + clearTimer(held.timer); + const { event } = held; + held = null; + sendCompileTiming(event); + }; + + runtime.onCompileTiming?.((event) => { + if (!mounted) { + mounted = true; + return; + } + if (held) clearTimer(held.timer); + held = { event, timer: setTimer(sendHeld, COMPILE_TIMING_SETTLE_MS) }; + heldCompileTimings.add(sendHeld); + }); + + runtime.onCompileError?.((event) => { + mounted = true; + // A keystroke that fails the pre-transpile dispatches no compile, but it is still part of the burst. + if (held) { + clearTimer(held.timer); + held.timer = setTimer(sendHeld, COMPILE_TIMING_SETTLE_MS); + } + const fp = fingerprint("sandpack.compile_error", event.message); + const emit = () => + telemetry.metric( + "sandpack.compile_error", + {}, + { framework: ctx.framework, ht_major: htMajor, fingerprint: fp }, + ); + if (opts.collapseCompileError) { + opts.collapseCompileError(emit, event.origin); + return; + } + if (seenFingerprints.has(fp)) return; + seenFingerprints.add(fp); + emit(); + }); + + runtime.onBundlerUnreachable?.((event) => { + telemetry.metric( + "sandpack.bundler_unreachable", + { duration_ms: event.durationMs }, + { ht_major: htMajor }, + ); + }); + + runtime.onSessionStart?.((event) => { + telemetry.metric( + "session.start_ms", + { duration_ms: event.elapsedMs }, + { framework: ctx.framework, ht_major: htMajor, outcome: event.outcome }, + ); + }); + + runtime.onHmr?.((event) => { + telemetry.metric( + "hmr.roundtrip_ms", + { duration_ms: event.durationMs }, + { framework: ctx.framework, ht_major: htMajor }, + ); + }); +} + +// ---- version.switch / bucket.resolve_ms ------------------------------------------ + +/** §5 `version.switch` — call before the remount effect tears down the old + * preview. `reason` carries the FROM version's `ht_major`, not the raw ref + * (which traces back to the user-controlled `?v=` param). */ +export function emitVersionSwitch( + telemetry: Telemetry, + params: { framework: string; toRef: string; fromRef?: string | null; bucket?: string }, +): void { + const attrs: HotAttrs = { + framework: params.framework, + ht_major: htMajorOf(params.toRef), + reason: htMajorOf(params.fromRef), + }; + if (params.bucket !== undefined) attrs.bucket = params.bucket; + telemetry.metric("version.switch", {}, attrs); +} + +/** §5 `bucket.resolve_ms`. Pair with `startClock()` around the + * `resolveDocsBucket`/`resolveStarterBucket` call. */ +export function emitBucketResolve( + telemetry: Telemetry, + params: { bucket: string; outcome: "ok" | "error"; durationMs: number }, +): void { + telemetry.metric( + "bucket.resolve_ms", + { duration_ms: params.durationMs }, + { bucket: params.bucket, outcome: params.outcome }, + ); +} + +/** A tiny stopwatch: `startClock()` now, call the returned function once the + * measured work finishes to get the elapsed milliseconds (rounded). Shared by + * every duration-metric call site so none of them hand-roll `performance.now()` + * subtraction. */ +export function startClock(now: () => number = () => performance.now()): () => number { + const startedAt = now(); + return () => Math.round(now() - startedAt); +} diff --git a/runner/apps/authoring/src/tokens.ts b/runner/apps/authoring/src/tokens.ts index 7a7de305b8..4f1d7b3c88 100644 --- a/runner/apps/authoring/src/tokens.ts +++ b/runner/apps/authoring/src/tokens.ts @@ -8,6 +8,7 @@ import { assertApiOk, readApiJson } from "./api.js"; import { getToken } from "./auth.js"; +import { apiHeaders } from "./telemetry/index.js"; /** A token as the listing shows it. No digest, no plaintext — the server never * selects the former and only ever answers the latter once, on mint. */ @@ -26,9 +27,11 @@ export interface MintedToken extends ApiToken { token: string; } +/** Every one of this file's `fetch` calls is an API call, so this is also + * where `x-hot-session` (T06, `apiHeaders`) rides along. */ function authHeaders(): Record { const token = getToken(); - return token ? { Authorization: `Bearer ${token}` } : {}; + return Object.fromEntries(apiHeaders(token ? { Authorization: `Bearer ${token}` } : undefined).entries()); } const FALLBACK = (status: number) => `Request failed (${status}).`; diff --git a/runner/apps/authoring/vite.config.ts b/runner/apps/authoring/vite.config.ts index d8696e6c54..3701b8ca8a 100644 --- a/runner/apps/authoring/vite.config.ts +++ b/runner/apps/authoring/vite.config.ts @@ -29,10 +29,14 @@ export default defineConfig({ }, build: { // Only emitted when there is somewhere to upload them. A build without upload - // (local, PR CI) would otherwise leave ~12 MB of .map files in dist/ that the - // plugin's post-upload cleanup never runs to remove — and a manual - // `wrangler deploy` would publish them. - sourcemap: uploadEnabled, + // (local, PR CI) would otherwise leave ~12 MB of .map files in dist/ that would + // need their own cleanup — and a manual `wrangler deploy` would publish them. + // "hidden": the map is still built and uploaded, but no + // `//# sourceMappingURL=` comment is written — Workers Assets' SPA + // fallback (DEV-2569) answers a stray `.map` request with + // `200 text/html`, which would otherwise decode as JSON and fail. Maps + // never ship in `dist/` at all — this only removes a dead pointer. + sourcemap: uploadEnabled ? "hidden" : false, // ⚠ Do not give the @babel/standalone chunk a hash-free name (reverted from #249, // DEV-2569). The intent was sound — Workers Assets serves this app with // `not_found_handling: "single-page-application"`, so a deploy rotates the hashed chunk @@ -63,7 +67,10 @@ export default defineConfig({ authToken: process.env.SENTRY_AUTH_TOKEN, disable: !uploadEnabled, release: RELEASE ? { name: RELEASE } : undefined, - sourcemaps: { filesToDeleteAfterUpload: ["dist/**/*.map"] }, + // No `filesToDeleteAfterUpload`: the maps must stay on disk after this + // plugin's Sentry upload, because the deploy workflow's next step + // uploads the SAME files to R2 (ADR §C.3) before deleting them. + sourcemaps: {}, }), ], resolve: { @@ -103,10 +110,17 @@ export default defineConfig({ // `--routes` flags in workers/api/package.json), so the bare prefix was // always wider here than on the deployment it stands in for. `/embed` has // the same shape but nothing is named as a sibling of it today. + // The three targets below default to ":8787" but honour `API_DEV_PORT` + // so a walkthrough needing both the API worker and this proxy can run + // each on its own port block (COMMON.md) without a collision. proxy: { - "^/api(?:/|$)": { target: "http://localhost:8787" }, - "^/d(?:/|$)": { target: "http://localhost:8787" }, - "/embed": { target: "http://localhost:8787" }, + "^/api(?:/|$)": { target: `http://localhost:${process.env.API_DEV_PORT ?? "8787"}` }, + "^/d(?:/|$)": { target: `http://localhost:${process.env.API_DEV_PORT ?? "8787"}` }, + "/embed": { target: `http://localhost:${process.env.API_DEV_PORT ?? "8787"}` }, + // The o11y worker, same-origin reasoning as `/api` above (Faro posts + // to same-origin `/telemetry/collect`, contract §6). Reads + // `O11Y_DEV_PORT` (default 4200) so the two stay in sync. + "^/telemetry(?:/|$)": { target: `http://localhost:${process.env.O11Y_DEV_PORT ?? "4200"}` }, }, }, }); diff --git a/runner/containers/o11y/Dockerfile b/runner/containers/o11y/Dockerfile new file mode 100644 index 0000000000..5aa08cef5f --- /dev/null +++ b/runner/containers/o11y/Dockerfile @@ -0,0 +1,73 @@ +# containers/o11y/Dockerfile — the Grafana box image (ADR-0041 §A, §I). +# +# Runs only Loki and Grafana, under a small POSIX-shell supervisor that is +# PID 1: it is the sole recipient of container signals so it can run the +# stop protocol before anything reaches Loki or Grafana directly (no `tini`, +# no `-g`/`init: true`; the supervisor forwards SIGTERM itself, in order). +# +# Hand-written, NOT generated by scripts/prepare-container.mjs (that script +# builds the Tier-2 sandbox image from `config/frameworks.json`, an unrelated +# concern). See containers/live/Dockerfile for the repo's other convention: +# pinned base image, ordered COPY+RUN layers, EXPOSE at the end. +# +# Base: grafana/grafana (Alpine — bash, curl with --aws-sigv4 already +# present, so no extra R2-signing binary is needed for the shutdown script). +# The Loki binary is copied in from the official Loki image; Loki's own +# image is `FROM distroless` variant with no shell, so it cannot host the +# supervisor. + +FROM grafana/loki:3.3.2 AS loki + +FROM grafana/grafana:11.4.0 + +USER root + +# The Altinity ClickHouse plugin (ADR-0041 §A), NOT grafana/clickhouse- +# datasource: Cloudflare's own Analytics Engine + Grafana docs +# (developers.cloudflare.com/analytics/analytics-engine/grafana/) point at +# `vertamedia-clickhouse-datasource` specifically — the AE SQL API answers a +# ClickHouse-HTTP-shaped endpoint authenticated with a single custom +# `Authorization: Bearer ` header, which is this plugin's supported +# shape (grafana/clickhouse-datasource expects host/port/user/password and +# a native or different HTTP auth handshake instead). Pinned so a catalog +# change never silently rewrites what runs in production; community-signed +# (ships its own MANIFEST.txt), no allow-unsigned flag needed. +ARG CLICKHOUSE_PLUGIN_VERSION=3.5.0 +RUN grafana cli --pluginsDir /var/lib/grafana/plugins plugins install \ + vertamedia-clickhouse-datasource ${CLICKHOUSE_PLUGIN_VERSION} + +COPY --from=loki /usr/bin/loki /usr/bin/loki + +RUN mkdir -p \ + /loki/wal /loki/chunks /loki/tsdb-index /loki/tsdb-cache \ + /loki/compactor /loki/rules /loki/rules-temp \ + && chown -R grafana:root /loki +# Deliberately NOT chown -R'd on /var/lib/grafana: that would copy-on-write +# the whole plugin tree a second time (it roughly doubled the image size in +# an earlier build — measured 865 MB -> 1.22 GB uncompressed for this one +# `chown`). The plugin's own files come out of `grafana cli ... install` +# world-readable/executable (0644/0755) even when the install RUN runs as +# root, so the non-root `grafana` user (472:0) can load them without it — +# verified by booting the image and hitting /grafana/api/health after +# switching to USER grafana below. + +COPY loki/loki-config.yaml /etc/loki/loki-config.yaml +COPY loki/loki-config.filesystem.yaml /etc/loki/loki-config.filesystem.yaml +COPY loki/runtime-config.yaml /etc/loki/runtime-config.yaml +COPY grafana/grafana.ini /etc/grafana/grafana.ini +COPY grafana/provisioning /etc/grafana/provisioning +COPY grafana/dashboards /etc/grafana/dashboards + +COPY supervisor/entrypoint.sh /entrypoint.sh +COPY supervisor/shutdown.sh /shutdown.sh +COPY supervisor/lib.sh /lib.sh +RUN chmod 0755 /entrypoint.sh /shutdown.sh /lib.sh + +# Ports are reached only through GrafanaBox.containerFetch (ADR-0041 §A) — +# the container itself is never publicly routable. See docs/observability- +# contract.md §1 for the port table. +EXPOSE 3000 3100 + +USER grafana + +ENTRYPOINT ["/entrypoint.sh"] diff --git a/runner/containers/o11y/compose.yml b/runner/containers/o11y/compose.yml new file mode 100644 index 0000000000..2ea939ea11 --- /dev/null +++ b/runner/containers/o11y/compose.yml @@ -0,0 +1,123 @@ +# containers/o11y/compose.yml — local stand-in for the Grafana box (ADR-0041 +# §I, docs/observability-contract.md §10). No `container_name`/`name:` — +# compose's automatic `_` naming keeps every project's +# resources distinct, so multiple worktrees run this under their own +# `COMPOSE_PROJECT_NAME`. +# +# `minio-data`/`clickhouse-data` are NAMED volumes (kept across `down`, +# unlike `-v`); the o11y worker's own Durable Object state persists too, so +# `dev.mjs --fresh` wipes both together (`resetO11yLocalState`) to avoid +# divergence. `box` itself is ephemeral — nothing in it survives a restart. +# +# Run with COMPOSE_PROJECT_NAME + O11Y_*_PORT env vars (each worktree uses +# its own block); unset, every port defaults to the contract's own values. + +services: + box: + build: + context: . + environment: + # No default: an empty WAKE_ID makes the supervisor REFUSE to write a + # marker. Every caller must set O11Y_WAKE_ID explicitly. + WAKE_ID: ${O11Y_WAKE_ID:-} + STORAGE: s3 + LOKI_S3_ENDPOINT: minio:9000 + LOKI_S3_REGION: auto + # Independent from MinIO's own root credentials: the C1 negative + # control (stop-roundtrip.mjs) runs the box against a restricted MinIO + # user. Defaults match MinIO's own default root credentials. + LOKI_S3_ACCESS_KEY_ID: ${O11Y_LOKI_S3_ACCESS_KEY_ID:-minioadmin} + LOKI_S3_SECRET_ACCESS_KEY: ${O11Y_LOKI_S3_SECRET_ACCESS_KEY:-minioadmin} + LOKI_S3_BUCKET: loki + LOKI_S3_INSECURE: "true" + GF_SERVER_ROOT_URL: http://localhost:${O11Y_GRAFANA_PORT:-3000}/grafana/ + # vertamedia-clickhouse-datasource needs a full URL + ONE header pair + # (X-ClickHouse-User/-Key), unlike production's single Bearer header. + # No standalone AE_SQL_TOKEN var — it reaches Grafana as the VALUE of + # O11Y_CLICKHOUSE_HEADER2_VALUE below. + O11Y_CLICKHOUSE_URL: http://clickhouse:8123 + O11Y_CLICKHOUSE_DATABASE: default + O11Y_CLICKHOUSE_HEADER1_NAME: X-ClickHouse-User + O11Y_CLICKHOUSE_HEADER1_VALUE: default + O11Y_CLICKHOUSE_HEADER2_NAME: X-ClickHouse-Key + O11Y_CLICKHOUSE_HEADER2_VALUE: ${AE_SQL_TOKEN:-local-dev-token} + O11Y_STOP_GRACE_SECONDS: ${O11Y_STOP_GRACE_SECONDS:-30} + ports: + - "127.0.0.1:${O11Y_GRAFANA_PORT:-3000}:3000" + - "127.0.0.1:${O11Y_LOKI_PORT:-3100}:3100" + # Must exceed O11Y_STOP_GRACE_SECONDS so the stop protocol's remaining + # steps can run before Docker SIGKILLs; raise both together, by hand. + # Default 30s was measured (~2.3s) against short local test streams — + # not yet validated against production traffic (ADR-0041 exit criterion 7). + stop_grace_period: 45s + # `standard-1` (docs.cloudflare.com/containers/platform-details/limits): + # 1/2 vCPU, 4 GiB memory, 8 GB disk — so local boot-time and memory + # measurements mean the same thing the sandbox probe will measure. + deploy: + resources: + limits: + cpus: "0.5" + memory: 4g + depends_on: + minio: + condition: service_healthy + clickhouse: + condition: service_started + + # Upstream MinIO stopped distributing public community images (2026-09). + # Replaced with Bitnami's frozen "legacy" mirror of the same release line, + # pinned by DIGEST; verified end to end against the C1 negative control + # below (a real denied PUT under a deny-`s3:PutObject` policy). + minio: + image: bitnamilegacy/minio@sha256:81cd091fb9f14b2e9e9bfa6dbc2bf2d46fdd5eafa6c5e7c9213baf4256ff13d6 # 2024.11.7, verified linux/amd64+arm64 (`docker buildx imagetools inspect`) + # A data service (unlike `box`): if Docker OOM-kills minio/clickhouse, + # every write after that point is silently dropped until restarted by + # hand (observed for real under Docker Desktop's default memory limit). + restart: unless-stopped + environment: + MINIO_ROOT_USER: ${O11Y_MINIO_ROOT_USER:-minioadmin} + MINIO_ROOT_PASSWORD: ${O11Y_MINIO_ROOT_PASSWORD:-minioadmin} + # Replaces the old `minio-init` one-shot container (quay.io/minio/mc + # is also gone) — Bitnami's own entrypoint creates the bucket itself + # before MinIO starts listening; the healthcheck confirms it on disk. + MINIO_DEFAULT_BUCKETS: loki + volumes: + # Bitnami's image data dir differs from upstream minio/minio's `/data` + # (`docker inspect` on the image: `Volumes: {"/bitnami/minio/data"}`). + - minio-data:/bitnami/minio/data + ports: + - "127.0.0.1:${O11Y_MINIO_PORT:-9000}:9000" + - "127.0.0.1:${O11Y_MINIO_CONSOLE_PORT:-9001}:9001" + healthcheck: + # Checks the bucket dir AND the live endpoint: Bitnami's entrypoint + # briefly runs a throwaway MinIO instance first, which a bare + # `/minio/health/live` check could latch "healthy" against. + test: ["CMD-SHELL", "test -d /bitnami/minio/data/loki && curl -f http://localhost:9000/minio/health/live"] + interval: 2s + timeout: 2s + retries: 30 + + # Analytics Engine stand-in (§10): a `runner_events` table shaped like the + # real AE SQL API response, queried through the same allowlisting helper. + clickhouse: + image: clickhouse/clickhouse-server:24.10-alpine + # Data service — see minio's own comment above (this is the exact + # container observed OOM-killed under Docker Desktop's default limit). + restart: unless-stopped + environment: + CLICKHOUSE_DB: default + # The Grafana datasource always sends a password; give the local user + # the same value so the health check exercises real auth. + CLICKHOUSE_PASSWORD: ${AE_SQL_TOKEN:-local-dev-token} + volumes: + - ./local/clickhouse-init.sql:/docker-entrypoint-initdb.d/clickhouse-init.sql:ro + # Only runs on a genuinely empty data dir; a persisted volume already + # has `runner_events` and this is skipped, which is the point. + - clickhouse-data:/var/lib/clickhouse + ports: + - "127.0.0.1:${O11Y_CLICKHOUSE_PORT:-8123}:8123" + - "127.0.0.1:${O11Y_CLICKHOUSE_NATIVE_PORT:-9009}:9000" + +volumes: + minio-data: + clickhouse-data: diff --git a/runner/containers/o11y/grafana/dashboards/ai-assist.json b/runner/containers/o11y/grafana/dashboards/ai-assist.json new file mode 100644 index 0000000000..21d4463b5f --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/ai-assist.json @@ -0,0 +1,466 @@ +{ + "uid": "o11y-ai-assist", + "title": "AI assist", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "chat.answer rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'chat.answer' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "chat.answer p95 duration by model", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob13 AS model, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'chat.answer' AND blob3 = '$environment' GROUP BY t, model ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "chat.answer cost (USD) by model", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob13 AS model, sum(_sample_interval * double4) AS usd FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'chat.answer' AND blob3 = '$environment' GROUP BY t, model ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "currencyUSD", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "timeseries", + "title": "chat.edit funnel (proposed/applied/undone)", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'chat.edit' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 5, + "type": "timeseries", + "title": "theme.ai rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'theme.ai' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 6, + "type": "timeseries", + "title": "import.url rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'import.url' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 7, + "type": "timeseries", + "title": "payload.boot rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'payload.boot' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/docs-embeds.json b/runner/containers/o11y/grafana/dashboards/docs-embeds.json new file mode 100644 index 0000000000..2d89a0751c --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/docs-embeds.json @@ -0,0 +1,336 @@ +{ + "uid": "o11y-docs-embeds", + "title": "Docs embeds", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "serve.share / serve.d / serve.embed rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 IN ('serve.share','serve.d','serve.embed') AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "snapshot.build rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'snapshot.build' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "web_vital by vital (embed/d surfaces)", + "description": "LCP, INP, CLS, TTFB — weighted quantile of double3 (value).", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS vital, quantileExactWeighted(0.75)(double3, toUInt32(_sample_interval)) AS p75 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'web_vital' AND blob4 IN ('embed','d') AND blob3 = '$environment' GROUP BY t, vital ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "table", + "title": "Broken embeds by demo id (error.uncaught, surface=embed)", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob12 AS demo_id, sum(_sample_interval * double1) AS errors FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'error.uncaught' AND blob4 = 'embed' AND blob3 = '$environment' GROUP BY demo_id ORDER BY errors DESC LIMIT 20", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 5, + "type": "logs", + "title": "Recent embed errors", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "expr": "{hot_surface=\"embed\"}", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/examples.json b/runner/containers/o11y/grafana/dashboards/examples.json new file mode 100644 index 0000000000..f0b748e83b --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/examples.json @@ -0,0 +1,415 @@ +{ + "uid": "o11y-examples", + "title": "Examples & features", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-90d", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "table", + "title": "Top guides by opens and engaged opens", + "description": "ADR-0042 — ranks docs guides (blob18) by example.open count; example.engaged sits alongside it so a guide with high opens but low engagement stands out.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob18 AS guide, sum(_sample_interval * double1 * (index1 = 'example.open')) AS opens, sum(_sample_interval * double1 * (index1 = 'example.engaged')) AS engaged FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 IN ('example.open', 'example.engaged') AND blob17 = 'docs' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY guide ORDER BY opens DESC LIMIT 20", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 2, + "type": "timeseries", + "title": "Docs opens by area", + "description": "example.open, kind=docs, grouped by area (blob19 — a resolved example's first breadcrumb element).", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob19 AS area, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'example.open' AND blob17 = 'docs' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, area ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "table", + "title": "Framework split by guide", + "description": "example.open, kind=docs, guide x framework — deliberately NOT filtered by $framework, since this panel exists to show the split that filter would hide.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob18 AS guide, blob6 AS framework, sum(_sample_interval * double1) AS opens FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'example.open' AND blob17 = 'docs' AND blob3 = '$environment' AND blob7 IN (${ht_major:sqlstring}) GROUP BY guide, framework ORDER BY opens DESC LIMIT 50", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 4, + "type": "table", + "title": "Starter ranking", + "description": "example.open, kind=starter — ref IS the framework id for a starter, so this panel is not filtered by $framework either.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob18 AS starter, sum(_sample_interval * double1) AS opens FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'example.open' AND blob17 = 'starter' AND blob3 = '$environment' AND blob7 IN (${ht_major:sqlstring}) GROUP BY starter ORDER BY opens DESC", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 5, + "type": "timeseries", + "title": "example.open ht_major distribution", + "description": "Every kind, grouped by ht_major — not filtered by $ht_major (this panel IS the breakdown).", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob7 AS ht_major, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'example.open' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) GROUP BY t, ht_major ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 6, + "type": "table", + "title": "Funnel: open → engaged → saved → shared → forked → downloaded, by area (docs)", + "description": "ADR-0042 §2's engagement events, joined by area via the boolean-multiply sum trick (version-health.json's own convention) rather than a second panel per metric. forked/downloaded added — controller audit finding: emitted but previously unread.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob19 AS area, sum(_sample_interval * double1 * (index1 = 'example.open')) AS opens, sum(_sample_interval * double1 * (index1 = 'example.engaged')) AS engaged, sum(_sample_interval * double1 * (index1 = 'example.saved')) AS saved, sum(_sample_interval * double1 * (index1 = 'example.shared')) AS shared, sum(_sample_interval * double1 * (index1 = 'example.forked')) AS forked, sum(_sample_interval * double1 * (index1 = 'example.downloaded')) AS downloaded FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 IN ('example.open', 'example.engaged', 'example.saved', 'example.shared', 'example.forked', 'example.downloaded') AND blob17 = 'docs' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY area ORDER BY opens DESC", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 7, + "type": "timeseries", + "title": "Deep-link share of opens", + "description": "% of example.open events whose entry/reason (blob9) is deep-link — how much of the traffic arrives already pointed at a specific example vs. via the picker/switch.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 32 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, 100 * sum(_sample_interval * double1 * (blob9 = 'deep-link')) / sum(_sample_interval * double1) AS deep_link_pct FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'example.open' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "max": 100, + "min": 0, + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/logs.json b/runner/containers/o11y/grafana/dashboards/logs.json new file mode 100644 index 0000000000..c53404d05c --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/logs.json @@ -0,0 +1,537 @@ +{ + "uid": "o11y-logs", + "title": "Logs", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "links": [ + { + "asDropdown": false, + "icon": "external link", + "includeVars": false, + "keepTime": true, + "tags": [], + "targetBlank": false, + "title": "Runner overview", + "type": "link", + "url": "/d/o11y-runner-overview/runner-overview" + }, + { + "asDropdown": false, + "icon": "external link", + "includeVars": false, + "keepTime": true, + "tags": [], + "targetBlank": false, + "title": "Observability self", + "type": "link", + "url": "/d/o11y-observability-self/observability-self" + } + ], + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "allValue": ".*", + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + }, + { + "name": "tenant", + "type": "datasource", + "label": "Tenant", + "query": "loki", + "regex": "", + "refresh": 1, + "multi": false, + "includeAll": false, + "hide": 0, + "current": { + "text": "Loki (worker)", + "value": "loki-worker" + } + }, + { + "name": "service_name", + "type": "query", + "label": "Service", + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "query": "label_values(service_name)", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "hot_surface", + "type": "query", + "label": "Surface", + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "query": "label_values(hot_surface)", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "hot_outcome", + "type": "query", + "label": "Outcome", + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "query": "label_values(hot_outcome)", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "search", + "type": "textbox", + "label": "Search (free text)", + "description": "A LogQL line filter (|= \"$search\"). Empty matches every line — an empty substring is contained in every string.", + "query": "", + "current": { + "text": "", + "value": "" + }, + "hide": 0 + }, + { + "name": "correlation", + "type": "textbox", + "label": "Correlation (cf.ray / session.id / demo id)", + "description": "Matched as a LogQL structured-metadata regex filter against cf_ray, session_id and hot_demo_id (never as a stream label — those are unbounded-cardinality values, ADR §B.4). Default \".*\" matches everything, including records with no value for a given field (Loki reads an absent structured-metadata field as \"\", which \".*\" also matches). Paste an exact id to narrow to one request/session/demo.", + "query": ".*", + "current": { + "text": ".*", + "value": ".*" + }, + "hide": 0 + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + }, + { + "name": "Deploys", + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "enable": true, + "iconColor": "blue", + "expr": "{hot_surface=\"o11y\"} |= `\"event\":\"deploy\"`", + "titleFormat": "Deploy", + "textFormat": "{{__line__}}" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "Log volume by service", + "description": "count_over_time across every service on the selected tenant, filtered by every dropdown variable, the free-text search and the correlation id.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "expr": "sum by (service_name) (count_over_time({service_name=~\"${service_name:regex}\", hot_surface=~\"${hot_surface:regex}\", hot_outcome=~\"${hot_outcome:regex}\", hot_framework=~\"${framework:regex}\", deployment_environment_name=\"$environment\"} |= \"$search\" | cf_ray=~\"$correlation\" or session_id=~\"$correlation\" or hot_demo_id=~\"$correlation\" [$__auto]))", + "queryType": "range", + "legendFormat": "{{service_name}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "bars", + "fillOpacity": 30, + "stacking": { + "mode": "normal" + } + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "logs", + "title": "API worker errors", + "description": "Loki worker tenant, service_name=demos-api, filtered to the API worker's own structured error line (workers/api/src/telemetry/lines.ts#logErrorLine: `console.error(JSON.stringify({\"log.kind\":\"error\",...}))`). Cloudflare's real OTLP export ships that whole console.log call as opaque BODY TEXT (never parsed into attributes, and severityNumber is dropped by this repo's own pipeline before Loki ever sees it — `detected_level` is unusable here), so the line filter below matches the literal body shape instead. Verified against a re-timestamped copy of the real captured console-log-line.json fixture with a log.kind=error body.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "expr": "{service_name=\"demos-api\"} |= `\"log.kind\":\"error\"`", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 3, + "type": "logs", + "title": "Authoring errors", + "description": "Loki browser tenant, hot_surface=authoring, `hot_kind=\"exception\"` structured-metadata filter (never a label — unbounded per-report cardinality) — Faro exceptions from the authoring app. Verified against a replayed exception fixture.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "expr": "{hot_surface=\"authoring\"} | hot_kind=\"exception\"", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 4, + "type": "logs", + "title": "Recent embed errors", + "description": "Loki browser tenant, hot_surface=embed — same query as the docs-embeds dashboard's own panel.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "expr": "{hot_surface=\"embed\"}", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 5, + "type": "logs", + "title": "Recent demo-runtime errors", + "description": "Loki browser tenant, hot_surface=demo-runtime exceptions: one line per collapsed preview error (F26/F10: one per fingerprint per edit burst, text = the §7 fingerprint shape, never the raw message) plus the Tier-1 compile branch's constant-titled report. Same query as the Tier-1 playground dashboard's own panel.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "expr": "{hot_surface=\"demo-runtime\"} | hot_kind=\"exception\"", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 7, + "type": "logs", + "title": "Sentry issues", + "description": "Loki worker tenant, service_name=demos-api / hot_surface=api, the issue-alert webhook's own line shape (normalise/sentry.ts: `sentry : [<id>] release=<release> <link>`). Controller audit finding: emitted but previously unread — verified against a replayed pipeline/fixtures/otlp/sentry-issue.json fixture.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 34 + }, + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "expr": "{service_name=\"demos-api\", hot_surface=\"api\"} |= \"sentry \"", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 8, + "type": "table", + "title": "Top error fingerprints by report count", + "description": "error.handled + error.uncaught grouped by fingerprint (blob11), highest count first — an approximation of \"new error fingerprints\": Grafana cannot read the o11y worker's own Durable Object fp: registry (the real first-seen source of truth, ADR §... T04), and Analytics Engine's SQL API only allowlists sum/avg/quantileExactWeighted/toStartOfInterval/toUInt32/now (workers/o11y/src/alerts/ae-query.ts#ALLOWED_AE_FUNCTIONS) — no COUNT(DISTINCT ...)/uniqExact, so a true first-seen-in-range query is not expressible here. Controller audit finding: emitted but previously unread.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 34 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob11 AS fingerprint, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 IN ('error.handled', 'error.uncaught') AND blob3 = '$environment' GROUP BY fingerprint ORDER BY cnt DESC LIMIT 20", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 6, + "type": "logs", + "title": "All matching lines", + "description": "Every variable on this dashboard applies: tenant, service, surface, outcome, framework, environment, free-text search and the correlation id (cf.ray / session.id / demo id, matched as structured metadata).", + "gridPos": { + "h": 10, + "w": 24, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "${tenant}" + }, + "expr": "{service_name=~\"${service_name:regex}\", hot_surface=~\"${hot_surface:regex}\", hot_outcome=~\"${hot_outcome:regex}\", hot_framework=~\"${framework:regex}\", deployment_environment_name=\"$environment\"} |= \"$search\" | cf_ray=~\"$correlation\" or session_id=~\"$correlation\" or hot_demo_id=~\"$correlation\"", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/observability-self.json b/runner/containers/o11y/grafana/dashboards/observability-self.json new file mode 100644 index 0000000000..4625f89f1c --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/observability-self.json @@ -0,0 +1,513 @@ +{ + "uid": "o11y-observability-self", + "title": "Observability self", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "links": [ + { + "asDropdown": false, + "icon": "external link", + "includeVars": false, + "keepTime": true, + "tags": [], + "targetBlank": false, + "title": "Logs", + "type": "link", + "url": "/d/o11y-logs/logs" + } + ], + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "o11y.ingest rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'o11y.ingest' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "o11y.drain duration p95 by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'o11y.drain' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "o11y.wake rate by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'o11y.wake' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "timeseries", + "title": "o11y.backlog oldest age (s)", + "description": "A cron gauge — sampling-correct average, not the count-sampling formula.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, sum(_sample_interval * double3) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'o11y.backlog' AND blob3 = '$environment' GROUP BY t ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "s", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 5, + "type": "timeseries", + "title": "o11y.alert fired/resolved", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'o11y.alert' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 6, + "type": "timeseries", + "title": "reconcile.run rate by outcome", + "description": "API worker cron — reconcile.run count, by outcome (ok/skipped/error).", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'reconcile.run' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 7, + "type": "timeseries", + "title": "reconcile.run usd (billing total written) by outcome", + "description": "reconcile.run double4 (§4) summed per bucket, by outcome — the total billing usd WRITTEN this run, NOT billing-minus-estimate (D-M15 fix round; contract §5, reconcile.ts#billingUsdWritten).", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double4) AS usd_written FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'reconcile.run' AND blob3 = '$environment' GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "currencyUSD", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 8, + "type": "logs", + "title": "o11y worker log stream", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "expr": "{service_name=\"demos-o11y\"}", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/runner-overview.json b/runner/containers/o11y/grafana/dashboards/runner-overview.json new file mode 100644 index 0000000000..d5e9a9079e --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/runner-overview.json @@ -0,0 +1,655 @@ +{ + "uid": "o11y-runner-overview", + "title": "Runner overview", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "links": [ + { + "asDropdown": false, + "icon": "external link", + "includeVars": false, + "keepTime": true, + "tags": [], + "targetBlank": false, + "title": "Logs", + "type": "link", + "url": "/d/o11y-logs/logs" + } + ], + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + }, + { + "name": "Deploys", + "datasource": { + "type": "loki", + "uid": "loki-worker" + }, + "enable": true, + "iconColor": "blue", + "expr": "{hot_surface=\"o11y\"} |= `\"event\":\"deploy\"`", + "titleFormat": "Deploy", + "textFormat": "{{__line__}}" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "Preview-ready rate by tier", + "description": "preview.ready_ms — share of previews that reached outcome=ready, by hot.tier.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob5 AS tier, 100 * sum(_sample_interval * double1 * (blob8 = 'ready')) / sum(_sample_interval * double1) AS pct FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'preview.ready_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, tier ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "Preview-ready p95 by tier", + "description": "preview.ready_ms duration_ms, weighted quantile, by hot.tier.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob5 AS tier, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'preview.ready_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, tier ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "Session start rate by outcome", + "description": "session.start (API worker) count by outcome.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'session.start' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "timeseries", + "title": "Error report rate (uncaught + handled)", + "description": "error.uncaught and error.handled counts. Excludes demo-runtime errors (errors inside a user's preview), which the Tier-1 dashboard shows.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, index1 AS metric, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 IN ('error.uncaught','error.handled') AND blob3 = '$environment' AND blob4 != 'demo-runtime' GROUP BY t, metric ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 5, + "type": "timeseries", + "title": "Pool gauge vs cap", + "description": "pool.gauge — awake containers (double3) against the cap (double8), by reason (live/builder). Sampling-correct average, not sum.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, sum(_sample_interval * double3) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'pool.gauge' AND blob3 = '$environment' GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 6, + "type": "timeseries", + "title": "Pool cap", + "description": "pool.gauge — the configured cap (double8), by reason.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, sum(_sample_interval * double8) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'pool.gauge' AND blob3 = '$environment' GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 7, + "type": "timeseries", + "title": "api.request 5xx rate by route_class", + "description": "api.request count where outcome=5xx, by route_class.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob10 AS route_class, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'api.request' AND blob3 = '$environment' AND blob8 = '5xx' GROUP BY t, route_class ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 8, + "type": "timeseries", + "title": "api.request p95 duration_ms by route_class", + "description": "api.request duration_ms, weighted quantile, by route_class.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob10 AS route_class, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'api.request' AND blob3 = '$environment' GROUP BY t, route_class ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 9, + "type": "timeseries", + "title": "Budget spend vs ceiling by tier", + "description": "budget.gauge (API worker */5 cron) — value (double3) is ALREADY spend as a percent of the ceiling, by tier (blob9=reason). Sampling-correct average, not sum. No separate cap/ceiling column is emitted for this metric (unlike pool.gauge's double8) — the fixed 100 threshold below IS the cap line: it marks the ceiling this percentage is measured against, since 100% always means \"at the ceiling\" by construction.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS tier, sum(_sample_interval * double3) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'budget.gauge' AND blob3 = '$environment' GROUP BY t, tier ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + }, + "thresholds": { + "mode": "absolute", + "steps": [ + { "color": "green", "value": null }, + { "color": "red", "value": 100 } + ] + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 10, + "type": "timeseries", + "title": "Budget usd spent by tier", + "description": "budget.gauge (API worker */5 cron) — usd (double4), by tier. Sampling-correct average of the current spend snapshot, not sum (this is a gauge reading, not a per-interval delta).", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS tier, sum(_sample_interval * double4) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'budget.gauge' AND blob3 = '$environment' GROUP BY t, tier ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "currencyUSD", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/tier1-playground.json b/runner/containers/o11y/grafana/dashboards/tier1-playground.json new file mode 100644 index 0000000000..7816579dc4 --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/tier1-playground.json @@ -0,0 +1,501 @@ +{ + "uid": "o11y-tier1-playground", + "title": "Tier-1 playground", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "sandpack.compile_ms p95 by framework", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob6 AS framework, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'sandpack.compile_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, framework ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "sandpack.compile_error rate by framework", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob6 AS framework, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'sandpack.compile_error' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, framework ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "sandpack.bundler_unreachable", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'sandpack.bundler_unreachable' AND blob3 = '$environment' AND blob7 IN (${ht_major:sqlstring}) GROUP BY t ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "timeseries", + "title": "preview.runtime_error rate by reason", + "description": "surface=demo-runtime only (§5); reason: uncaught, console, network, stderr. Ladder-deduped by fingerprint upstream (§7).", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'preview.runtime_error' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 5, + "type": "timeseries", + "title": "hmr.roundtrip_ms p95", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'hmr.roundtrip_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 6, + "type": "logs", + "title": "Recent demo-runtime errors", + "description": "Loki browser tenant, hot_surface=demo-runtime exceptions: one line per collapsed preview error (F26/F10: one per fingerprint per edit burst, text = the §7 fingerprint shape, never the raw message) plus the Tier-1 compile branch's constant-titled report. The text behind the fingerprint AE carries.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "loki", + "uid": "loki-browser" + }, + "expr": "{hot_surface=\"demo-runtime\"} | hot_kind=\"exception\"", + "queryType": "range" + } + ], + "options": { + "showTime": true, + "wrapLogMessage": true, + "sortOrder": "Descending" + } + }, + { + "id": 7, + "type": "timeseries", + "title": "bucket.resolve_ms p95 by bucket", + "description": "F24 fix round: not filtered by framework/ht_major — the §5 registry row for `bucket.resolve_ms` lists only `bucket, outcome` (the docs-bucket resolve fires before a framework is chosen; the starter-bucket resolve fires before it, too), so those columns are always empty on this metric and a filter on them matched zero rows.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob16 AS bucket, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'bucket.resolve_ms' AND blob3 = '$environment' GROUP BY t, bucket ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 8, + "type": "timeseries", + "title": "bucket.resolve_ms error rate by bucket", + "description": "F24 fix round: not filtered by framework/ht_major — the §5 registry row for `bucket.resolve_ms` lists only `bucket, outcome` (the docs-bucket resolve fires before a framework is chosen; the starter-bucket resolve fires before it, too), so those columns are always empty on this metric and a filter on them matched zero rows.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob16 AS bucket, 100 * sum(_sample_interval * double1 * (blob8 = 'error')) / sum(_sample_interval * double1) AS pct FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'bucket.resolve_ms' AND blob3 = '$environment' GROUP BY t, bucket ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/tier2-sessions.json b/runner/containers/o11y/grafana/dashboards/tier2-sessions.json new file mode 100644 index 0000000000..0897e96714 --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/tier2-sessions.json @@ -0,0 +1,383 @@ +{ + "uid": "o11y-tier2-sessions", + "title": "Tier-2 sessions", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "timeseries", + "title": "session.start rate by outcome", + "description": "ready, at_capacity, container_starting, boot_timeout, budget_denied, error.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'session.start' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 2, + "type": "timeseries", + "title": "session.start_ms p95 by reason (cold/warm)", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'session.start_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "container.boot_ms p95 by outcome", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob8 AS outcome, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'container.boot_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) GROUP BY t, outcome ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 4, + "type": "timeseries", + "title": "session.end awake seconds, p95 by reason", + "description": "pagehide, sleep_after, teardown_failed, budget_closed.", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, quantileExactWeighted(0.95)(double3, toUInt32(_sample_interval)) AS p95 FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'session.end' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "s", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 5, + "type": "timeseries", + "title": "Pool gauge (live/builder) vs cap", + "description": "pool.gauge — both series in one panel: double3 (awake, by reason) and double8 (cap, overall) — I3.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob9 AS reason, sum(_sample_interval * double3) / sum(_sample_interval) AS avg_value FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'pool.gauge' AND blob3 = '$environment' GROUP BY t, reason ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + }, + { + "refId": "B", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, sum(_sample_interval * double8) / sum(_sample_interval) AS cap FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'pool.gauge' AND blob3 = '$environment' GROUP BY t ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/dashboards/version-health.json b/runner/containers/o11y/grafana/dashboards/version-health.json new file mode 100644 index 0000000000..0ac203351b --- /dev/null +++ b/runner/containers/o11y/grafana/dashboards/version-health.json @@ -0,0 +1,297 @@ +{ + "uid": "o11y-version-health", + "title": "Version health", + "tags": [ + "o11y", + "runner" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "editable": false, + "time": { + "from": "now-6h", + "to": "now" + }, + "refresh": "", + "templating": { + "list": [ + { + "name": "environment", + "type": "query", + "label": "Environment", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob3 AS environment FROM runner_events", + "refresh": 1, + "sort": 1, + "multi": false, + "includeAll": false, + "current": {} + }, + { + "name": "framework", + "type": "query", + "label": "Framework", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT DISTINCT blob6 AS framework FROM runner_events WHERE blob6 != ''", + "refresh": 1, + "sort": 1, + "multi": true, + "includeAll": true, + "current": { + "text": "All", + "value": [ + "$__all" + ] + } + }, + { + "name": "ht_major", + "type": "custom", + "label": "HT major", + "query": "15,16,17,18,19,next,none", + "current": { + "text": "All", + "value": [ + "$__all" + ] + }, + "options": [ + { + "text": "All", + "value": "$__all", + "selected": true + }, + { + "text": "15", + "value": "15" + }, + { + "text": "16", + "value": "16" + }, + { + "text": "17", + "value": "17" + }, + { + "text": "18", + "value": "18" + }, + { + "text": "19", + "value": "19" + }, + { + "text": "next", + "value": "next" + }, + { + "text": "none", + "value": "none" + } + ], + "hide": 0, + "includeAll": true, + "multi": true + } + ] + }, + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "name": "Annotations & Alerts", + "type": "dashboard" + } + ] + }, + "panels": [ + { + "id": 1, + "type": "table", + "title": "preview.ready_ms by framework × ht_major", + "description": "\"next\" is a real ht_major value here, not a separate column — sort/filter on it to highlight the channel.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT blob6 AS framework, blob7 AS ht_major, sum(_sample_interval * double1) AS samples, 100 * sum(_sample_interval * double1 * (blob8 = 'ready')) / sum(_sample_interval * double1) AS ready_pct, quantileExactWeighted(0.95)(double2, toUInt32(_sample_interval)) AS p95_ms FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'preview.ready_ms' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY framework, ht_major ORDER BY framework, ht_major", + "format": "table", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": {}, + "overrides": [] + }, + "options": {} + }, + { + "id": 2, + "type": "timeseries", + "title": "sandpack.compile_error rate by ht_major", + "description": "Day-over-day comparability per ht_major; \"next\" highlighted via a series override.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob7 AS ht_major, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'sandpack.compile_error' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, ht_major ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [ + { + "matcher": { + "id": "byName", + "options": "next" + }, + "properties": [ + { + "id": "color", + "value": { + "mode": "fixed", + "fixedColor": "purple" + } + }, + { + "id": "custom.lineWidth", + "value": 3 + } + ] + } + ] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + }, + { + "id": 3, + "type": "timeseries", + "title": "version.switch count by ht_major (to)", + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "targets": [ + { + "refId": "A", + "datasource": { + "type": "vertamedia-clickhouse-datasource", + "uid": "clickhouse-runner-events" + }, + "query": "SELECT toStartOfInterval(timestamp, INTERVAL '$interval' SECOND) AS t, blob7 AS ht_major, sum(_sample_interval * double1) AS cnt FROM $table WHERE $timeFilterByColumn(timestamp) AND index1 = 'version.switch' AND blob3 = '$environment' AND blob6 IN (${framework:sqlstring}) AND blob7 IN (${ht_major:sqlstring}) GROUP BY t, ht_major ORDER BY t", + "format": "time_series", + "table": "runner_events", + "round": "0s", + "intervalFactor": 1 + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "fillOpacity": 10 + } + }, + "overrides": [ + { + "matcher": { + "id": "byName", + "options": "next" + }, + "properties": [ + { + "id": "color", + "value": { + "mode": "fixed", + "fixedColor": "purple" + } + }, + { + "id": "custom.lineWidth", + "value": 3 + } + ] + } + ] + }, + "options": { + "legend": { + "displayMode": "table", + "placement": "bottom", + "calcs": [ + "last", + "max" + ] + } + } + } + ] +} diff --git a/runner/containers/o11y/grafana/grafana.ini b/runner/containers/o11y/grafana/grafana.ini new file mode 100644 index 0000000000..1b4c212eed --- /dev/null +++ b/runner/containers/o11y/grafana/grafana.ini @@ -0,0 +1,117 @@ +; ADR-0041 §A, §B.5, §H. Pinned settings a config test guards +; (pipeline/o11y-box-config.test.mjs). Values that must vary per environment +; (root_url host:port, plugin token, datasource endpoints) are supplied by +; GF_* environment variables or provisioning-file env expansion instead of +; being hard-coded here, so this file never needs editing per deploy. +; +; Trap: any GF_SERVER_*, GF_AUTH_PROXY_*, GF_LIVE_* env var set in compose.yml +; or wrangler.jsonc silently overrides the pinned values below. Do not set +; those — override only GF_SERVER_ROOT_URL (host/port) and datasource-level +; values via provisioning env expansion. + +[server] +protocol = http +http_port = 3000 +domain = localhost +; The Worker/compose sets GF_SERVER_ROOT_URL to the real host; this default +; keeps `docker compose up` usable standalone. +root_url = http://localhost:3000/grafana/ +serve_from_sub_path = true + +[auth] +disable_login_form = true +disable_signout_menu = true +oauth_auto_login = false + +[auth.anonymous] +enabled = false + +[auth.basic] +; auth.proxy is meant to be the ONLY trusted identity (X-O11Y-GRAFANA-USER, +; set by the Worker; a client-sent copy is stripped — ADR-0041 §B.5). Basic +; auth left enabled would let anyone who reaches the box's HTTP port at all +; authenticate as the default admin, independent of whether the Worker's +; proxy happens to forward an Authorization header. Verified: with this +; disabled, `curl -u admin:admin ...` gets 401, not 200. +enabled = false + +[auth.proxy] +enabled = true +header_name = X-O11Y-GRAFANA-USER +header_property = username +auto_sign_up = true +sync_ttl = 15 +whitelist = +headers = +enable_login_token = false + +[users] +allow_sign_up = false +auto_assign_org = true +auto_assign_org_role = Viewer +; A signed-in Viewer gets Explore (Grafana gates Explore on this +; flag, not on the Editor/Admin role) and can make transient, in-session +; panel edits (add/remove a panel, change a query) on the dashboard they're +; looking at. Nothing here grants a Viewer the ability to SAVE those edits +; back to a provisioned dashboard or datasource: dashboards stay read-only +; in effect because `allowUiUpdates: false` in +; provisioning/dashboards/dashboards.yaml refuses the save, and every +; datasource in provisioning/datasources/datasources.yaml is `editable: +; false`. And since Grafana's sqlite state is disposable (a fresh DB on +; every wake, ADR-0041 §A), even a Viewer edit that somehow landed would +; never survive the next sleep/wake cycle. Verified: a Viewer session can +; open /explore and run a LogQL query; a Viewer POST to save a dashboard +; still 403s. +viewers_can_edit = true + +[live] +; ADR-0041 §A: Grafana Live disabled — see also ADR-0041 alternatives +; (revision 2's Alloy design). This DOES turn +; Live off for Grafana's own frontend — `GET /api/frontend/settings` reports +; `liveEnabled: false` with this set (verified on 11.4.0), so the shipped +; Grafana UI never opens a Live socket and never retries one on an idle tab. +; It does NOT refuse a client that dials `/api/live/ws` directly, though: +; both the plain HTTP upgrade (101) and a full Centrifuge `connect` +; handshake (a real client id, ping/pong) succeed regardless of this value — +; reproduced on 11.4.0 and 12.2.0, over both HTTP Basic auth and a real +; cookie session, matching the still-open grafana/grafana#72072. Safe for +; the product surface, not a hard refusal of the endpoint itself. +; GrafanaBox.containerFetch (workers/o11y/src/box.ts) still +; rejects any /api/live/* path before it reaches this container as +; defense-in-depth against a client that bypasses Grafana's own frontend — +; the container is never reachable except through the Worker (ADR-0041 §A). +max_connections = 0 + +[analytics] +reporting_enabled = false +check_for_updates = false +check_for_plugin_updates = false +feedback_links_enabled = false + +[security] +disable_gravatar = true +cookie_secure = true +cookie_samesite = strict +allow_embedding = false +; No default admin account to fall back to once [auth.basic] is off. +disable_initial_admin_creation = true + +[dashboards] +default_home_dashboard_path = + +[plugins] +allow_loading_unsigned_plugins = +; Grafana 11's default background preinstall of grafana-lokiexplore-app +; fetches a zip from grafana.com on every cold start otherwise: an unpinned, +; not-image-baked dependency on exit criterion 6 (cold start). Every plugin +; this box needs is installed at build time (Dockerfile), pinned by version. +preinstall_disabled = true + +[log] +mode = console +level = info + +[database] +; sqlite state is disposable (ADR-0041 §A) — no external DB, ephemeral disk. +type = sqlite3 +path = /var/lib/grafana/grafana.db diff --git a/runner/containers/o11y/grafana/provisioning/dashboards/dashboards.yaml b/runner/containers/o11y/grafana/provisioning/dashboards/dashboards.yaml new file mode 100644 index 0000000000..765e513193 --- /dev/null +++ b/runner/containers/o11y/grafana/provisioning/dashboards/dashboards.yaml @@ -0,0 +1,14 @@ +# Dashboards provider — the 8 provisioned JSON dashboards live under +# containers/o11y/grafana/dashboards/ (ADR-0041 §F.2, ADR-0042). +apiVersion: 1 + +providers: + - name: default + orgId: 1 + folder: "" + type: file + disableDeletion: false + updateIntervalSeconds: 30 + allowUiUpdates: false + options: + path: /etc/grafana/dashboards diff --git a/runner/containers/o11y/grafana/provisioning/datasources/datasources.yaml b/runner/containers/o11y/grafana/provisioning/datasources/datasources.yaml new file mode 100644 index 0000000000..fc62c2dfe2 --- /dev/null +++ b/runner/containers/o11y/grafana/provisioning/datasources/datasources.yaml @@ -0,0 +1,62 @@ +# ADR-0041 §A, §B.5. One Loki datasource per tenant (X-Scope-OrgID), plus the +# Altinity ClickHouse plugin against the local ClickHouse container in dev or +# the Analytics Engine SQL API in production. Fixed `uid`s: Grafana's sqlite +# state is disposable (a fresh DB on every wake), so a dashboard (T09) that +# references a datasource by uid would break on every wake if these were +# auto-generated. Grafana expands ${VAR} in provisioning files (documented +# since v8.0) — do not hard-code an endpoint or credential here. +apiVersion: 1 + +datasources: + - name: Loki (browser) + uid: loki-browser + type: loki + access: proxy + url: http://localhost:3100 + jsonData: + httpHeaderName1: X-Scope-OrgID + maxLines: 1000 + secureJsonData: + httpHeaderValue1: browser + editable: false + isDefault: false + + - name: Loki (worker) + uid: loki-worker + type: loki + access: proxy + url: http://localhost:3100 + jsonData: + httpHeaderName1: X-Scope-OrgID + maxLines: 1000 + secureJsonData: + httpHeaderValue1: worker + editable: false + isDefault: true + + # vertamedia-clickhouse-datasource ("the Altinity plugin", ADR-0041 §A) — + # NOT grafana-clickhouse-datasource. Cloudflare's own Analytics Engine + + # Grafana docs point at this exact plugin id and this exact auth shape: a + # full URL plus ONE custom HTTP header, never host/port/user/password + # fields (the AE SQL API rejects Basic auth outright — verified against a + # real ClickHouse HTTP endpoint too, which 403s a bare `Bearer` scheme). + # Local ClickHouse and production Analytics Engine need genuinely + # different header shapes (ClickHouse HTTP: X-ClickHouse-User/-Key; + # AE SQL API: a single `Authorization: Bearer <token>`), so the header + # NAMES are env-driven too, not just the values — compose.yml supplies + # the local pair; GrafanaBox's envVars (phase 2) supply the production + # pair. Never hard-code either shape here. + - name: ClickHouse (runner_events) + uid: clickhouse-runner-events + type: vertamedia-clickhouse-datasource + access: proxy + url: ${O11Y_CLICKHOUSE_URL} + jsonData: + httpHeaderName1: ${O11Y_CLICKHOUSE_HEADER1_NAME} + httpHeaderName2: ${O11Y_CLICKHOUSE_HEADER2_NAME} + defaultDatabase: ${O11Y_CLICKHOUSE_DATABASE} + addCorsHeader: false + secureJsonData: + httpHeaderValue1: ${O11Y_CLICKHOUSE_HEADER1_VALUE} + httpHeaderValue2: ${O11Y_CLICKHOUSE_HEADER2_VALUE} + editable: false diff --git a/runner/containers/o11y/local/clickhouse-init.sql b/runner/containers/o11y/local/clickhouse-init.sql new file mode 100644 index 0000000000..4654c61153 --- /dev/null +++ b/runner/containers/o11y/local/clickhouse-init.sql @@ -0,0 +1,60 @@ +-- docs/observability-contract.md §4, §10. Local stand-in for the Workers +-- Analytics Engine `runner_events` dataset: the SAME columns the real AE SQL +-- API answers with (index1, blob1-20, double1-20, timestamp, +-- _sample_interval), so the o11y worker's one allowlisting SQL helper +-- (ADR-0041 §4 "Reading rule") runs unmodified against either backend. Do +-- NOT rename these to the §4 table's friendly names (metric, service_name, +-- ...) — that column mapping lives in the query helper, not the schema. +CREATE DATABASE IF NOT EXISTS default; + +CREATE TABLE IF NOT EXISTS default.runner_events +( + `timestamp` DateTime64(3) DEFAULT now64(3), + `index1` String, + `blob1` String DEFAULT '', + `blob2` String DEFAULT '', + `blob3` String DEFAULT '', + `blob4` String DEFAULT '', + `blob5` String DEFAULT '', + `blob6` String DEFAULT '', + `blob7` String DEFAULT '', + `blob8` String DEFAULT '', + `blob9` String DEFAULT '', + `blob10` String DEFAULT '', + `blob11` String DEFAULT '', + `blob12` String DEFAULT '', + `blob13` String DEFAULT '', + `blob14` String DEFAULT '', + `blob15` String DEFAULT '', + `blob16` String DEFAULT '', + `blob17` String DEFAULT '', + `blob18` String DEFAULT '', + `blob19` String DEFAULT '', + `blob20` String DEFAULT '', + `double1` Float64 DEFAULT 0, + `double2` Float64 DEFAULT 0, + `double3` Float64 DEFAULT 0, + `double4` Float64 DEFAULT 0, + `double5` Float64 DEFAULT 0, + `double6` Float64 DEFAULT 0, + `double7` Float64 DEFAULT 0, + `double8` Float64 DEFAULT 0, + `double9` Float64 DEFAULT 0, + `double10` Float64 DEFAULT 0, + `double11` Float64 DEFAULT 0, + `double12` Float64 DEFAULT 0, + `double13` Float64 DEFAULT 0, + `double14` Float64 DEFAULT 0, + `double15` Float64 DEFAULT 0, + `double16` Float64 DEFAULT 0, + `double17` Float64 DEFAULT 0, + `double18` Float64 DEFAULT 0, + `double19` Float64 DEFAULT 0, + `double20` Float64 DEFAULT 0, + -- Real AE rows are sampled at write AND read time; the local shim never + -- samples, so every row carries the AE-equivalent constant. Every count + -- reads `SUM(_sample_interval * double1)`, never `COUNT()` (§4). + `_sample_interval` Float64 DEFAULT 1 +) +ENGINE = MergeTree +ORDER BY (index1, timestamp); diff --git a/runner/containers/o11y/local/redact.mjs b/runner/containers/o11y/local/redact.mjs new file mode 100644 index 0000000000..3f099174ee --- /dev/null +++ b/runner/containers/o11y/local/redact.mjs @@ -0,0 +1,21 @@ +// containers/o11y/local/redact.mjs +// +// Split out of stop-roundtrip.mjs so it is unit-testable — that file runs +// `main()` unconditionally at module scope, so importing it would run the +// whole real docker-compose roundtrip. Scrubs known secret VALUES (not a +// pattern match) out of text before it is ever printed. + +/** + * @param {string} text + * @param {ReadonlyArray<string | undefined | null>} secrets + * @returns {string} + */ +export function scrubSecrets(text, secrets) { + let out = text ?? ""; + for (const secret of secrets) { + // A falsy/empty secret would match everywhere (`"".split("")` splits + // every character) — skip those rather than mangling the text. + if (secret) out = out.split(secret).join("<redacted>"); + } + return out; +} diff --git a/runner/containers/o11y/local/stop-roundtrip.mjs b/runner/containers/o11y/local/stop-roundtrip.mjs new file mode 100644 index 0000000000..8e033148f9 --- /dev/null +++ b/runner/containers/o11y/local/stop-roundtrip.mjs @@ -0,0 +1,602 @@ +#!/usr/bin/env node +// containers/o11y/local/stop-roundtrip.mjs +// +// Proves ADR-0041 exit criterion 1 and 2 against a REAL docker compose +// stack: push OTLP lines, SIGTERM the box, confirm the clean-shutdown +// marker, force-recreate it, and confirm every line is still queryable. +// Then the negative control: SIGKILL a second wake, confirm NO marker. +// +// Usage: node containers/o11y/local/stop-roundtrip.mjs. Exit 0 = every +// check passed. Run through `rtk proxy` — judge by exit code. + +import { spawnSync } from "node:child_process"; +import { setTimeout as sleep } from "node:timers/promises"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; +import { randomUUID } from "node:crypto"; +import { scrubSecrets } from "./redact.mjs"; +import { defaultComposeProjectName } from "../../../scripts/dev-lib.mjs"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const O11Y_DIR = join(__dirname, ".."); + +// This script's own up-front `down -v` (below) wipes whatever project +// name it resolves to. The default must never collide with a persistent +// dev stack's own default (`defaultComposeProjectName()`, imported from +// dev-lib.mjs so the two can't drift) or compose.yml's own "o11y-t01" +// example. +// +// A distinct DEFAULT alone is not enough: an inherited +// `COMPOSE_PROJECT_NAME` env var would still win and let `down -v` wipe +// the real dev stack, so reusing the dev stack's project name is a hard refusal. +const DEV_STACK_DEFAULT_PROJECT = defaultComposeProjectName(); +const REQUESTED_PROJECT = process.env.COMPOSE_PROJECT_NAME || "o11y-stop-roundtrip"; +if (REQUESTED_PROJECT === DEV_STACK_DEFAULT_PROJECT) { + console.error( + `error: COMPOSE_PROJECT_NAME="${DEV_STACK_DEFAULT_PROJECT}" is dev.mjs's own persistent dev-stack project name ` + + `for this worktree (scripts/dev-lib.mjs's defaultComposeProjectName()) — this script's own \`down -v\` would ` + + `wipe its named volumes (minio-data/clickhouse-data). Refusing to run under this project name; set ` + + `COMPOSE_PROJECT_NAME to something else (or unset it to use this script's own "o11y-stop-roundtrip" default).`, + ); + process.exit(1); +} +const PROJECT = REQUESTED_PROJECT; +const GRAFANA_PORT = process.env.O11Y_GRAFANA_PORT || "4200"; +const LOKI_PORT = process.env.O11Y_LOKI_PORT || "4201"; +const MINIO_PORT = process.env.O11Y_MINIO_PORT || "4202"; +const MINIO_CONSOLE_PORT = process.env.O11Y_MINIO_CONSOLE_PORT || "4203"; +const CLICKHOUSE_PORT = process.env.O11Y_CLICKHOUSE_PORT || "4204"; +const CLICKHOUSE_NATIVE_PORT = process.env.O11Y_CLICKHOUSE_NATIVE_PORT || "4205"; +const MINIO_USER = process.env.O11Y_MINIO_ROOT_USER || "minioadmin"; +const MINIO_PASSWORD = process.env.O11Y_MINIO_ROOT_PASSWORD || "minioadmin"; + +const RUN_ID = randomUUID().slice(0, 8); + +let failures = 0; +const results = []; + +function record(name, ok, detail) { + results.push({ name, ok, detail }); + console.log(`${ok ? "PASS" : "FAIL"} ${name}${detail ? ` — ${detail}` : ""}`); + if (!ok) failures++; +} + +const BASE_ENV = { + ...process.env, + COMPOSE_PROJECT_NAME: PROJECT, + O11Y_GRAFANA_PORT: GRAFANA_PORT, + O11Y_LOKI_PORT: LOKI_PORT, + O11Y_MINIO_PORT: MINIO_PORT, + O11Y_MINIO_CONSOLE_PORT: MINIO_CONSOLE_PORT, + O11Y_CLICKHOUSE_PORT: CLICKHOUSE_PORT, + O11Y_CLICKHOUSE_NATIVE_PORT: CLICKHOUSE_NATIVE_PORT, + O11Y_MINIO_ROOT_USER: MINIO_USER, + O11Y_MINIO_ROOT_PASSWORD: MINIO_PASSWORD, +}; + +// MINIO_USER/MINIO_PASSWORD (root creds) are always scrubbed from a +// failure log; `opts.redact` lets a call site add its own per-run secrets. +// Output is redacted, not suppressed — a failure still prints for +// diagnosis, just never a real credential. +function sh(cmd, args, opts = {}) { + const { env: extraEnv, redact: redactValues = [], ...restOpts } = opts; + const res = spawnSync(cmd, args, { + cwd: O11Y_DIR, + encoding: "utf8", + env: { ...BASE_ENV, ...extraEnv }, + ...restOpts, + }); + if (res.status !== 0 && !opts.allowFail) { + const secrets = [MINIO_USER, MINIO_PASSWORD, ...redactValues]; + const scrub = (text) => scrubSecrets(text ?? "", secrets); + console.error(`command failed: ${scrub(`${cmd} ${args.join(" ")}`)}\n${scrub(res.stdout)}\n${scrub(res.stderr)}`); + } + return res; +} + +// A trailing plain-object argument to `compose(...)` is treated as options +// for `sh()` (e.g. `{ allowFail: true }` or `redact`) and popped off before +// building the compose args, so every existing call site (which only ever +// passes strings) is unaffected. +function compose(...args) { + let opts = {}; + const last = args[args.length - 1]; + if (last !== null && typeof last === "object" && !Array.isArray(last)) { + opts = args.pop(); + } + return sh("docker", ["compose", "-p", PROJECT, "-f", "compose.yml", ...args], opts); +} + +// --- S3-signed helpers against the MinIO bucket, mirroring lib.sh --------- + +function curlS3(args, opts = {}) { + return sh( + "curl", + [ + "-sS", + "--max-time", + "15", + "--aws-sigv4", + "aws:amz:auto:s3", + "--user", + `${MINIO_USER}:${MINIO_PASSWORD}`, + ...args, + ], + { allowFail: true, ...opts }, + ); +} + +function markerExists(wakeId) { + const res = curlS3([ + "-o", + "/dev/null", + "-w", + "%{http_code}", + "-I", + `http://localhost:${MINIO_PORT}/loki/state/wakes/${wakeId}/clean`, + ]); + return res.stdout.trim() === "200"; +} + +// --- Loki push / query ----------------------------------------------------- + +function otlpBody(tenant, lines, extra = {}) { + const nowNs = String(Date.now() * 1_000_000); + return { + resourceLogs: [ + { + resource: { + attributes: [ + { key: "service.name", value: { stringValue: tenant === "browser" ? "demos-authoring" : "demos-api" } }, + { key: "service.version", value: { stringValue: "roundtrip-sha" } }, + { key: "deployment.environment.name", value: { stringValue: "local" } }, + { key: "hot.surface", value: { stringValue: tenant === "browser" ? "authoring" : "api" } }, + { key: "hot.tier", value: { stringValue: "1" } }, + { key: "hot.framework", value: { stringValue: "react" } }, + { key: "hot.ht_major", value: { stringValue: "18" } }, + { key: "hot.outcome", value: { stringValue: "ready" } }, + ...Object.entries(extra).map(([key, value]) => ({ key, value: { stringValue: String(value) } })), + ], + }, + scopeLogs: [ + { + logRecords: lines.map((line) => ({ + timeUnixNano: nowNs, + body: { stringValue: line }, + severityText: "INFO", + })), + }, + ], + }, + ], + }; +} + +async function pushLines(tenant, lines, extra = {}) { + const res = await fetch(`http://localhost:${LOKI_PORT}/otlp/v1/logs`, { + method: "POST", + headers: { "Content-Type": "application/json", "X-Scope-OrgID": tenant }, + body: JSON.stringify(otlpBody(tenant, lines, extra)), + }); + return res.status; +} + +async function queryLines(tenant, serviceName, sinceMs) { + const startNs = String((sinceMs - 3_600_000) * 1_000_000); + const url = new URL(`http://localhost:${LOKI_PORT}/loki/api/v1/query_range`); + url.searchParams.set("query", `{service_name="${serviceName}"}`); + url.searchParams.set("start", startNs); + url.searchParams.set("limit", "100"); + const res = await fetch(url, { headers: { "X-Scope-OrgID": tenant } }); + if (res.status !== 200) return { status: res.status, lines: [] }; + const body = await res.json(); + const lines = (body?.data?.result ?? []).flatMap((stream) => stream.values.map((v) => v[1])); + return { status: res.status, lines }; +} + +async function waitReady(timeoutMs = 90_000) { + const start = Date.now(); + while (Date.now() - start < timeoutMs) { + try { + const [g, l] = await Promise.all([ + fetch(`http://localhost:${GRAFANA_PORT}/grafana/api/health`).then((r) => r.status).catch(() => 0), + fetch(`http://localhost:${LOKI_PORT}/ready`).then((r) => r.status).catch(() => 0), + ]); + if (g === 200 && l === 200) return Date.now() - start; + } catch { + // keep polling + } + await sleep(1000); + } + return -1; +} + +function boxContainerId() { + const res = compose("ps", "-q", "box"); + return res.stdout.trim(); +} + +/** The supervisor's own stdout log lines for a container — used to confirm + * which branch of `run_stop_protocol` actually + * refused the marker (a listing failure vs. an upload failure vs. neither + * applying), rather than inferring it only from "no marker" (which both + * branches, and several others, all produce identically). */ +function boxLogs(containerId) { + const res = sh("docker", ["logs", containerId], { allowFail: true }); + return `${res.stdout}\n${res.stderr}`; +} + +// --- main ------------------------------------------------------------------- + +async function main() { + console.log(`o11y box stop-roundtrip — project=${PROJECT} run=${RUN_ID}`); + console.log(`ports: grafana=${GRAFANA_PORT} loki=${LOKI_PORT} minio=${MINIO_PORT}`); + + // compose.yml's minio/clickhouse use named volumes, so a PRIOR run that + // crashed before its own `down -v` (O11Y_ROUNDTRIP_KEEP unset) could + // leave volumes around for THIS run to silently reuse. This `down -v` + // guarantees a genuinely empty start regardless. Harmless when nothing is left over. + compose("down", "-v"); + + console.log("\n== bring up minio + clickhouse =="); + // `minio-init` (a one-shot `mc mb` container) is not used — quay.io/minio/mc + // is not pullable. Bitnami's `minio` image creates the `loki` bucket + // itself via MINIO_DEFAULT_BUCKETS (compose.yml) before its healthcheck + // goes green, so `--wait` (blocks until every started service is + // healthy/running) is sufficient. + const bringUpRes = compose("up", "-d", "--wait", "minio", "clickhouse"); + record( + "minio became healthy (bucket created — compose.yml's MINIO_DEFAULT_BUCKETS)", + bringUpRes.status === 0, + `exit=${bringUpRes.status}`, + ); + + // ---- run 1: clean stop ----------------------------------------------- + const wakeIdClean = `roundtrip-clean-${RUN_ID}`; + console.log(`\n== run 1 (clean stop): wakeId=${wakeIdClean} ==`); + { + const bootStart = Date.now(); + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { env: { O11Y_WAKE_ID: wakeIdClean } }); + // compose.yml reads O11Y_WAKE_ID at container-create time; the env above + // must reach the `up` invocation directly. + const readyMs = await waitReadyForBox(); + record("run1: box became ready", readyMs >= 0, `${readyMs}ms`); + console.log(`local boot: ${readyMs}ms`); + } + + // Behavioural check that [live] max_connections=0 actually turns Live off + // for Grafana's own frontend — `liveEnabled` in /api/frontend/settings is + // what the shipped Grafana UI reads before ever opening a socket. This + // does NOT check the raw /api/live/ws endpoint itself (that remains + // reachable — see grafana.ini's own note); it checks the one surface the + // product actually consults. + { + const res = await fetch(`http://localhost:${GRAFANA_PORT}/grafana/api/frontend/settings`, { + headers: { "X-O11Y-GRAFANA-USER": "roundtrip-probe@handsontable.com" }, + }); + const body = res.status === 200 ? await res.json() : null; + record( + "run1: /api/frontend/settings reports liveEnabled=false", + res.status === 200 && body?.liveEnabled === false, + `HTTP ${res.status} liveEnabled=${body?.liveEnabled}`, + ); + } + + const pushedBrowser = [`roundtrip-browser-a-${RUN_ID}`, `roundtrip-browser-b-${RUN_ID}`, `roundtrip-browser-c-${RUN_ID}`]; + const pushedWorker = [`roundtrip-worker-a-${RUN_ID}`, `roundtrip-worker-b-${RUN_ID}`]; + const pushStart = Date.now(); + const bStatus = await pushLines("browser", pushedBrowser, { "hot.demo_id": "r-roundtrip" }); + const wStatus = await pushLines("worker", pushedWorker, { "hot.demo_id": "r-roundtrip" }); + record("run1: browser push accepted", bStatus === 204, `HTTP ${bStatus}`); + record("run1: worker push accepted", wStatus === 204, `HTTP ${wStatus}`); + await sleep(500); + + const containerId1 = boxContainerId(); + record("run1: box container found", Boolean(containerId1)); + + const termStart = Date.now(); + sh("docker", ["kill", "-s", "TERM", containerId1]); + sh("docker", ["wait", containerId1]); + const termMs = Date.now() - termStart; + const exitCode1 = sh("docker", ["inspect", containerId1, "--format", "{{.State.ExitCode}}"]).stdout.trim(); + record("run1: SIGTERM stop wall time recorded", true, `${termMs}ms`); + record("run1: box exited 0 on SIGTERM", exitCode1 === "0", `exit=${exitCode1}`); + record("run1: clean-shutdown marker present", markerExists(wakeIdClean), `state/wakes/${wakeIdClean}/clean`); + + // ---- restart from a genuinely fresh container ------------------------- + console.log("\n== restart (fresh container, same MinIO bucket) =="); + const wakeIdRestart = `roundtrip-restart-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { env: { O11Y_WAKE_ID: wakeIdRestart } }); + const readyMs2 = await waitReadyForBox(); + record("restart: box became ready", readyMs2 >= 0, `${readyMs2}ms`); + + const { status: bqStatus, lines: bLines } = await queryLines("browser", "demos-authoring", pushStart); + const { status: wqStatus, lines: wLines } = await queryLines("worker", "demos-api", pushStart); + record("restart: browser query 200", bqStatus === 200, `HTTP ${bqStatus}`); + record("restart: worker query 200", wqStatus === 200, `HTTP ${wqStatus}`); + const browserSetOk = pushedBrowser.every((l) => bLines.includes(l)) && bLines.length === pushedBrowser.length; + const workerSetOk = pushedWorker.every((l) => wLines.includes(l)) && wLines.length === pushedWorker.length; + record( + "restart: 100% of pushed browser lines queryable (exact set)", + browserSetOk, + `expected ${JSON.stringify(pushedBrowser)} got ${JSON.stringify(bLines)}`, + ); + record( + "restart: 100% of pushed worker lines queryable (exact set)", + workerSetOk, + `expected ${JSON.stringify(pushedWorker)} got ${JSON.stringify(wLines)}`, + ); + + // ---- run 1b: zero-ingest wake ------------------------------------------- + // + // A wake that pushes NOTHING to Loki never produces a new uploader-named + // index object, so shutdown.sh writes no marker for it — the ledger's + // own "clean" comes from having nothing provisional to lose + // (`resolveOverWakes`), never from a marker that doesn't exist. + // + // Measured here: a Loki with no writes this wake exits 1 on SIGTERM, not + // 0. `shutdown.sh` only checks the upload when `loki_exit -eq 0`, so this + // takes the SAME no-marker path via a different branch; recorded as + // evidence, not asserted. + console.log("\n== run 1b (zero ingest): no OTLP push at all =="); + const wakeIdZero = `roundtrip-zero-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { env: { O11Y_WAKE_ID: wakeIdZero } }); + const readyMsZero = await waitReadyForBox(); + record("run1b: box became ready", readyMsZero >= 0, `${readyMsZero}ms`); + + const containerIdZero = boxContainerId(); + record("run1b: box container found", Boolean(containerIdZero)); + sh("docker", ["kill", "-s", "TERM", containerIdZero]); + sh("docker", ["wait", containerIdZero]); + const exitCodeZero = sh("docker", ["inspect", containerIdZero, "--format", "{{.State.ExitCode}}"]).stdout.trim(); + record("run1b: SIGTERM completed and exit code recorded (T03B-D1: a truly-untouched Loki exits 1, not 0 — informational, not asserted)", true, `exit=${exitCodeZero}`); + record( + "run1b: no marker written for a zero-ingest wake (shutdown.sh's own contract — the ledger, not this script, is what now treats this as clean)", + !markerExists(wakeIdZero), + `state/wakes/${wakeIdZero}/clean`, + ); + + // ---- run 2: negative control — SIGKILL writes no marker --------------- + console.log("\n== run 2 (negative control): SIGKILL =="); + const wakeIdKill = `roundtrip-kill-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { env: { O11Y_WAKE_ID: wakeIdKill } }); + const readyMs3 = await waitReadyForBox(); + record("run2: box became ready", readyMs3 >= 0, `${readyMs3}ms`); + const killPushStart = Date.now(); + const killPushStatus = await pushLines("browser", [`roundtrip-kill-canary-${RUN_ID}`], { "hot.demo_id": "r-roundtrip" }); + record("run2: canary push accepted", killPushStatus === 204, `HTTP ${killPushStatus}`); + await sleep(300); + + const containerId2 = boxContainerId(); + sh("docker", ["kill", "-s", "KILL", containerId2]); + sh("docker", ["wait", containerId2]); + const exitCode2 = sh("docker", ["inspect", containerId2, "--format", "{{.State.ExitCode}}"]).stdout.trim(); + record("run2: box exit code is the SIGKILL code (137)", exitCode2 === "137", `exit=${exitCode2}`); + record("run2: NO marker written for a SIGKILL'd wake", !markerExists(wakeIdKill), `state/wakes/${wakeIdKill}/clean`); + // The clean run's own marker must still be the only one with this run id. + record( + "run2: the earlier clean-run marker is untouched", + markerExists(wakeIdClean), + `state/wakes/${wakeIdClean}/clean`, + ); + + // Data-loss proof: recreate the box and confirm the canary (pushed but + // never index-uploaded, SIGKILLed mid-ingest) is genuinely gone — also + // the negative control for the "restart" section's positive query above. + console.log("\n== restart after SIGKILL (data-loss negative control) =="); + const wakeIdAfterKill = `roundtrip-after-kill-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { env: { O11Y_WAKE_ID: wakeIdAfterKill } }); + const readyMs4 = await waitReadyForBox(); + record("run2 restart: box became ready", readyMs4 >= 0, `${readyMs4}ms`); + const { status: killQueryStatus, lines: killQueryLines } = await queryLines("browser", "demos-authoring", killPushStart); + record("run2 restart: browser query 200", killQueryStatus === 200, `HTTP ${killQueryStatus}`); + record( + "run2 restart: the SIGKILL'd canary line is genuinely lost (never uploaded)", + !killQueryLines.includes(`roundtrip-kill-canary-${RUN_ID}`), + `got ${JSON.stringify(killQueryLines)}`, + ); + + // ---- C1 negative control: a failed FINAL index upload writes no marker ---- + // + // The bug this proves fixed: an earlier check only asked "does an + // uploader-named index object exist?" — but the shipper also uploads + // periodically while Loki runs, so an object can exist from an EARLIER + // upload even if the LAST, shutdown-time one silently failed. Reproduces + // that shape: seed a fake object, then run the box against a MinIO user + // denied PutObject on `index/*` only, so the real shutdown-time upload + // fails and no NEW uploader-named object appears. + console.log("\n== C1 negative control: failed final index upload =="); + const RESTRICTED_USER = `restricted-${RUN_ID}`; + const RESTRICTED_PASSWORD = `restricted-pw-${RUN_ID}`; + const RESTRICTED_POLICY = `deny-index-put-${RUN_ID}`; + setupRestrictedMinioUser(RESTRICTED_USER, RESTRICTED_PASSWORD, RESTRICTED_POLICY); + + const wakeIdC1 = `roundtrip-c1-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { + env: { + O11Y_WAKE_ID: wakeIdC1, + O11Y_LOKI_S3_ACCESS_KEY_ID: RESTRICTED_USER, + O11Y_LOKI_S3_SECRET_ACCESS_KEY: RESTRICTED_PASSWORD, + }, + }); + const readyMsC1 = await waitReadyForBox(); + record("C1: box (restricted S3 credential) became ready", readyMsC1 >= 0, `${readyMsC1}ms`); + + const containerIdC1 = boxContainerId(); + const uploaderName = sh("docker", [ + "exec", containerIdC1, "cat", "/loki/tsdb-index/uploader/name", + ], { allowFail: true }).stdout.trim(); + record("C1: read this instance's uploader name", Boolean(uploaderName), uploaderName || "(empty)"); + + // Seed a fake "earlier successful upload" using the ADMIN credential — + // the restricted user could not have written this itself, standing in + // for an upload before the pre-SIGTERM snapshot. + const dayNow = Math.floor(Date.now() / 1000 / 86400); + const seededKey = `index/index/${dayNow}/9999999999-${uploaderName}-seeded.tsdb.gz`; + const seedRes = curlS3(["-o", "/dev/null", "-w", "%{http_code}", "-X", "PUT", "--data", "seed", + `http://localhost:${MINIO_PORT}/loki/${seededKey}`]); + record("C1: seeded a pre-existing uploader-named index object (admin credential)", seedRes.stdout.trim() === "200", `HTTP ${seedRes.stdout.trim()} key=${seededKey}`); + + await pushLines("browser", [`roundtrip-c1-line-${RUN_ID}`], { "hot.demo_id": "r-roundtrip" }); + await sleep(300); + + sh("docker", ["kill", "-s", "TERM", containerIdC1]); + sh("docker", ["wait", containerIdC1]); + const exitCodeC1 = sh("docker", ["inspect", containerIdC1, "--format", "{{.State.ExitCode}}"]).stdout.trim(); + record( + "C1: no marker written when the final index upload is blocked, despite a pre-existing uploader-named object", + !markerExists(wakeIdC1), + `state/wakes/${wakeIdC1}/clean, box exit=${exitCodeC1}`, + ); + // Confirm WHICH branch refused the marker, not just trust "no marker" + // alone. Measured live: denying `s3:PutObject` on `index/*` also rejects + // the ingester's own periodic CHUNK flush, which Loki retries in a + // backoff loop past `STOP_GRACE_SECONDS`, so the branch actually observed + // is shutdown.sh's grace-timeout log line, not a clean non-zero `wait` + // exit. Accepts all three shapes a real run could produce (grace-timeout, + // a clean non-zero exit, or a future Loki version's "not new" comparison) + // rather than asserting only the one this run happened to take. + const logsC1 = boxLogs(containerIdC1); + const putBlockedEvidence = /loki did not exit within \d+s of SIGTERM/.test(logsC1) + || /loki exited with code [1-9]\d*/.test(logsC1) + || /is new since before SIGTERM/.test(logsC1); + record( + "C1: the log confirms the PUT-blocked scenario is what refused it (grace-timeout, a non-zero loki exit, or an explicit not-new comparison)", + putBlockedEvidence, + putBlockedEvidence ? "found" : "none of the expected log shapes found in supervisor log", + ); + + // ---- D1 negative control: a failed pre-SIGTERM LISTING (not a blocked + // PUT) also writes no marker ------------------- + // + // C1 above proves the PUT-blocked path. This proves the OTHER path: + // `r2_list_prefix`'s own listing call fails (ListBucket denied), which + // must make `snapshot_ok=0` and refuse the marker BEFORE any upload + // confirmation is attempted. + console.log("\n== D1 negative control (B-I2): a failed pre-SIGTERM listing writes no marker =="); + const LIST_DENY_USER = `list-deny-${RUN_ID}`; + const LIST_DENY_PASSWORD = `list-deny-pw-${RUN_ID}`; + const LIST_DENY_POLICY = `deny-list-${RUN_ID}`; + // ListBucket is evaluated against the BUCKET's own ARN, never `/*` — get + // this wrong and the "negative control" would pass for testing nothing. + setupRestrictedMinioUser( + LIST_DENY_USER, + LIST_DENY_PASSWORD, + LIST_DENY_POLICY, + { Effect: "Deny", Action: ["s3:ListBucket"], Resource: ["arn:aws:s3:::loki"] }, + "D1: restricted MinIO user/policy created (deny ListBucket on the bucket itself)", + ); + + const wakeIdD1 = `roundtrip-d1-${RUN_ID}`; + sh("docker", [ + "compose", "-p", PROJECT, "-f", "compose.yml", + "up", "-d", "--no-deps", "--force-recreate", "box", + ], { + env: { + O11Y_WAKE_ID: wakeIdD1, + O11Y_LOKI_S3_ACCESS_KEY_ID: LIST_DENY_USER, + O11Y_LOKI_S3_SECRET_ACCESS_KEY: LIST_DENY_PASSWORD, + }, + }); + const readyMsD1 = await waitReadyForBox(); + // Ready under a ListBucket-denied credential proves boot itself does not + // need ListBucket, so a later "no marker" is attributable to shutdown-time listing. + record("D1: box (ListBucket-denied credential) became ready — boot itself does not need ListBucket", readyMsD1 >= 0, `${readyMsD1}ms`); + + const containerIdD1 = boxContainerId(); + const pushStatusD1 = await pushLines("browser", [`roundtrip-d1-line-${RUN_ID}`], { "hot.demo_id": "r-roundtrip" }); + record("D1: push accepted under the ListBucket-denied credential — ingest itself does not need ListBucket either", pushStatusD1 === 204, `HTTP ${pushStatusD1}`); + await sleep(300); + + sh("docker", ["kill", "-s", "TERM", containerIdD1]); + sh("docker", ["wait", containerIdD1]); + const exitCodeD1 = sh("docker", ["inspect", containerIdD1, "--format", "{{.State.ExitCode}}"]).stdout.trim(); + record( + "D1: no marker written when the pre-SIGTERM index LISTING is blocked", + !markerExists(wakeIdD1), + `state/wakes/${wakeIdD1}/clean, box exit=${exitCodeD1}`, + ); + const logsD1 = boxLogs(containerIdD1); + record( + "D1: the log confirms the LISTING-failure branch specifically refused it (shutdown.sh's own snapshot_ok gate)", + /pre-SIGTERM index listing failed/.test(logsD1), + logsD1.includes("pre-SIGTERM index listing failed") ? "found" : "not found in supervisor log", + ); + + console.log(`\n${failures === 0 ? "ALL CHECKS PASSED" : `${failures} CHECK(S) FAILED`}`); + process.exitCode = failures === 0 ? 0 : 1; + + // Tear the stack down when done, so a run of this script never leaves + // containers/networks for the operator to notice and clean up by hand. + // Set O11Y_ROUNDTRIP_KEEP=1 to skip this while debugging a failure. + if (process.env.O11Y_ROUNDTRIP_KEEP !== "1") { + console.log("\n== tearing down (set O11Y_ROUNDTRIP_KEEP=1 to skip) =="); + compose("down", "-v"); + } else { + console.log("\nO11Y_ROUNDTRIP_KEEP=1 set — stack left running."); + } + + async function waitReadyForBox() { + return waitReady(); + } +} + +// `denyStatement`: a single extra `Deny` statement layered on top of the +// same base `Allow s3:* on the whole bucket` every restricted user starts +// from. Defaults to C1's own PutObject-on-`index/*` deny so existing +// callers are unaffected. +function setupRestrictedMinioUser( + user, + password, + policyName, + denyStatement = { Effect: "Deny", Action: ["s3:PutObject"], Resource: ["arn:aws:s3:::loki/index/*"] }, + label = "restricted MinIO user/policy created (deny PutObject on index/*)", +) { + const policy = JSON.stringify({ + Version: "2012-10-17", + Statement: [ + { Effect: "Allow", Action: ["s3:*"], Resource: ["arn:aws:s3:::loki", "arn:aws:s3:::loki/*"] }, + denyStatement, + ], + }); + // quay.io/minio/mc is not pullable — the Bitnami `minio` image ships the + // real `mc` binary INSIDE the container itself, so this execs into the + // already-running service instead of `docker run`-ing a second image. + // `-T`: spawnSync has no TTY, and `compose exec` defaults `-t` ON. + const script = [ + `mc alias set c1 http://localhost:9000 "${MINIO_USER}" "${MINIO_PASSWORD}" >/dev/null`, + `cat > /tmp/policy.json <<'EOF'\n${policy}\nEOF`, + `mc admin policy create c1 ${policyName} /tmp/policy.json`, + `mc admin user add c1 ${user} "${password}"`, + `mc admin policy attach c1 ${policyName} --user ${user}`, + ].join(" && "); + // A transient failure here would otherwise print the FULL `mc admin ...` + // command line unredacted (both root and restricted-user passwords). + // Redacted rather than suppressed: `sh()` always scrubs MINIO_USER/ + // MINIO_PASSWORD; `redact: [password]` adds this call's own secret too. + const res = compose("exec", "-T", "minio", "/bin/sh", "-c", script, { redact: [password] }); + record(label, res.status === 0, res.stdout.trim().split("\n").pop()); +} + +main().catch((err) => { + console.error(err); + process.exitCode = 1; +}); diff --git a/runner/containers/o11y/loki/loki-config.filesystem.yaml b/runner/containers/o11y/loki/loki-config.filesystem.yaml new file mode 100644 index 0000000000..240c52c1a9 --- /dev/null +++ b/runner/containers/o11y/loki/loki-config.filesystem.yaml @@ -0,0 +1,88 @@ +{ + "auth_enabled": true, + "server": { + "http_listen_port": 3100, + "grpc_listen_port": 9095, + "log_level": "info" + }, + "common": { + "path_prefix": "/loki", + "replication_factor": 1, + "ring": { + "kvstore": { "store": "inmemory" } + } + }, + "ingester": { + "wal": { + "enabled": true, + "dir": "/loki/wal", + "flush_on_shutdown": true + }, + "chunk_idle_period": "1h", + "chunk_retain_period": "30s", + "max_chunk_age": "2h" + }, + "limits_config": { + "shard_streams": { "enabled": false }, + "max_global_streams_per_user": 20000, + "reject_old_samples": true, + "reject_old_samples_max_age": "7d", + "ingestion_rate_mb": 16, + "ingestion_burst_size_mb": 32, + "max_line_size": "256KB", + "otlp_config": { + "resource_attributes": { + "ignore_defaults": true, + "attributes_config": [ + { + "action": "index_label", + "attributes": [ + "service.name", + "deployment.environment.name", + "hot.surface", + "hot.tier", + "hot.framework", + "hot.ht_major", + "hot.outcome" + ] + } + ] + } + } + }, + "runtime_config": { + "file": "/etc/loki/runtime-config.yaml" + }, + "schema_config": { + "configs": [ + { + "from": "2024-01-01", + "store": "tsdb", + "object_store": "filesystem", + "schema": "v13", + "index": { "prefix": "index/", "period": "24h" } + } + ] + }, + "storage_config": { + "tsdb_shipper": { + "active_index_directory": "/loki/tsdb-index", + "cache_location": "/loki/tsdb-cache", + "resync_interval": "5m" + }, + "filesystem": { + "directory": "/loki/chunks" + } + }, + "compactor": { + "working_directory": "/loki/compactor", + "retention_enabled": false + }, + "querier": { + "max_concurrent": 10, + "query_ingesters_within": "168h" + }, + "analytics": { + "reporting_enabled": false + } +} diff --git a/runner/containers/o11y/loki/loki-config.yaml b/runner/containers/o11y/loki/loki-config.yaml new file mode 100644 index 0000000000..06efed8ab1 --- /dev/null +++ b/runner/containers/o11y/loki/loki-config.yaml @@ -0,0 +1,95 @@ +{ + "auth_enabled": true, + "server": { + "http_listen_port": 3100, + "grpc_listen_port": 9095, + "log_level": "info" + }, + "common": { + "path_prefix": "/loki", + "replication_factor": 1, + "ring": { + "kvstore": { "store": "inmemory" } + } + }, + "ingester": { + "wal": { + "enabled": true, + "dir": "/loki/wal", + "flush_on_shutdown": true + }, + "chunk_idle_period": "1h", + "chunk_retain_period": "30s", + "max_chunk_age": "2h" + }, + "limits_config": { + "shard_streams": { "enabled": false }, + "max_global_streams_per_user": 20000, + "reject_old_samples": true, + "reject_old_samples_max_age": "7d", + "ingestion_rate_mb": 16, + "ingestion_burst_size_mb": 32, + "max_line_size": "256KB", + "otlp_config": { + "resource_attributes": { + "ignore_defaults": true, + "attributes_config": [ + { + "action": "index_label", + "attributes": [ + "service.name", + "deployment.environment.name", + "hot.surface", + "hot.tier", + "hot.framework", + "hot.ht_major", + "hot.outcome" + ] + } + ] + } + } + }, + "runtime_config": { + "file": "/etc/loki/runtime-config.yaml" + }, + "schema_config": { + "configs": [ + { + "from": "2024-01-01", + "store": "tsdb", + "object_store": "s3", + "schema": "v13", + "index": { "prefix": "index/", "period": "24h" } + } + ] + }, + "storage_config": { + "tsdb_shipper": { + "active_index_directory": "/loki/tsdb-index", + "cache_location": "/loki/tsdb-cache", + "resync_interval": "5m" + }, + "aws": { + "endpoint": "${LOKI_S3_ENDPOINT}", + "region": "${LOKI_S3_REGION}", + "access_key_id": "${LOKI_S3_ACCESS_KEY_ID}", + "secret_access_key": "${LOKI_S3_SECRET_ACCESS_KEY}", + "bucketnames": "${LOKI_S3_BUCKET}", + "s3forcepathstyle": true, + "insecure": ${LOKI_S3_INSECURE} + } + }, + "compactor": { + "working_directory": "/loki/compactor", + "retention_enabled": false, + "delete_request_store": "s3" + }, + "querier": { + "max_concurrent": 10, + "query_ingesters_within": "168h" + }, + "analytics": { + "reporting_enabled": false + } +} diff --git a/runner/containers/o11y/loki/runtime-config.yaml b/runner/containers/o11y/loki/runtime-config.yaml new file mode 100644 index 0000000000..a8ec0ccec5 --- /dev/null +++ b/runner/containers/o11y/loki/runtime-config.yaml @@ -0,0 +1,10 @@ +{ + "overrides": { + "browser": { + "max_query_lookback": "720h" + }, + "worker": { + "max_query_lookback": "2160h" + } + } +} diff --git a/runner/containers/o11y/r2-lifecycle-rules.json b/runner/containers/o11y/r2-lifecycle-rules.json new file mode 100644 index 0000000000..560643e2bf --- /dev/null +++ b/runner/containers/o11y/r2-lifecycle-rules.json @@ -0,0 +1,36 @@ +{ + "rules": [ + { + "id": "o11y-loki-browser-chunks-30d", + "enabled": true, + "conditions": { "prefix": "browser/" }, + "deleteObjectsTransition": { + "condition": { "type": "Age", "maxAge": 2592000 } + } + }, + { + "id": "o11y-loki-worker-chunks-90d", + "enabled": true, + "conditions": { "prefix": "worker/" }, + "deleteObjectsTransition": { + "condition": { "type": "Age", "maxAge": 7776000 } + } + }, + { + "id": "o11y-loki-index-90d", + "enabled": true, + "conditions": { "prefix": "index/" }, + "deleteObjectsTransition": { + "condition": { "type": "Age", "maxAge": 7776000 } + } + }, + { + "id": "o11y-loki-state-30d", + "enabled": true, + "conditions": { "prefix": "state/" }, + "deleteObjectsTransition": { + "condition": { "type": "Age", "maxAge": 2592000 } + } + } + ] +} diff --git a/runner/containers/o11y/supervisor/entrypoint.sh b/runner/containers/o11y/supervisor/entrypoint.sh new file mode 100644 index 0000000000..2f5e3672b9 --- /dev/null +++ b/runner/containers/o11y/supervisor/entrypoint.sh @@ -0,0 +1,76 @@ +#!/bin/bash +# containers/o11y/supervisor/entrypoint.sh — PID 1 of the Grafana box. +# +# Starts Loki and Grafana as background children and is the ONLY process +# that receives container signals (no tini/`-g`, no `init: true`): a plain +# SIGTERM to the container reaches this trap, never Loki or Grafana +# directly, so the stop protocol in shutdown.sh always runs first. +# +# A SIGKILL bypasses this script entirely (nothing traps SIGKILL) — that is +# the point: no marker is ever written for an unclean stop (ADR-0041 exit +# criterion 12). +set -u + +# shellcheck source=./lib.sh +. /lib.sh +# shellcheck source=./shutdown.sh +. /shutdown.sh + +LOKI_CONFIG_FILE="/etc/loki/loki-config.yaml" +if [ "${STORAGE:-s3}" = "filesystem" ]; then + LOKI_CONFIG_FILE="/etc/loki/loki-config.filesystem.yaml" +fi + +LOKI_PID="" +GRAFANA_PID="" +STOPPING=0 + +term_handler() { + STOPPING=1 + log "received SIGTERM" + run_stop_protocol + local rc=$? + log "stop protocol finished with status $rc" + exit "$rc" +} +trap term_handler TERM + +log "wakeId=${WAKE_ID:-<unset>} storage=${STORAGE:-s3} config=${LOKI_CONFIG_FILE}" + +log "starting loki" +/usr/bin/loki -config.file="$LOKI_CONFIG_FILE" -config.expand-env=true & +LOKI_PID=$! + +log "starting grafana" +grafana server \ + --homepath="${GF_PATHS_HOME:-/usr/share/grafana}" \ + --config="${GF_PATHS_CONFIG:-/etc/grafana/grafana.ini}" \ + --packaging=docker \ + cfg:default.log.mode="console" \ + cfg:default.paths.data="${GF_PATHS_DATA:-/var/lib/grafana}" \ + cfg:default.paths.logs="${GF_PATHS_LOGS:-/var/log/grafana}" \ + cfg:default.paths.plugins="${GF_PATHS_PLUGINS:-/var/lib/grafana/plugins}" \ + cfg:default.paths.provisioning="${GF_PATHS_PROVISIONING:-/etc/grafana/provisioning}" & +GRAFANA_PID=$! + +log "loki pid=${LOKI_PID} grafana pid=${GRAFANA_PID}" + +# Reap an unexpected crash of either child (not triggered by our own +# SIGTERM) as a hard failure, instead of hanging forever with one process +# left running. +while true; do + wait -n "$LOKI_PID" "$GRAFANA_PID" + rc=$? + if [ "$STOPPING" -eq 1 ]; then + # term_handler already ran (and will exit); nothing left to do here. + exit "$rc" + fi + if ! kill -0 "$LOKI_PID" 2>/dev/null; then + log "loki exited unexpectedly with code $rc" + exit "$rc" + fi + if ! kill -0 "$GRAFANA_PID" 2>/dev/null; then + log "grafana exited unexpectedly with code $rc" + exit "$rc" + fi +done diff --git a/runner/containers/o11y/supervisor/lib.sh b/runner/containers/o11y/supervisor/lib.sh new file mode 100644 index 0000000000..6ffb6ad2a3 --- /dev/null +++ b/runner/containers/o11y/supervisor/lib.sh @@ -0,0 +1,79 @@ +#!/bin/bash +# Shared helpers for entrypoint.sh and shutdown.sh. Sourced, never executed +# directly. Bash (not POSIX sh): the base image ships bash, and the trap / +# `wait -n` behaviour the supervisor needs is bash's, not busybox ash's. +set -u + +log() { + # A structured-ish line on stdout; this box exports no Workers Logs of its + # own (ADR-0041 §B.6) — this is operator-visible container stdout only. + printf '%s supervisor: %s\n' "$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)" "$*" +} + +# Build the base curl args for a signed request against the Loki bucket. +# The credentials reach the box as LOKI_S3_* envVars, scoped to that bucket +# only (ADR-0041 §A) — the shutdown script never touches any other bucket. +r2_curl_base() { + local scheme="https" + if [ "${LOKI_S3_INSECURE:-false}" = "true" ]; then + scheme="http" + fi + R2_BASE_URL="${scheme}://${LOKI_S3_ENDPOINT}/${LOKI_S3_BUCKET}" + R2_SIGV4_ARGS=(--aws-sigv4 "aws:amz:${LOKI_S3_REGION:-auto}:s3" \ + --user "${LOKI_S3_ACCESS_KEY_ID}:${LOKI_S3_SECRET_ACCESS_KEY}") +} + +# r2_list_prefix <prefix> — object keys under a prefix, one per line, via +# the S3 ListObjectsV2 XML API. Proves the index actually landed in the +# bucket (never trust a local directory or a 202-style POST /flush). +# +# A curl failure, non-XML/error response, or truncated listing must all be +# treated as "cannot confirm," never as "found nothing" — not solved with +# `set -o pipefail` alone, since a genuinely empty listing also makes +# `grep -o` exit 1. Instead: capture curl's own exit status explicitly, +# require a real `<ListBucketResult` root, and refuse a truncated listing. +# Every caller must check this function's OWN exit status, never +# `$(... || true)`. +r2_list_prefix() { + local prefix="$1" + r2_curl_base + local body + body="$(curl -fsS --max-time 15 "${R2_SIGV4_ARGS[@]}" \ + "${R2_BASE_URL}/?list-type=2&prefix=$(printf '%s' "$prefix" | sed 's/\//%2F/g')")" || return 1 + case "$body" in + *"<ListBucketResult"*) ;; + *) + log "r2_list_prefix: response for prefix '${prefix}' has no <ListBucketResult> root — treating as a failure, not an empty listing" + return 1 + ;; + esac + if printf '%s' "$body" | grep -q '<IsTruncated>true</IsTruncated>'; then + log "r2_list_prefix: truncated listing for prefix '${prefix}' (over 1000 keys) — cannot confirm the full set" + return 1 + fi + printf '%s' "$body" | grep -o '<Key>[^<]*</Key>' | sed -e 's/<Key>//' -e 's#</Key>##' + return 0 +} + +# r2_put <key> <file> — PUT then HEAD to confirm (200s only; never trust +# the PUT response alone). +r2_put_and_verify() { + local key="$1" file="$2" + r2_curl_base + local put_code + put_code=$(curl -sS --max-time 15 -o /dev/null -w '%{http_code}' \ + "${R2_SIGV4_ARGS[@]}" -X PUT --data-binary "@${file}" \ + "${R2_BASE_URL}/${key}") + if [ "$put_code" != "200" ]; then + log "marker PUT failed: HTTP ${put_code}" + return 1 + fi + local head_code + head_code=$(curl -sS --max-time 15 -o /dev/null -w '%{http_code}' \ + "${R2_SIGV4_ARGS[@]}" -I "${R2_BASE_URL}/${key}") + if [ "$head_code" != "200" ]; then + log "marker HEAD confirmation failed: HTTP ${head_code}" + return 1 + fi + return 0 +} diff --git a/runner/containers/o11y/supervisor/shutdown.sh b/runner/containers/o11y/supervisor/shutdown.sh new file mode 100644 index 0000000000..33c0937718 --- /dev/null +++ b/runner/containers/o11y/supervisor/shutdown.sh @@ -0,0 +1,165 @@ +#!/bin/bash +# Sourced by entrypoint.sh, never executed directly. Defines run_stop_protocol, +# called from entrypoint's SIGTERM trap with LOKI_PID/GRAFANA_PID as globals. +# +# ADR-0041 §A stop protocol: stop Loki (SIGTERM, wait for a 0 exit) -> confirm +# THIS instance's TSDB index landed in R2 (never trust an empty local dir or +# POST /flush, which answers before anything is written) -> only then write +# state/wakes/<wakeId>/clean -> stop Grafana. Fail-closed; WAKE_ID missing is +# refused outright. +# +# An uploader-name match alone is not proof of a NEW upload (the shipper also +# uploads periodically while Loki runs), so the check snapshots uploader-named +# keys BEFORE SIGTERM and requires a key NOT in that snapshot after exit. A +# false "unclean" is the safe failure direction (a replay costs a dedupe'd +# duplicate, ADR-0041 §B.3). + +STOP_GRACE_SECONDS="${O11Y_STOP_GRACE_SECONDS:-30}" + +# Loki's own `reject_old_samples_max_age: 7d` means a backlogged/reopened +# record can write a NEW index table up to 7 days in the past; this bounds +# how far back a stop check must look. Overridable for tests. +INDEX_DAY_SPAN_DAYS="${O11Y_INDEX_DAY_SPAN_DAYS:-7}" + +# The snapshot-diff decision lives in its own testable functions (separate +# from `run_stop_protocol`, which needs a real LOKI_PID to exercise at all). +# See `pipeline/o11y-shutdown-snapshot.test.mjs` for the deterministic proof. + +# snapshot_index_keys <day_now> — every uploader-named key under EVERY day +# prefix the wake's pushed records could span (`day_now` down through +# `day_now - INDEX_DAY_SPAN_DAYS`, inclusive), one per line. Prints nothing and returns 1 if ANY +# one day's listing could not be confirmed (r2_list_prefix itself fails +# closed — see lib.sh) — the caller MUST treat that as "cannot confirm," +# never as "confirmed empty." +snapshot_index_keys() { + local day_now="$1" + local keys="" offset day listing + for offset in $(seq 0 "$INDEX_DAY_SPAN_DAYS"); do + day=$((day_now - offset)) + if ! listing="$(r2_list_prefix "index/index/${day}/")"; then + return 1 + fi + keys="${keys}${listing} +" + done + printf '%s' "$keys" + return 0 +} + +# confirm_new_upload <uploader_name> <before_keys> <after_keys> — true (0) +# iff a key bearing <uploader_name> is new in <after_keys> vs <before_keys>. +confirm_new_upload() { + local uploader_name="$1" before_keys="$2" after_keys="$3" + local before_matches after_matches new_matches + before_matches="$(printf '%s\n' "$before_keys" | grep -F "$uploader_name" | sort -u || true)" + after_matches="$(printf '%s\n' "$after_keys" | grep -F "$uploader_name" | sort -u || true)" + new_matches="$(comm -13 <(printf '%s\n' "$before_matches") <(printf '%s\n' "$after_matches"))" + [ -n "$new_matches" ] +} + +run_stop_protocol() { + local marker_ok=1 + + if [ -z "${WAKE_ID:-}" ]; then + log "refusing stop protocol: WAKE_ID is not set" + fi + + # --- 1. stop Loki gracefully ------------------------------------------- + # The uploader-name path is overridable via LOKI_UPLOADER_NAME_FILE, which + # makes `run_stop_protocol` itself directly testable + # (`pipeline/o11y-shutdown-snapshot.test.mjs`) without a real `/loki` filesystem. + local uploader_name_file="${LOKI_UPLOADER_NAME_FILE:-/loki/tsdb-index/uploader/name}" + local loki_exit=1 + if [ -n "${LOKI_PID:-}" ] && kill -0 "$LOKI_PID" 2>/dev/null; then + local uploader_name="" + if [ -r "$uploader_name_file" ]; then + uploader_name="$(cat "$uploader_name_file" 2>/dev/null || true)" + fi + if [ -z "$uploader_name" ] && [ "${STORAGE:-s3}" = "s3" ]; then + log "could not read ${uploader_name_file} before stop — index upload cannot be confirmed" + fi + + # Snapshot BEFORE SIGTERM: a periodic shipper upload can already exist + # this wake, so "does one exist" after exit isn't proof of the FINAL + # upload. `snapshot_ok` ensures a FAILED listing is never read as empty. + local day_now before_keys="" snapshot_ok=0 + if [ "${STORAGE:-s3}" = "s3" ] && [ -n "$uploader_name" ]; then + day_now=$(( $(date -u +%s) / 86400 )) + if before_keys="$(snapshot_index_keys "$day_now")"; then + snapshot_ok=1 + else + log "pre-SIGTERM index listing failed — cannot confirm a new upload this wake; will refuse the marker" + fi + fi + + log "sending SIGTERM to loki (pid $LOKI_PID)" + kill -TERM "$LOKI_PID" 2>/dev/null + + local waited=0 + while kill -0 "$LOKI_PID" 2>/dev/null; do + if [ "$waited" -ge "$STOP_GRACE_SECONDS" ]; then + log "loki did not exit within ${STOP_GRACE_SECONDS}s of SIGTERM; giving up on a clean marker" + break + fi + sleep 1 + waited=$((waited + 1)) + done + + if ! kill -0 "$LOKI_PID" 2>/dev/null; then + wait "$LOKI_PID" + loki_exit=$? + log "loki exited with code $loki_exit after ${waited}s" + else + loki_exit=1 + fi + + # --- 2. confirm THIS instance uploaded a NEW index object ------------ + # Requires `snapshot_ok` too — a stop that could not even confirm the + # BEFORE state must never write a marker, no matter how the + # after-listing or Loki's own exit code turn out. + if [ "$loki_exit" -eq 0 ] && [ -n "${WAKE_ID:-}" ] && [ "${STORAGE:-s3}" = "s3" ] && [ -n "$uploader_name" ] && [ "$snapshot_ok" -eq 1 ]; then + local after_keys + if after_keys="$(snapshot_index_keys "$day_now")"; then + if confirm_new_upload "$uploader_name" "$before_keys" "$after_keys"; then + log "new index upload confirmed (not present before SIGTERM)" + marker_ok=0 + else + log "no index object bearing uploader name '${uploader_name}' is new since before SIGTERM under index/index/{$((day_now - INDEX_DAY_SPAN_DAYS))..${day_now}}/ — not writing a marker (plan B territory, see ADR-0041 exit criterion 1)" + marker_ok=1 + fi + else + log "post-exit index listing failed — cannot confirm a new upload this wake; treating this stop as unclean" + marker_ok=1 + fi + else + marker_ok=1 + fi + else + log "loki is not running; nothing to stop, no marker" + fi + + # --- 3. write the clean-shutdown marker, only if every check held ------ + if [ "$marker_ok" -eq 0 ]; then + local marker_file + marker_file="$(mktemp)" + printf '{"wakeId":"%s","at":"%s"}' "$WAKE_ID" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" > "$marker_file" + if r2_put_and_verify "state/wakes/${WAKE_ID}/clean" "$marker_file"; then + log "clean-shutdown marker written: state/wakes/${WAKE_ID}/clean" + marker_ok=0 + else + log "marker PUT/HEAD did not both return 200 — treating this stop as unclean" + marker_ok=1 + fi + rm -f "$marker_file" + fi + + # --- 4. stop Grafana ----------------------------------------------------- + if [ -n "${GRAFANA_PID:-}" ] && kill -0 "$GRAFANA_PID" 2>/dev/null; then + log "sending SIGTERM to grafana (pid $GRAFANA_PID)" + kill -TERM "$GRAFANA_PID" 2>/dev/null + wait "$GRAFANA_PID" 2>/dev/null + log "grafana exited with code $?" + fi + + return "$marker_ok" +} diff --git a/runner/docs/TESTING.md b/runner/docs/TESTING.md index 328360b673..3e05b33f87 100644 --- a/runner/docs/TESTING.md +++ b/runner/docs/TESTING.md @@ -109,6 +109,8 @@ covers the dependency: | `E2E_API_TOKEN` | An authed write round-trip against the deployed API (`share-create-live.spec.ts`, `session-abandoned-create.spec.ts`) | `e2e-live.yml` | A persistent API token minted on `/api-tokens` (ADR-0037). It does not expire, so a token that stops validating **fails** the run — it means revoked or broken. Absent, the step is skipped. | | `E2E_AI=1` | A live LLM answer (`ai-live.spec.ts`, lands with [#187](https://github.com/handsontable/examples/pull/187)) | `e2e-live.yml` weekly canary | Real budget, shared 8/min-per-IP rate bucket — a 429 skips rather than fails. | | `E2E_STARTER_MATRIX=1` | Every starter × major through a live container session | `e2e-starter-matrix.yml` (manual + monthly) | Serialized against the global container cap; never fold matrix cases into the PR suite. | +| `E2E_TELEMETRY=1` | A `VITE_TELEMETRY_LOCAL=1` build (contract §10) | `page.route`-mocked specs (`telemetry-faro.spec.ts`, `example-analytics.spec.ts`) run in `ci.yml`'s `e2e-telemetry` job, every PR. Combined with `E2E_LIVE=1`, `telemetry-metrics.spec.ts` needs a real local API worker + Docker instead, and runs in `e2e-o11y-local.yml` (manual + nightly + o11y-path PRs) | Same gate, two very different cost profiles — mocked network vs. a real `wrangler dev` — hence two different workflow homes. | +| `E2E_O11Y_LOCAL=1` | The whole local o11y stack: Docker compose (ClickHouse/MinIO), a real o11y `wrangler dev`, a real API `wrangler dev`, applied D1 migrations | `e2e-o11y-local.yml` (manual + nightly + o11y-path PRs) | `o11y-local.spec.ts` proves telemetry reaches the real ingest pipeline end to end, never mocked — the standing check that T02–T08's wiring still holds. | Two rules that keep gates honest: diff --git a/runner/docs/adr/0040-hourly-buckets-and-pool-pressure.md b/runner/docs/adr/0040-hourly-buckets-and-pool-pressure.md index c1c5af3013..9322682491 100644 --- a/runner/docs/adr/0040-hourly-buckets-and-pool-pressure.md +++ b/runner/docs/adr/0040-hourly-buckets-and-pool-pressure.md @@ -1,6 +1,6 @@ # ADR-0040: Hour-of-day buckets, and measuring pool pressure -**Status:** Accepted (extends [ADR-0022](0022-self-enforced-spend-ceiling.md)) +**Status:** Accepted (extends [ADR-0022](0022-self-enforced-spend-ceiling.md)); decisions A, B, C.2 and C.3 superseded by [ADR-0041](0041-observability-stack.md) — hour buckets, awake-seconds per hour and sampled peak concurrency become Workers Analytics Engine points instead of D1 rows. Decision C.1 stands as written (`at_capacity` as a `usage_daily` counter, also emitted as a `session.start` outcome point), and so does D (privacy). None of A–C was implemented before supersession. ## Context diff --git a/runner/docs/adr/0041-observability-stack.md b/runner/docs/adr/0041-observability-stack.md new file mode 100644 index 0000000000..e4e7e55adb --- /dev/null +++ b/runner/docs/adr/0041-observability-stack.md @@ -0,0 +1,974 @@ +# ADR-0041: Observability on Cloudflare — a sleeping Loki + Grafana box, OTLP inward, Sentry for uncaught errors + +**Status:** Proposed — design approved 2026-09-23 (revision 3), implemented, local +end-to-end walkthrough and every sandbox probe complete (§L "Results"). 12 of 15 exit +criteria pass outright with real evidence; criteria 5 and 13 pass pending the two +confirmations named below; criterion 8 (Volume) is **Mixed**, not a pass — Analytics +Engine points and the raw Workers Logs pool both pass at 10× with real margin, but the +exported-logs allotment does not (§D above has the numbers and the fallback). The +design's own §L trigger (criterion 1, 2 or 7 failing) is not engaged. +**Stays Proposed, not Accepted, pending exactly two items**: exit criterion 5's +CPU/memory measurement inside a real Workers isolate (every measurement so far is a +Node-process proxy — no isolate profiling access was available), and exit criterion 13's +real-object retention expiry (a 1-day R2 lifecycle test is running against real objects; +calendar time has not yet passed as of this pass — see `docs/run-and-deploy.md`'s +Launch plan for how to close both). +Supersedes ADR-0040 decisions A, B, C.2 and C.3; amends ADR-0022 (o11y spend cap, +per-script billing rows), ADR-0038 (WAF exception extended to `/telemetry/*`); adds +routes under ADR-0020. **No longer deviates from ADR-0007**: `/grafana/*` gates +through the same Handsontable login broker as every other internal surface, not +Cloudflare Access — see §H. ADR-0042 ships with this +ADR, and stays at the same status (Proposed) until this one flips to Accepted. +ADR-0043 follows after launch (not yet dispatched). + +## Context + +The runner reports faults through Sentry and nothing else. Both SDKs run errors-only, +Workers Logs keeps seven days at a 10 % head sample, and every number the team looks +at is a D1 counter aggregated at write time (`usage_daily`, `analytics_daily`, +`cost_ledger`) and rendered by `/admin`. There are no traces, no metric history, no +browser performance data, and the only join between a Sentry issue and a Worker log is +the `cf-ray` header, promoted to a Sentry tag by hand. + +The missing thing is the ability to answer, from data, the questions the runner is run +on: how long an example takes to show a grid, by tier and Handsontable major; how full +the container pool is at 14:00 UTC; whether the last release raised compile errors on +the `next` bucket; what the AI assistant costs per answer; which docs guides people +reach for a live example of (ADR-0042). + +Constraints: + +1. **Self-hosted, on Cloudflare, next to the app**, in the main account (ADR-0010). No + VPS, no Grafana Cloud, no SaaS beyond Sentry. +2. **OpenTelemetry protocol as the contract** from the Worker inward. +3. **Grafana as the UI, Grafana Faro as the browser SDK.** +4. **Sentry stays connected for uncaught errors**, and those errors also land in the new + stack. +5. **Cost at app scale.** The runner's containers cost $1–13 a month because they sleep; + observability follows the same pattern. +6. **Anonymous by construction** stays in force for analytics (`analytics.ts`, + AGENTS.md). Operational logs are governed by the explicit rule in §E.4. +7. **Deploy only after localhost tests**, from one feature branch (decision 2026-09-23). + Facts that exist only on real Cloudflare are measured with throwaway probes on the + sandbox account, never the production account (§L). + +Platform facts that shaped the design (verified 2026-09-22/23, including Loki, Sentry +and Faro source): + +- **Workers tracing and export** (open beta): auto spans for fetch, D1, KV, R2, DO, + handlers, RPC; custom spans via `tracing.enterSpan`; OTLP export of traces and logs to + any endpoint with custom headers. From 2026-10-01 spans are billed as events **in the + same 20 M/month Workers Logs pool** as log lines, and export includes **10 M events per + signal**. No `spanContext()`, so no browser-to-worker `traceparent` join. No metrics + export. Sampling is per invocation and per Worker, never per route. Export retry + semantics are undocumented. Fetch-handler spans carry `url.full`, + `user_agent.original`, city and ASN; preview hostnames carry the sandbox token. +- **Workers Logs**: every sampled invocation writes an invocation log plus each + `console.*` line; `observability.logs.invocation_logs: false` removes the former. The + Sandbox SDK logs a warning on every request to a stale preview URL, and Tier-2 + container stdout lands in the API worker's logs. +- **Containers** bill CPU on use and memory/disk on provisioned size, nothing while + asleep. Disk is ephemeral. Inbound only via Worker/DO. `jurisdiction: "eu"` placement + exists since 2026-04-05. `onStop` reports a host loss the DO did not observe as + `{ exitCode: 0, reason: "exit" }`, indistinguishable from a clean exit. Cloudflare + "does not guarantee that any container instance will run for any set period of time." +- **R2** is S3-compatible; EU jurisdiction on buckets; lifecycle rules work by key prefix. +- **Loki 3.x** ingests OTLP natively and synchronously; `otlp_config` can label only + **resource** attributes. Query-time dedupe needs same stream, same timestamp, same line + **and** same structured metadata (`TestMergeIteratorNoDedupDifferentStructuredMetadata`), + and metric queries dedupe only across iterators, so two copies inside one chunk are + both counted. When an OTLP record has no timestamp, Loki stamps `time.Now()`. + `POST /flush` returns before anything is written. The TSDB index head rotates every + 15 minutes and uploads afterwards. The compactor rewrites index only, and its retention + markers live on local disk. Default push limits are 4 MB/s with a 6 MB burst; + `shard_streams` is on by default with shard numbers from in-memory state; + `query_ingesters_within` defaults to 3 h. +- **Grafana Faro** v2.12: `faro.receiver` stamps every entry with `time.Now()`, answers 202 + even when its exporter fails, drops after a 2 s timeout, rate-limits at 50 req/s, and + looks maps up by URL path. `persistent: false` session tracking uses `sessionStorage`. + The Performance and CSP instrumentations send full URLs with query strings. +- **Workers Analytics Engine**: 20 blobs + 20 doubles + 1 index per point, 3-month + retention, 10 M data points and 1 M read queries per month included (currently not + billed), SQL API; it samples at write time and read time (`_sample_interval`). + Grafana reads it through the Altinity ClickHouse plugin. No local emulation. +- **Sentry**: Team plan, so issue-alert webhooks only, no per-event `error.created`. The + API worker's fetch catch-all converts every throw into an explicit `captureException` + plus a 500, and the snapshot-job alarm reports without rethrowing, so almost no Worker + error is "uncaught" in the SDK's sense. Spend alerts reach anyone only through + `Sentry.captureMessage` in `reconcile.ts`. React 19 render crashes caught by + `Sentry.ErrorBoundary` never reach `window.onerror`. +- **Edge**: the zone rule behind ADR-0038 403s any body containing `<script` outside + `/api/*`; `workers_dev` defaults to on and bypasses zone rules. + +Alternatives considered and rejected: + +- **VPS with the LGTM stack** — outside constraint 1. +- **Serverless store (Pipelines → Iceberg → R2 SQL)** — no LogQL, two open-beta + services. Kept as the escape hatch (§L). +- **Cloudflare's own Workers Observability as the store** — 7-day, non-EU, not Grafana. + Kept as the temporary safety net (`persist: true`, §H). +- **Alloy with `faro.receiver` in the box** (revision 2's design) — `time.Now()` + stamping makes every browser record land at drain time and never dedupe on re-replay; + 202-on-failure makes the drain's ledger lie; its rate limit refuses backlog replays; its + path-based map lookup breaks per-release storage. Once the Worker converts Faro to + OTLP itself (§C.1), Alloy has nothing left to do: the drain pushes straight to Loki. +- **Vanilla OpenTelemetry JS in the browser** — experimental SDKs, no bundled error or + vitals capture. Faro stays as the capture SDK; only its wire format stops at the Worker. +- **Mimir** — a WAL on ephemeral disk and ingestion only while awake. Analytics Engine + ingests while the box sleeps. +- **Tempo, and exporting traces at all, in the first release** — no browser join exists, + span attributes carry URLs, user agents, geo and preview tokens, and spans compete with + logs for the same 20 M pool. Deferred (§C.4). +- **A Queue in front of the inbox** — billed per message operation and a second write + path next to the inbox Durable Object, which already batches. +- **Self-hosted Sentry or GlitchTip**; **`@microlabs/otel-cf-workers`** — not the Grafana + stack; deprioritised by its maintainer. + +## Decision + +### A. One sleeping box: Loki + Grafana in a Container; R2 is the store + +A new Worker `handsontable-demos-o11y` (`workers/o11y/`, own `wrangler.jsonc` with +`workers_dev: false` and `preview_urls: false`, own deploy job) owns two Durable Object +classes: `InboxWriter` (§B) and `GrafanaBox`, a Container class running **Loki and +Grafana** on `standard-1`, pinned with `jurisdiction: "eu"`. There is no collector in the +box. + +- **Loki**, single binary, TSDB single store in an EU R2 bucket, two tenants: + `browser` (browser and beacon streams) and `worker` (Worker streams). Configuration in + §B.4. +- **Grafana** served from `/grafana/`, provisioned from git (one Loki datasource per + tenant via `X-Scope-OrgID`, the Altinity plugin against the Analytics Engine SQL API, + dashboards), `auth.proxy` behind the Worker (§B.5), **Grafana Live disabled**, sqlite + state disposable. +- **Secrets reach the container** only as `envVars` set by `GrafanaBox` at start from + Worker secrets (`LOKI_S3_*`, `AE_SQL_TOKEN`). The Slack webhook never enters the box. + +**Wake.** Two triggers only: + +1. A Grafana visit through the login broker (ADR-0007 — not Cloudflare Access). The + Worker renews the activity timer on every HTTP + request to `/grafana/*`, **including the waking page's own `meta refresh` poll while + the box is still booting** (counting a request only once the box is ready would let + a visit wake with nothing yet in the backlog SIGTERM itself around 20s into boot, + before the person who opened it ever saw Grafana); the box stops after 15 idle + minutes, and after 4 hours awake regardless (the next request shows the waking page). +2. A backlog: a `*/10` cron in the o11y worker asks `InboxWriter.backlog()` and wakes the + box when the oldest uncommitted object is older than 60 minutes or the backlog exceeds + 64 MB. The cron never wakes the box when drains are paused (§G). + +**Stop protocol**, the same for every stop the Worker initiates: the drain finishes; if +no Grafana request arrived in the last 10 minutes the Worker calls `stop()` (otherwise +the idle timer does, later, so a drain never SIGTERMs someone reading a dashboard); the +container's shutdown script stops Loki gracefully, confirms the index is uploaded to R2, +and only then writes a **clean-shutdown marker** `state/wakes/<wakeId>/clean` into the Loki +bucket, the one bucket its credentials reach. The o11y worker reads `state/` through an R2 +binding on the same bucket; no lifecycle rule touches that prefix except a 30-day expiry. +The marker, not `onStop`, is what the ledger trusts (§B.3). **Exception:** a +wake that ends without ever having a `provisional` key (an empty backlog, or a +Grafana-visit-only wake with nothing to drain) never gets an index upload and so never +gets a marker — that is expected, not an unclean stop, and the ledger now resolves such a +wake as clean without requiring one. A wake that *did* push data still requires the real +marker. + +**Waking page**: the Handsontable logo, one line of text, `<meta http-equiv="refresh" +content="3">`, no script, served by the Worker while the box is not ready. + +**Cost model**: drain wakes × measured drain-wake duration + visit hours, at +$0.038–0.074 per awake hour on `standard-1`. The target is ≈ $5–8/month at current +traffic; exit criterion L.7 recomputes it from the measured drain-wake duration and +fails above $10. **Measured on the real sandbox platform, corrected 1×/10× traffic +scale (§L.7): $0.21/month at 1×, $0.33/month at 10×** — the design's own $5–8 target was itself a +conservative upper estimate; drain-wake frequency is capped by the 60-minute backlog-age +trigger, not by traffic volume, so 20× more records only adds ~16s of drain time per wake, +not 20× the awake-hour cost. **Per forgotten tab**: every provisioned dashboard ships with +auto-refresh off, so an open tab left idle costs nothing beyond the visit that opened it; a +viewer who turns refresh back on keeps the box awake for as long as the tab stays open, up +to the 4-hour hard cap — $0.15–$0.30 per forgotten tab at the awake-hour rate above. + +### B. Ingest never waits for the box + +**B.1 Routes**, all owned by the o11y worker on the main hostname, passed as `--routes` +flags (ADR-0020), beside the API worker's `/api/*`, `/d/*`, `/embed/*`: + +| Route | Source | +|---|---| +| `POST /telemetry/collect` | Faro from the authoring app | +| `POST /telemetry/lite` | lite beacon from `/d` and `/embed` | +| `POST /telemetry/v1/logs` | Cloudflare OTLP log export | +| `POST /telemetry/deploy` | CI deploy events | +| `POST /telemetry/hooks/sentry` | Sentry issue-alert webhook | +| `/grafana/*` | Grafana, waking page | +| `POST /grafana/_o11y/reopen` | manual ledger re-open (§B.3) | + +ADR-0038's WAF exception is extended from `/api/*` to `/telemetry/*`: Faro errors and +Sentry payloads legitimately contain `<script`. The compensating controls are the gates +in §B.5. + +**B.2 `InboxWriter`: one owner of the inbox lifecycle.** One instance, named `main`, +addressed through `.jurisdiction("eu")`. The CPU-heavy steps run in the **stateless route +handler**, so ingest is not serialised through one object; `InboxWriter` only checks, +stores and packs. For each accepted request, in this order: + +1. **Decode and scrub** in the route handler into OTLP log records (§C.1): convert Faro + and beacon payloads; decode Cloudflare's export (protobuf or JSON, whichever spike (b) + observes); keep only allowlisted attributes; hoist `hot.*` and `service.*` to resource + attributes; run the server-side scrubber (§E.4). Nothing derived from the arrival time + is added yet. Records over 256 KB are dropped. +2. **Hash** each record (SHA-256) at this point, before any arrival-time value exists, so + a redelivered body hashes identically. +3. **Stamp** timestamps (§C.2), which may use the arrival time as a clamp or fallback. + The arrival time itself never becomes part of a stored record; `InboxWriter` keeps it + on the storage row. +4. **Deduplicate** in `InboxWriter`: each hash is checked against a 24-hour set in DO + storage; a record already seen is dropped. This is what makes a duplicate Cloudflare + delivery a non-event. +5. **Append** the records to DO SQLite storage in rows of at most 1 MB, and answer `2xx` + only after the transaction commits. Nothing is held in memory across requests. +6. **Pack**, from a 60-second alarm or at 4 MB stored: write one gzipped NDJSON object + per tenant to `inbox/<tenant>/<yyyy-mm-dd>/<hh>/<seq>.ndjson.gz`, where `<seq>` is a + counter persisted in DO storage, incremented in the same transaction that records the + key, zero-padded to 12 digits. The key's state is recorded as `written`; the packed + rows are deleted. "4 MB stored" is a real, enforced bound on the packed object's own + decompressed NDJSON size (`PACK_OBJECT_MAX_DECOMPRESSED_BYTES`), not only a + flush-cadence hint — the alarm takes pending rows in arrival order up to that budget + (always at least one row, even if a single row alone is over budget) and commits that + object; a tenant with more pending rows than fit in one object is packed across several + objects, looping within the same alarm invocation up to a per-invocation cap on packed + objects and rescheduling itself immediately when rows remain, rather than one unbounded + in-memory gzip per alarm. The alarm's own read of pending rows is bounded too: `row:<n>` + is zero-padded (12 digits, matching `<seq>`'s own width), so native ascending key order + equals arrival order without an in-memory sort, and the alarm pages `row:` in small + chunks, accumulated up to one packed object's own byte budget per round, rather than + one unbounded `list()` — see contract §8 and `workers/o11y/src/inbox/pack.ts`. + +One writer means key order equals arrival order; storage-backed buffering means a +deploy, eviction or host restart between two alarms loses nothing. + +Analytics Engine points for browser metrics are written by the route handler after step 3 +(§F.1), so they exist while the box sleeps. `example.*` events (ADR-0042) produce Analytics Engine points only +and are never packed into the inbox. The exact first-seen registry for error fingerprints +(§F.3) is updated at step 4. + +**B.3 Drain and ledger.** The ledger lives in `InboxWriter` storage, next to the keys it +describes, so backlog and state are readable without starting the container. Each key is +`written` → `provisional(wakeId)` → `committed`, or `rejected`. + +- **At each cron tick and at the start of each wake**, `InboxWriter` resolves every wake + that still owns provisional keys and is **over** — a newer `wakeId` has started, or + `GrafanaBox`'s container state reports it not running: marker present → those keys + become `committed`; marker absent → they go back to `written` (re-opened). A wake that is + still running is left alone. `POST /grafana/_o11y/reopen` re-opens a time window by hand. +- `backlog()` counts only `written` keys, after that resolution step, so a box kept awake + by a visitor never triggers wake attempts for its own provisional keys, and a crashed + wake's keys count again as soon as the crash is noticed. +- **The drain** runs in `GrafanaBox`, as a Durable Object alarm loop that handles a bounded + number of objects per invocation and reschedules itself, under the o11y worker's + `limits.cpu_ms` (set to the value spike (b) needs, at most 300 000). It runs only on a + freshly woken box, before anything else is pushed: it + replays re-opened keys first, then new `written` keys, in key order, each object's + records pushed to Loki's `/otlp/v1/logs` with the tenant header in requests of at most + 1 MB decompressed. A key becomes `provisional(wakeId)` only after every one of its + requests returned `2xx`. `429` and `5xx` are retried with backoff within the wake; a + `400` (for example `too_far_behind`) is logged with Loki's message. Two cases defer a + key instead (it stays `written`, nothing is rejected, the batch goes on): a `429` whose + body names Loki's stream limit, which is never retried, and an inbox object that + cannot be read. A stream-limited tenant's keys are skipped for the rest of the wake so + the other tenant keeps draining (contract §8, "Drain refusals"). A single too-old + record inside an otherwise-good packed object does not 400 (and so reject) the whole + key — `drainKey` drops individual log records older than `reject_old_samples_max_age` + minus a margin *before* pushing, counts + the dropped ones on the `o11y.drain` point, and still pushes the good siblings in the + same key. A key is marked `rejected` only when the push itself still 400s after that + filtering (a genuine, not-just-stale, rejection), which raises an alert (§F.3). Within + one wake, a per-record hash set guarantees no record is pushed twice. +- **What an unclean stop costs.** The whole wake's keys are replayed on the next wake. + Data that Loki had already indexed before the crash is then stored twice, in different + chunks; queries return it once, because timestamps, labels and structured metadata are + deterministic (§C.2) and `shard_streams` is off. The duplicate storage stays until R2 + lifecycle deletes it, not until compaction. Re-opening across a Loki configuration or + label change is not allowed, because it would break that determinism. + +**B.4 Loki configuration**: `ingester.wal.flush_on_shutdown: true`; +`limits_config.shard_streams.enabled: false`; `max_chunk_age: 2h` (a 1-hour out-of-order +window, enough because the drain replays in key order into an empty ingester); +`reject_old_samples_max_age: 7d`; `query_ingesters_within: 168h`, so backfilled data is +queryable while the box is awake; `ingestion_rate_mb` and `ingestion_burst_size_mb` +raised to what spike (b) measures, at least 16/32; `max_line_size: 256KB`; +`otlp_config` promoting the `hot.*` resource attributes to labels. **Retention through +R2 lifecycle, not the compactor**: one rule per tenant chunk prefix (`browser` 30 days, +`worker` 90 days) and the index prefix at 90 days, with `max_query_lookback` per tenant so +nothing past retention is queried. For days 31–90 the shared index still lists `browser` +chunks that lifecycle has already deleted; `max_query_lookback: 30d` on that tenant means +no query ever reaches those entries, and the index bytes they cost are negligible, so the +index is not split per tenant. Compactor retention stays off: its markers would live +on ephemeral disk, and `retention_delete_delay` outlasts any wake. + +**B.5 Gates**, every route authenticated or gated: + +| Route | Gate | +|---|---| +| `collect`, `lite` | `Origin`/`Referer` host is the production host (or `localhost` in the `local` environment), payload environment matches, `BOT_RE` user-agent filter, size caps, item-kind allowlist, unknown attributes dropped, the Workers rate-limiting binding, then the server-side scrubber. `navigator.webdriver` is a browser-side check only (§E.4). | +| `v1/logs` | `x-o11y-secret` header set on the export destination, constant-time compare | +| `deploy` | GitHub OIDC token (issuer, audience, repository, workflow), secret fallback | +| `hooks/sentry` | `sentry-hook-signature` HMAC | +| `/grafana/*`, `reopen` | the Worker's own HMAC-signed session cookie (`O11Y_SESSION_SECRET`), minted once from a Handsontable login broker token (ADR-0007, K1 — not Cloudflare Access); a client-sent `auth.proxy` header is stripped | + +Every drop writes an `o11y.ingest` point with the gate as the reason. + +**B.6 The observer does not observe itself**: the o11y worker exports no Workers Logs and +no traces, with invocation logs off and its own lines persisted in Cloudflare's dashboard +only. Its self-metrics (`o11y.*`) go to Analytics Engine like every other metric, which is +not a loop: they never pass through its own ingest routes. + +### C. OTLP from the Worker inward + +**C.1 Hops.** + +| Hop | Format | +|---|---| +| Authoring app → o11y worker | Faro JSON (the SDK's native transport) | +| Embeds → o11y worker | the §C.5 beacon payload | +| API worker → o11y worker | OTLP logs, Cloudflare's export | +| o11y worker (ingest) | everything normalised to OTLP log records, stored as OTLP JSON in the inbox | +| drain → Loki | OTLP `/otlp/v1/logs`, synchronous | +| metrics | Analytics Engine points, written at ingest by the o11y worker (browser) and directly by the API worker (server) | + +**C.2 Attributes, identity and time.** Every record carries `service.name`, +`service.version`, `deployment.environment.name` and the `hot.*` set (`surface`, `tier`, +`framework`, `ht_major`, `outcome`) as **resource attributes**. Per exit criterion 15's +own precise wording, Loki labels +`service.name`, `deployment.environment.name` and every `hot.*` key — seven of the eight — +from `containers/o11y/loki/loki-config.yaml`'s own `otlp_config.resource_attributes` +promotion list. `service.version` is a resource attribute (queryable, present on every +record) but is **deliberately never promoted to a label**: it is per-deploy-SHA, and a +label with that cardinality would fragment Loki's index into one stream per deploy. Exit +criterion 15 checks exactly this seven-key set arrives as labels, confirmed live against +the real committed config for all four sources (§L). The exact names, allowed values and +Analytics Engine slots live in +[`docs/observability-contract.md`](../observability-contract.md). +`hot.demo_id`, `session.id` and `cf.ray` are structured metadata only; never labels, never +Analytics Engine indexes. The user pseudonym, emails, IPs, user-agent strings, query +strings, authored code, chat text and console output are never sent to the o11y stack. + +- `session.id` is an in-memory id minted per page load by the app (§E.4). It joins a page + load's browser records with the API calls that page made (sent as `x-hot-session`); it + is not a visitor session and does not survive a reload. +- `service.version` is the deploying `GITHUB_SHA` on each deployable. Authoring and API + deploy independently (path-gated in `master.yml`), so their versions usually differ; + each deploy job posts `{service, sha, cf_version_id}` to `/telemetry/deploy`, which + becomes a Loki line and a Grafana annotation. +- **Timestamps are event time, deterministically**: browser item timestamps are clamped + to the envelope's `received_at` ± 5 minutes; beacon timestamps likewise; OTLP records + keep `time_unix_nano`, falling back to `observed_time_unix_nano`, then to `received_at`, + so no record ever reaches Loki without one. + +**C.3 Symbolication in the Worker, at drain.** CI uploads the authoring build's maps to +the EU maps bucket under `sourcemaps/<sha>/<original asset path>.map` before deleting +them from `dist` (the Sentry Vite plugin's in-build deletion is turned off; one CI step +uploads to both destinations, then deletes). At drain, for exception records only, the +o11y worker resolves app-chunk frames with `@jridgewell/trace-mapping` against the map for the +record's `service.version`, parsing lazily per file and caching in the isolate within a +fixed memory budget; frames from the Babel compiler chunk and third-party files are left +as they are. Maps expire with the browser tenant (30 days). Symbolication never runs on +the public ingest route. Exit criterion L.5 bounds its cost. + +**C.4 Traces are deferred, and not exported.** No trace destination is configured. +Worker traces are sampled at 1 % into Cloudflare's own dashboard for the 7-day view while +`persist: true` holds (§H). Tempo, and a trace export with a span-attribute allowlist +that removes URLs, user agents, geo and preview hosts, return together when +`spanContext()` makes the browser join real. + +**C.5 Lite beacon for `/d` and `/embed`.** The existing ES5 reporter in `monitor.ts` gains +a standalone mode: with no parent runner frame, it sends uncaught errors (up to the +existing ceiling) and sampled web vitals (10 % of page views) with `navigator.sendBeacon` +to same-origin `/telemetry/lite`, under 2 KB per payload. It is injected at the serve seam +in `share.ts` with the `monitor-inject.ts` guards and the DEV-2580 rules (self-removing +tag, no whitespace). Docs pages link with `noreferrer` and send +`strict-origin-when-cross-origin`, so no embed payload carries a docs page path; +embeds are identified by demo id. + +### D. Worker signals + +- `observability.logs`: `head_sampling_rate: 1.0`, `invocation_logs: false`, + `persist: true`, `destinations: ["o11y-logs"]`. +- `observability.traces`: `head_sampling_rate: 0.01`, `persist: true`, no destination. +- One structured JSON line per non-proxy request (route class, status, duration, + `cf.ray`, `session.id`, `hot.demo_id`, `service.version`), replacing the + bracketed-prefix convention, plus an `api.request` Analytics Engine point. The preview + proxy path emits nothing per request; stale-preview requests are answered before the + Sandbox SDK where the code can recognise them, so its per-request warning does not fire. +- Every error that escapes a handler is logged as a structured line **by our code**: the + fetch catch-all, the snapshot-job alarm's report path, the Durable Object alarms and the + cron handler. The Worker half of constraint 4 therefore does not depend on how + Cloudflare exports an uncaught exception with invocation logs off. +- Custom spans (`tracing.enterSpan`) around session start, container boot, snapshot + build, chat, theme AI, import and payload boot, for the dashboard view. +- A `*/5` cron in the **API worker**, which owns the KV session meters and D1, writes + `pool.gauge`, `budget.gauge` (tier, percent of ceiling) and checks the o11y heartbeat + (§F.3). +- **Volume budget** (exit criterion L.8): at current traffic, Workers Logs events (lines + plus spans) stay under half of the 20 M pool, exported log events under half of the 10 M + logs allotment, Analytics Engine points under half of 10 M. Tier-2 container stdout is + counted in the first. Because every count and alert reads Analytics Engine, which is not + sampled at ingest, lowering the log sampling rate is the fallback that costs text, never + alerts. + **Measured (projected from real per-session/per-request counts × `traffic-baseline.md`, + at the ADR's own required 10× headroom): Analytics Engine points pass comfortably + (≈4.14M of the 10M dataset, well under half). The raw Workers Logs pool also passes + (≈6.6M of 20M, under half) — but the exported-logs allotment does not clear its own half + at 10×: real measured Tier-2 container stdout (12–22 lines for a Vite-family starter's + boot alone, ~22 for a webpack/Angular-family starter's boot, plus 2 lines per 60-second + keepalive poll — Cloudflare's own Sandbox SDK's structured logging of its own health + checks, not the dev server's own output) pushes the projected 10× total to ≈6.6M/month + against the 5M half of the 10M exported-logs allotment.** This is exactly the situation + this paragraph's own fallback exists for: lower `head_sampling_rate` before or during + launch if real production volume confirms this projection, which drops exported/logged + text but never drops a count, an Analytics Engine point, or an alert. See + `docs/run-and-deploy.md`'s Launch plan for the concrete pre-launch measurement and the + sampling-rate action. +- The comment at `wrangler.jsonc:10-13` ("full fidelity is a spike amplifier") is answered + by `invocation_logs: false`, the silent proxy path and the budget above, not reversed. + +### E. Sentry: what "uncaught" means, what moves, what stays + +**E.1 Definition.** Uncaught means an error that escapes a handler: + +- **Browser**: `window.onerror`, `unhandledrejection`, and render crashes caught by + `Sentry.ErrorBoundary`. +- **Worker**: anything that escapes a fetch, alarm or scheduled handler, including errors + the fetch catch-all turns into a 500 and snapshot-job failures the alarm reports + without rethrowing. + +These stay in Sentry, keep its grouping and regression detection, and also reach the new +stack (§E.2). + +**Moves to the new stack only**: diagnostic reports about handled conditions — upstream +failures reported with tags (npm registry, import URL), the preview boot-window report, +`reportError` calls for recoverable UI failures, and demo-runtime preview events. They +become Faro reports or structured lines with an `error.handled` point and lose Sentry's +grouping; the exact new-fingerprint alert (§F.3) replaces its "new issue" signal. The +Outcome of each implementing task lists every call site and its classification. + +**Stays in Sentry, unconverted**: the budget-alert `captureMessage` in `reconcile.ts` and +its `rehomeBudgetAlert` hook. It is the spend alert channel. + +**E.2 Tee.** Faro's errors instrumentation sees `window.onerror` and rejections; +`Sentry.ErrorBoundary`'s `onError` also calls the facade, so render crashes reach Faro. +Worker errors reach Loki through the structured lines of §D. Sentry's `beforeSend` +pushes the Sentry event id as a Faro event, and the Faro page-load id becomes a Sentry +tag. The issue-alert webhook (new issue, regression, resolved) becomes a Loki line with +issue id, title, release and link. + +**E.3 Switch.** The trim ships behind `SENTRY_SCOPE` (API worker var) and +`VITE_SENTRY_SCOPE` (authoring build), both `full` by default, where moved reports go to +**both** Sentry and the new stack. The launch plan flips both to `uncaught` only after +data has been seen end to end in Grafana, alerts have fired once, and volume sits inside +the budget. Until then nothing that reaches Sentry today stops reaching it. + +**E.4 Faro configuration and the operational-log rule.** + +- Faro: session tracking **disabled**; the facade mints the page-load id in memory and + sets it on every item; no `user` meta; only the errors and web-vitals instrumentations + (Performance, CSP, console and view instrumentations off, because they send full URLs + or console text); transport to same-origin `/telemetry/collect`. +- One scrubber, `scrubTelemetry`, runs in the browser **and authoritatively at ingest** + for Faro, beacon and OTLP records alike: strip query strings and fragments from every + URL-valued field; `redactPreviewHosts` on every string; reduce any browser meta to the + device and browser classes `analytics.ts` uses; remove Babel code frames explicitly + (`stripCodeFrame`: the gutter-numbered source lines and caret markers), because + `normalizeMonitorMessage` does not; drop unknown attributes. +- The rule: **operational logs may carry a page-load id, a demo id and a cf-ray as + structured metadata, and nothing else that identifies a person or a request's content. + Browser streams are kept 30 days, Worker streams 90 days.** This is a separate class from + analytics, which keeps its rule unchanged: `example.*` events are counts only and never + reach Loki. +- Local testing: Faro runs on a local path only when the build was made with + `VITE_TELEMETRY_LOCAL=1` and the host is `localhost`/`127.0.0.1`, with environment + `local`. That path checks neither `import.meta.env.DEV` nor `navigator.webdriver`, + because Playwright serves a production `vite preview` under automation. The production + gate (`resolveReporting`) is unchanged and stays closed under automation; a post-build + check fails if the local path survives into a production bundle. + +### F. What is metered + +**F.1 Store.** Counts and latencies go to Analytics Engine; Loki holds the text. (Contract +§6's own table matches this ruling — a Faro measurement or web-vitals +item is AE-only, never a stored Loki record.) The +positional slot layout and the full metric registry, with allowed outcomes, are in +[`docs/observability-contract.md`](../observability-contract.md) §4–§5; this section +names the signals, the contract fixes their shape. Every +count is `SUM(_sample_interval * count)`, every percentile a weighted quantile, and every +query goes through one helper that allowlists Analytics Engine's documented functions. +Browser metrics are extracted from Faro measurements at ingest (§B.2); server metrics are +written by the API worker. + +**F.2 Catalogue.** + +| Journey | Signals | +|---|---| +| **Play** | `preview.ready_ms` (pick → `data-preview-status="ready"`, by tier/framework/ht_major — the headline metric); `sandpack.compile_ms`; `sandpack.compile_error` (fingerprint, no code); `sandpack.bundler_unreachable`; `preview.runtime_error` groups (ladder-deduped); `version.switch`; `bucket.resolve_ms` | +| **Edit live** | `session.start` (server; outcomes `ready`, `at_capacity`, `container_starting`, `boot_timeout`, `budget_denied`, `error`) and `session.start_ms` (client, cold/warm); `container.boot_ms`; `hmr.roundtrip_ms` where a reliable hook exists; `session.end` with reason and awake seconds; `pool.gauge` every 5 min | +| **Share & build** | `snapshot.build`; `serve.share`, `serve.d`, `serve.embed` (demo id as a blob); web vitals on `/share` and `/d` | +| **Embed on docs** | beacon `error.uncaught` and `web_vital` by demo id and `ht_major`; broken-embed and slow-embed lists by demo id | +| **Assist** | `chat.answer` (model, tokens, USD, latency, outcome), `chat.edit`, `theme.ai`, `import.url`, `payload.boot` | +| **Examples** | ADR-0042 | +| **Platform** | `api.request` (route class, status class, duration); `error.handled`; `budget.gauge` every 5 min; `reconcile.run`; o11y self: `o11y.ingest`, `o11y.drain`, `o11y.wake`, `o11y.backlog`, `o11y.alert` | + +Sampling: authoring Faro 100 %; beacon errors 100 %, vitals 10 %; Worker lines 100 %; +Worker traces 1 % into Cloudflare's dashboard only. + +Dashboards, provisioned from git: Runner overview (with deploy annotations), Tier-2 +sessions, Tier-1 playground, Version health (framework × ht_major, `next` highlighted), +Docs embeds, AI assist, Examples & features (ADR-0042), Observability self. Cost moves to +ADR-0043, because spend truth is D1. + +**F.3 Alerting runs outside the box, with state.** Grafana holds no alert rules: its +state would not survive a sleep. + +| Signal | Owner | Latency | +|---|---|---| +| Uncaught error, new issue, regression | Sentry | seconds | +| Spend thresholds (200/500/800) | `reconcile.ts` `captureMessage` to Sentry, as today | nightly | +| `at_capacity` rate, 5xx rate from `api.request`, preview-ready rate per tier, session start p95, embed error rate per demo id, compile-error rate per `ht_major` day over day, snapshot-build failed rate per framework, LiteLLM error rate (`chat.answer` + `theme.ai`), inbox backlog age, a `rejected` inbox key, the o11y spend cap | o11y worker `*/10` cron over Analytics Engine and `InboxWriter` → Slack | minutes | +| New handled-error fingerprint | the exact first-seen registry in `InboxWriter` (not sampled data), excluding `surface = demo-runtime`, whose keystroke ladders are authored-code output | minutes | +| The o11y stack itself stale (no cron tick or ingest for 30 min) | the API worker's `*/5` cron reads the o11y heartbeat over a service binding and sends `captureMessage` to Sentry | minutes | + +Alert state (firing, resolved, last notified) lives in Durable Object storage: a rule +notifies once when it fires and once when it resolves, never on every tick. Thresholds are +starting values, tuned after launch: preview-ready below 97 % (Tier-1) or 95 % (Tier-2) +over 1 h; session start p95 above 20 s; `at_capacity` above 5/h; 5xx above 1 % over +15 min; LiteLLM errors above 5 %; compile errors on one `ht_major` doubling day over +day; snapshot builds failing above 50 % per framework over 30 min with at least 10 +failed; an embed above 20 % errors with more than 50 views in 24 h; backlog older +than 2 h. + +### G. Cost is its own number, with its own cap + +- `recordContainerUsage` takes the SKU as a parameter; `o11y_container` carries the box's + awake seconds, reported by the o11y worker to the API worker over the `API` service + binding (the API worker owns D1 and KV; the o11y worker binds neither). +- `reconcile.ts` iterates over the scripts it reconciles, `handsontable-demos-api` and + `handsontable-demos-o11y`, and writes each script's billing rows under distinct SKUs + (`o11y_container`, `o11y_workers`), so the per-SKU upsert never overwrites the app's + rows. +- `/admin` shows app, observability and total. +- `O11Y_BUDGET_USD` (default $15/month) joins the guardrail settings. When month-to-date + observability spend crosses it, drains pause, visit wakes still work, and one Slack line + is posted. Metrics and alerts keep working, because they read Analytics Engine, not Loki. + Log text is kept only while the pause is shorter than the inbox's 7-day lifecycle and + Loki's 7-day `reject_old_samples_max_age`; a longer pause loses the oldest log text, and + that is the price of the cap. +- **Why the cap ($15) sits above the exit ceiling ($10)**: the ceiling is a design check + at current traffic, measured once in exit criterion 7; the cap is a runtime brake. A cap + at the ceiling would pause drains on the first busy month or heavy week of Grafana use, + which is exactly when the logs matter. The $5 gap is headroom for traffic growth; a + month that reaches it is a signal to revisit the design, not normal operation. + Product tiers keep acting on the total. The bound is honest: drains are capped by the + pause; visits are bounded by the 15-minute idle stop and the 4-hour limit (§A), not by + construction. + +### H. Access, jurisdiction, retention + +- **Login broker, not Cloudflare Access**: every `@handsontable.com` account, as for + `/admin` — the same Handsontable login broker (ADR-0007). A callback page under + `/grafana/_o11y/` reads the broker's fragment token once and exchanges it for the + Worker's own HMAC-signed session cookie (`gates/session.ts`); Grafana Viewer via + `auth.proxy`. This **conforms to ADR-0007 rather than deviating from it**. The earlier + claim in this section — that the broker "hands a JWT to a SPA and cannot gate a proxied + third-party HTML application" — was wrong: hot-mcp's own `create_app` runtime gates a + proxied app the identical way, and a point-in-time production probe (the §M + access-broker delta has the exact curl command and its `302` result, dated 2026-09-24) + confirmed the callback host was allowed as of that date — expected to still hold, but + re-run that probe before launch rather than assuming it (see the §M access-broker delta + also for the one risk this design inherits rather than fixes, DEV-3088). +- **EU-pinned**: the container (`jurisdiction: "eu"`), both Durable Objects, the inbox, + Loki and maps buckets. +- **Deploy order** for the mutual service bindings: the o11y worker first (binding the + existing API worker), then the API worker with its `O11Y` binding. +- **Not EU-pinned, stated plainly**: Cloudflare's own Workers Observability store keeps + full lines and 1 % of spans for 7 days while `persist: true`, which is switched off 30 + days after this ADR is accepted; Analytics Engine; the Workers edge; Sentry, whose + ingest host is `ingest.us.sentry.io` (unchanged); D1, which has an EEUR location hint, + not a jurisdiction, and cannot gain one later. +- **Retention**: Loki `browser` 30 days, `worker` 90 days (§B.4); inbox objects 7 days; + maps 30 days; Analytics Engine 3 months (platform); D1 rollups unbounded; + `analytics_visitors` 180 days as today. + +### I. Local development + +Layout: `workers/o11y/` (Worker, both DOs), `containers/o11y/` (Dockerfile, Loki and +Grafana config, provisioning, `compose.yml`), `pipeline/fixtures/otlp/`, +`pipeline/o11y-*.test.mjs`. The whole new stack runs locally with Docker: the box through +`wrangler dev` or `compose.yml`; the o11y worker with Miniflare's R2, DO and cron; Loki on +Miniflare's local S3 endpoint for R2 or MinIO; Analytics Engine replaced by a ClickHouse +container holding an AE-shaped `runner_events` table, queried with the same SQL through +the allowlisting helper; the login broker's session check by a fail-closed `DEV_ADMIN` +bypass in `.dev.vars`. +Cloudflare's OTLP export does not run locally; its fixtures are real bodies captured by +the sandbox probe, scrubbed, plus hand-built edge cases. `pnpm o11y:dev` starts the box, +the Worker and the fixture replay. + +### J. ADR-0042 and ADR-0043 + +ADR-0042 (example analytics) ships with this ADR: it needs the ingest path and Grafana, +and its events are counts only. ADR-0043 (`/admin` reads in Grafana) follows after launch; +the manual ledger re-open does not wait for it and lives on `/grafana/_o11y/reopen`. + +### K. Tests + +Every implementing change carries tests that fail with the change reverted +(docs/TESTING.md, the presence gate): gate tests per §B.5 row; `InboxWriter` tests for +normalisation, dedupe, storage-backed buffering across a simulated restart, sequence +persistence and key order; ledger tests for provisional/committed/re-opened/rejected +transitions and marker handling; a label test asserting the Loki series for each source +carry exactly the contract labels (exit criterion 15); scrubber tests per rule on real inputs (a Babel code +frame, a preview-host URL, a user-agent string); a config test pinning the Loki keys of +§B.4 and the `observability` block of §D; symbolication against a real `vite build`; +beacon injection and an `acorn` ES5 parse; `e2e/o11y-local.spec.ts`, which drives a +real browser against the real o11y worker and asserts the resulting points land in the +local Analytics Engine stand-in (ClickHouse rows) — not a Loki query. Loki queryability +itself is proven separately: `containers/o11y/local/stop-roundtrip.mjs`'s own +`query_range` calls against a real local Loki (the clean-stop/reopen roundtrip, criteria +1–2), and the local end-to-end walkthrough's live Grafana Logs-panel checks +(`docs/run-and-deploy.md`'s local-dev section). + +### L. Delivery and exit criteria + +**Probes.** Cloudflare-only facts are measured with throwaway resources on the sandbox +account: separate Worker names, a probe-only config, `wrangler whoami` before every +deploy, synthetic traffic only, every resource deleted afterwards and listed. + +**Order**: shared contract and scaffolds; the box and its stop protocol (spike a); +ingest and `InboxWriter` (spike b); drain, ledger and Grafana access; API worker signals; +Faro and the beacon; alerts and cost; ADR-0042; dashboards; CI and runbook; the local +end-to-end walkthrough and launch. + +**Exit criteria**, each a pass/fail with recorded evidence: + +1. **Clean stop**: lines pushed, clean stop, fresh wake → 100 % of lines queryable and the + marker present, written with the **production-scoped** R2 token, not a local + all-bucket credential. **Plan B** if Loki does not upload the index on graceful shutdown: the + stop protocol waits for the next 15-minute index rotation and its upload before + stopping, and wakes are thinned to backlog > 3 h or > 128 MB, which keeps the cost + model within criterion 7; if that fails too, the serverless store replaces Loki. +2. **Unclean stop**: SIGKILL mid-drain, next wake → re-open → `count_over_time` and a log + query both equal a single clean replay. +3. **Event time**: stored timestamps equal event time (clamped for browsers) for Faro, + beacon and OTLP records, including OTLP records without `time_unix_nano`. +4. **Duplicate delivery**: the same export body delivered twice produces one copy. +5. **Symbolication**: an exception from a real `vite build` resolves to `src/…` file and + line using at most 500 ms CPU and 64 MB of isolate memory; Babel-chunk frames are + skipped, not parsed. +6. **Cold start**: wake-to-ready at most 90 s, worst of five runs on Cloudflare. +7. **Drain wake**: one hour of production-shaped traffic drains in at most 5 minutes of + wall time, no alarm invocation exceeds the configured `cpu_ms`, CPU per object is + recorded, and the cost model recomputed from the measured duration stays at or under + $10/month at current traffic. +8. **Volume**: each allotment of §D under half, projected from measured per-session and + per-request counts times production traffic. +9. **Idle tab**: an open, idle Grafana tab lets the box stop at the 15-minute idle timeout. +10. **Placement**: the Container's Durable Object namespace accepts `.jurisdiction("eu")` + and the instance runs in an EU region. +11. **Worker errors**: an error thrown in a fetch handler, a DO alarm and the cron handler + each produce a structured line in Loki with `invocation_logs: false`. +12. **Stop semantics**: what `onStop` reports for our own `stop()` is recorded, and a + Worker-initiated stop does not escalate to SIGKILL before the clean marker is written. +13. **Retention**: R2 lifecycle deletes expired chunk prefixes per tenant, and queries past + `max_query_lookback` return nothing. +14. **Image**: compressed size recorded and at most 1 GB, and it fits the instance disk. +15. **Labels**: records from each source (Faro, beacon, Cloudflare export, deploy events) + arrive in Loki with every `hot.*`, `service.name` and `deployment.environment.name` + label populated, and `hot.demo_id`, `session.id` and `cf.ray` present as structured + metadata only, never as labels. Checked with Loki's label and series APIs, locally and + on the sandbox probe with a real Cloudflare export. + +If criterion 1, 2 or 7 fails with its plan B, the design is rewritten toward the +serverless store before more is built. + +**Results (local end-to-end pass plus every sandbox probe):** + +| # | Criterion | Result | +|---|---|---| +| 1 | Clean stop, production-scoped token | **PASS** — sandbox probe with a real bucket-scoped R2 token (T03B), local `compose`/`wrangler dev` both pass (T03-D2 fixed: the drain now posts a real OTLP `resourceLogs` envelope, not bare NDJSON) | +| 2 | Unclean stop, reopen | **PASS, fully** — real SIGKILL mid-drain on the sandbox platform (T03B): reopen, replay, `count_over_time` and a log query both equal one clean replay. Independently reproduced locally (T11): an interrupted wake's canary record reopens and replays to exactly one Loki line, both query forms agreeing | +| 3 | Event time | **PASS** (local) — clamped browser timestamps, un-clamped OTLP `time_unix_nano` (T02); a real, old-dated captured OTLP fixture was genuinely rejected by Loki's 7-day window this pass, which is only possible if its stored timestamp preserved real event time | +| 4 | Duplicate delivery | **PASS** (local) — the same body delivered twice produces one copy; re-confirmed live and repeatedly this pass (`o11y.ingest` outcome `duplicate`) | +| 5 | Symbolication | **PASS at the local/Node level; not verified inside a real Workers isolate** — every measurement (T03, T11) uses a Node-process CPU/memory proxy, explicitly labelled as a proxy; no task had a way to profile a real Workers isolate | +| 6 | Cold start | **PASS** — sandbox: 46.5s worst-of-5 (T01), 3–22s after the T03-D3 fix (T03) | +| 7 | Drain wake (time + cost) | **PASS at the corrected traffic scale** — sandbox (T03B): 28s wake-to-drain-complete at 1× (≈432 records/hr, T05's own per-session line count), 44s at 10× (≈4325/hr); cost $0.21/month at 1×, $0.33/month at 10× — both far under the $10 ceiling | +| 8 | Volume | **Mixed, measured, not a breakeven guess** — Analytics Engine points and the raw Workers Logs pool both pass at 10× with real margin; the **exported-logs allotment does not** (§D above has the numbers and the fallback) | +| 9 | Idle tab | **PASS by mechanism** — sandbox (T01): the box's own quiet-timer stopped it after 17.65 minutes with zero HTTP requests, which is what an idle tab with Grafana Live disabled also produces; never independently reproduced with a literal open browser tab | +| 10 | Placement | **PASS** — sandbox: EU region `mxp04` (Milan) | +| 11 | Worker errors → structured line | **PASS** — fetch-handler and cron paths confirmed live (T05, T11); the DO-alarm path is unit/pipeline-tested (T01–T03) but not independently reproduced live | +| 12 | Stop semantics | **PASS** — `onStop` is recorded and, by design, claims nothing about cleanliness (T01); "no SIGKILL before the clean marker" is the same platform behaviour criteria 1 and 2 already confirm | +| 13 | Retention | **Mechanism PASS, real expiry PENDING the calendar** — R2 lifecycle rules apply and read back correctly (T01, T10); T03B's own 1-day retention-clock test (`t03-retention-clock-test/`, `o11y-probe-t03-loki`) started 2026-09-23T14:15:22Z and has not yet reached 24h as of this pass | +| 14 | Image size | **PASS** — 212.9 MB compressed, real `linux/amd64` build (T01), under the 1 GB bound; uncompressed size against the `standard-1` 8 GB disk was not separately recorded by any task | +| 15 | Labels | **PASS, all four sources, both tenants** — confirmed live against the real committed `loki-config.yaml`: Faro (`demos-authoring`) and the lite beacon (`demos-embed`), browser tenant; the Cloudflare export (`demos-api`) and deploy events (`demos-o11y`), worker tenant — all seven labels populated, `service.version` present as a resource attribute but deliberately never promoted to a label (see §C.2), `hot.demo_id`/`session.id`/`cf.ray` never labels | + +§L's own trigger (criterion 1, 2 or 7 failing its plan B) is **not** engaged — all three pass. +Two items keep this ADR at **Proposed** rather than **Accepted** (below): criterion 5's +real-isolate measurement (no task had Workers isolate profiling access) and criterion 13's +calendar-pending retention confirmation. Criterion 8's exported-logs finding is real and +measured, not a missing-evidence gap; it is carried as a named pre-launch action in +`docs/run-and-deploy.md` rather than as a blocker to this ADR's status, because the ADR's +own §D already names the exact fallback (lower `head_sampling_rate`) for exactly this +situation. + +### M. Implementation deltas (full detail in git history under +the deleted `runner/tasks/o11y/`) + +Deltas already folded as direct edits above (§A cost, §A wake/stop, §B.3 drain-rejection, +§C.2 labels, §D volume, §L results) are not repeated here. The rest, grouped by section, +where they add information beyond what §A–§L already say: + +- **§B.2 ingest.** Hashing (step 2) uses each record's own raw, un-clamped source + timestamp alongside its body and attributes — not the clamped `time_unix_nano` a later + step computes — so two real deliveries of the same content at different real times still + hash differently, and the same body redelivered still dedupes. One aggregated + `o11y.ingest` point is written per *request* (not per record), so a batch of N duplicate + records reads as one `duplicate` point with `count = N`, not N separate points. + Every §3 resource attribute a source has no natural value for defaults to `"none"` + (`"unknown"` for `service.version` specifically, confirmed against real Cloudflare + export samples that never carry it at all) — this default is what makes exit criterion + 15 pass for worker-origin sources, not a defensive fallback. A real + Cloudflare OTLP export's ray id arrives as `cloudflare.ray_id`, remapped to the + contract's own `cf.ray`. `HotAttrs` fields with no dotted `hot.*` resource- + attribute counterpart (`bucket`, `reason`, `fingerprint`, and others with no + browser call site emitting them yet) travel over an AE-only channel, read from a Faro + item's raw `context` before the browser's own scrub allowlist would otherwise drop + them — this channel needed its own allowlist extension (`AE_ONLY_ATTRIBUTE_KEYS`) + before it worked for real. +- **§B.2 pack.** The pack alarm's "at 4 MB stored" trigger (step 6 above) is a real, + enforced upper bound on a single packed object's decompressed size + (`PACK_OBJECT_MAX_DECOMPRESSED_BYTES = 4 MB`), not only a flush-cadence hint. An + over-threshold burst is capped by **splitting into extra keys, not by cutting the + alarm's own accumulation short**: `packTenant` takes pending rows in arrival order up + to the budget and returns only that prefix (`consumedRowKeys`); `InboxWriter.alarm()` + loops `packTenant`/`commitPackedObject` per tenant over the leftover rows until + nothing remains or a 25-packed-object per-invocation cap is hit, rescheduling the + alarm immediately (`setAlarm(Date.now())`) when objects still remain. A single row + over budget on its own is still packed alone (row size is already bounded to + `INBOX_ROW_MAX_BYTES`, ~1 MB, well under the 4 MB object budget) rather than blocking + progress. +- **§B storage API, DO 128-key batch limit** (confirmed platform fact). Cloudflare's + SQLite-backed Durable Object storage API caps + `get`/`put`/`delete` at 128 keys/key-value pairs per call + (<https://developers.cloudflare.com/durable-objects/api/storage-api/>, fetched + 2026-09-24: "Supports up to 128 keys at a time" / "up to 128 key-value pairs at a + time"). Local `workerd` accepts 500+ in one call with no error, so this codebase's + own test doubles (`workers/o11y/src/inbox/storage.ts#memoryStorage()`, + `pipeline/fixtures/o11y-harness.mjs`) enforce the same cap explicitly rather than + relying on `workerd` to catch a violation. Every multi-key call in `InboxWriter` — + dedupe's `checkDuplicates`, the fingerprint registry's writes/prune, `pruneLedger`, + wake resolution, `markKeysProvisional`, manual reopen, the pack commit, and + `ingest`'s own transaction `put` — chunks through `storage.ts`'s + `getManyChunked`/`putChunked`/`deleteChunked`. `finalizeWakeResolution` and + `reopenWindow` run their whole put+delete sequence inside one + `storage.transaction()` — chunking alone, without that, would let a crash between + chunks leave a partial write (an orphaned `provisional:<wakeId>` key whose + `wake:<id>` is already gone). +- **§B.3 drain reads whole objects.** Each drained key's packed object is read into + memory whole before its records are pushed + to Loki — this is bounded, not unbounded, because the object it reads was itself + capped at write time (`PACK_OBJECT_MAX_DECOMPRESSED_BYTES`, ~4 MB), plus at + most one further oversized single row (`INBOX_ROW_MAX_BYTES`, ~1 MB) packed alone + when it alone exceeds the object budget. So one drain-time read is bounded to roughly + 4–5 MB, never the whole tenant's backlog at once. This bound is a property of the + PACK side (`inbox/pack.ts`), restated here as the drain-side consequence, not as a + claim about `drain.ts`'s own internals. +- **§B.2 ingest, worker tenant.** A Worker's own `console.log(JSON.stringify(...))` line + (the structured request/error lines §D describes) arrives through Cloudflare's real OTLP + log export as **opaque body text**, not as OTLP attributes — confirmed with a real + captured export. The o11y + worker parses a JSON-object body and merges its keys into the same attribute bag a + real OTLP attribute would land in, through the existing allowlist, with every + §3 resource-attribute key **stripped from the parsed body first and given the lowest + merge priority** — a body key cannot spoof `service.name`/`deployment.environment.name`/ + any `hot.*` label. +- **§B.2 ingest, worker tenant — fingerprint** (live at merge). The API worker's + own handled-error lines (`reportDiagnostic`, + `workers/api/src/telemetry/diagnostic.ts`) carry `hot.fingerprint` (contract §3 + AE-only key) in the same structured JSON body the bullet above describes. The read + half (`workers/o11y/src/normalise/otlp.ts#toIngestItem`/`apiFingerprintFeed`) reads + `bodyJsonAttrs["hot.fingerprint"]` (the pre-`hoistAttributes` bag + `tryParseJsonBodyAttrs` already builds) and feeds it into the `fp:` registry only + when ALL of: the REAL resource `service.name === "demos-api"` (read from + `finalResourceAttrs`, the resource attribute after hoisting/defaults — never from + `bodyJsonAttrs`, the same anti-spoof rule `RESOURCE_ATTR_KEY_SET` already enforces + for every other resource attribute); the parsed body's `log.kind === "error"`; the + value matches contract §7's own shape, via `isValidFingerprint` — ONE shared + validator, also used by the browser path's `resolveFingerprint`, never a second, + independently drifting copy: a validator anchored on the FIRST + `:` would reject the `:`-joined call-site paths `reportDiagnostic`'s own real callers + send — `"npm-registry:version-exists"`, `"npm-registry:versions"` — so neither could + ever satisfy this condition, gate aside. + + **Two preconditions, both must hold:** + 1. **Service-name remap:** a real Cloudflare OTLP export's resource `service.name` is the + deployed script's own name (`handsontable-demos-api`), not the contract's short + `demos-api` — `normalise/otlp.ts#remapCloudflareServiceName` strips the shared + `handsontable-` script-name prefix whenever what remains is one of the contract's + own `SERVICE_NAMES`, applied before `hoistAttributes`, general across every + deployable. Confirmed against the captured real-export fixtures + (`pipeline/fixtures/otlp/json/console-log-line*.json`, + `cloudflare-invocation-log.json`), every one of which carries the raw script name. + 2. **Shared fingerprint validator:** it accepts a `:`-joined `context`, so + `reportDiagnostic`'s own real call sites' fingerprints pass the shape check at + all — see above. + + Without BOTH, this gate is a correctly-gated no-op against real production traffic, + not a silent bypass; with both, it fires for real. **This makes it LIVE at merge** + (the o11y worker's `*/10` new-fingerprint cron runs unconditionally, regardless of + `SENTRY_SCOPE`/`VITE_SENTRY_SCOPE` — those only gate whether a moved report ALSO + reaches Sentry, §E.3), not gated behind any later `SENTRY_SCOPE` flip — + `docs/run-and-deploy.md`'s runbook is updated with this as an explicit precondition + check, not just a flip-time one. + + The fourth condition — not Tier-2 container stdout — is attempted by construction, + not guaranteed: the B cross-note fix (two bullets below) makes `tryParseJsonBodyAttrs` + refuse to parse ANY body whose own `log.kind` isn't one of this worker's trusted + shapes, so `bodyJsonAttrs` is empty for a body with no matching sentinel — but the + sentinel is body TEXT, not a resource attribute, and (known gap, not fully resolved — + see `normalise/otlp.ts#tryParseJsonBodyAttrs`'s + own doc comment for the full analysis) nothing in this pipeline can currently + tell a genuine `lines.ts` line apart from a Tier-2 container's own authored stdout + that happens to print the same shape, since both would share this Worker's + `service.name` once the service-name remap normalises it. Accepted, bounded residual: a forged line can + only mint a `fp:` entry and a notify-only, mrkdwn-escaped Slack line, the same + noise class already accepted for the browser path — never Sentry, PII or code + execution. Deliberately NOT `hot.surface !== "demo-runtime"` (the browser path's own + rule) — a worker-tenant record's `hot.surface` resource attribute defaults to + `"none"` when nothing sets it, which would admit any body reaching + `/telemetry/v1/logs`, forged or not. +- **§C.1 hops.** Faro's real browser transport posts a `TransportBody` + (`{meta, exceptions?, logs?, measurements?, events?, traces?}`), not an array of + self-contained items the way every contract function's own types assume — the ingest + route reconstructs items from the four typed arrays. +- **§C.3 symbolication.** A Faro exception's stack trace reaches the drain as V8-shaped + text in the record body — the pre-implementation contract had no field carrying frame + data for this to resolve at all. +- **§D Worker signals.** `container.boot_ms` (not `session.start`'s own `boot_timeout` + outcome) is what fires when the Tier-2 boot window is exceeded — the original design + would have double-counted a session that later times out after already reporting + `session.start` `ready` once (a design correction made before shipping, not + after). Several §5 metrics remain real but never observed in practice: `pool.gauge` + `reason="builder"` (no signal tracks `BuilderSandbox` concurrency the way live sessions + are tracked), `snapshot.build` `reason="inline"` (only the detached build path is + instrumented), `session.end` `reason="sleep_after"` (nothing observes the Sandbox SDK's + own idle-timeout stop) — all named gaps, not silently dropped. A cron + failure inside `ctx.waitUntil()` is structurally unreachable by `@sentry/cloudflare`'s + own auto-capture (its `scheduled` instrumentation only wraps the synchronous handler + invocation) — every cron branch calls `Sentry.captureException` explicitly in its own + catch (confirmed live: a forced cron failure without this produced zero Sentry + envelopes; with it, exactly one). +- **§E Sentry.** The full call-site inventory found one real §11 violation: `App.tsx`'s + `versions-fetch` diagnostic was unconditional before this ADR's switch existed, exactly + the shape §E.1 already names as "handled." §E.3 is binding for demo-runtime preview + events: `reportDemoEvent` keeps its full pre-ADR Sentry behaviour (including the + `DEMO_SURFACE` environment re-homing) under `full` scope, unreachable under `uncaught` + — "the re-homing disappears once the scope flips" is literally true only after the + flip, not at implementation time. +- **§B.3 drain, a key with a mixed 400/2xx outcome** ("accepted chunks skip §B.3"). + Pushing every chunk of a key even after an earlier one 400'd, while still + classifying the whole key `rejected` if ANY chunk 400'd — including when another chunk + landed 2xx — would be wrong: a `rejected` key never becomes `provisional`, so those + already-accepted + bytes never pass the §B.3 marker/commit check any wake's clean stop confirms + durability through: an unclean stop right after the push, before Loki's own local + flush, could lose them with no automatic replay (only a manual reopen, which — being + a full key replay — would re-derive the identical classification anyway). Instead: a + key with at least one accepted (2xx) chunk stays `provisional`, following the + normal durability path; only a key with ZERO accepted chunks stays `rejected`. The + permanent chunk loss stays operator-visible via a `rejectedEvent:` audit log + (`ledger.ts#recordPartialReject`, contract §8) rather than the key's own ledger state. +- **§B.3/§F.3 storage housekeeping, remainder** ("prune ceiling"). + Three gaps in the prune mechanism: (1) a 500-row/tick batch limit falls behind + ADR §D's own 10× traffic + projection at roughly 3× today's traffic — raised to 5,000/tick (still chunked to the + real 128-key limit per call, see the DO storage batch-limit bullet above), with the exact arithmetic in + `dedupe.ts#HASH_PRUNE_BATCH_LIMIT`'s doc comment; (2) `newFingerprintsSince` listed the + entire (alphabetically, not chronologically, ordered) `fp:` prefix every ten-minute + alert tick — a new `fpts:<firstSeenMs>:<fingerprint>` time-ordered secondary index + (contract §8) makes this a bounded range read instead, with a truncation-safe cursor + in `alerts/rules.ts#newFingerprintRule` (never skips an unread fingerprint, at the cost + of a bounded duplicate report under sustained flood — the same trade-off already + accepted for the cursor's grace window); (3) `rejected-inbox-key` fired on + `rejectedKeyCount() > 0` and never resolved (rejected `key:` entries are deliberately + never pruned by date alone — see the row-19 bullet's `rejectedEvent:` log and the + §B storage API bullet above) — it now fires on a RECENT (last hour) count from that + same log instead, resolving once new rejections stop. +- **§F metering.** ADR-0042's `example.*` events needed the same AE-only attribute-channel + extension as §B.2 above (`kind`→`hot.metric_kind`, since `hot.kind` is reserved for the + Faro item kind, `ref`, `area`) before `kind`/`ref`/`area` survived the browser scrub at + all. A post-fork landing needs a one-shot, non-storage URL marker (`?fork=1`, + stripped via `history.replaceState` on read) to classify as `entry="fork"` rather than + `"deep-link"`, because `onFork`'s navigation is a full page reload — the same + hard-navigation pattern the rest of the app already uses for every route change, which + destroys any in-memory alternative. +- **§H access.** `ACCESS_AUD` is still the committed `""` placeholder as of this ADR's own + fold — no Access application was ever minted; `docs/run-and-deploy.md`'s Launch plan + names this as the first pre-condition to confirm before any real deploy. +- **§H access (supersedes the bullet above).** The `/grafana/*` Access application + named above was never created before launch. It was replaced with the + Handsontable login broker (ADR-0007) instead of finishing it — `gates/session.ts` + (session cookie, `DEV_ADMIN` bypass), `gates/broker.ts` (the one-time `/broker/userinfo` + call), `grafana/login.ts` (login/callback/session/logout). A real production probe + (curl against + `mcp-auth-proxy-j0tb.onrender.com/broker/login`, 2026-09-24) confirmed `302` to + Google for `return_to=https://demos.handsontable.com/grafana/_o11y/callback?n=…`, so the + callback path is allowed today (see `docs/run-and-deploy.md`'s step 5 for the exact + command; a separate local round trip against a *stubbed* broker proves the + Worker's own code, not the real broker's live configuration, and should not be read + as a second production probe). **This widens DEV-3088's blast radius, it does not just + inherit it**: the broker's + `return_to` allowlist is host-suffix-only, so it also admits anonymous Tier-2 preview + hosts under `*.demos.handsontable.com`, letting anyone harvest a team member's 1h broker + token. Before this change, a stolen token could not reach Grafana at all (`ACCESS_AUD` was `""`, + so Access refused everything); now, it can be exchanged for a Grafana session. This + narrows the exposure: `gates/session.ts#computeSessionTtlSeconds` caps the session at + `min(now + 12h, brokerTokenExp)` (falling back to 1h when the token carries no readable + `exp`) instead of a flat 12h, so a stolen token buys close to its own remaining + lifetime, not up to 11 extra hours — narrows, does not close. DEV-3088 itself remains + open and is filed and tracked separately from this ADR. +- **§I local development.** `wrangler dev`'s local Container reaches `compose.yml`'s + standalone `minio`/`clickhouse` services (started without the `box` service) via + Docker's own `host.docker.internal`, since the two are never on the same Docker network. + +## Consequences + +- **ADR-0040** decisions A (hour dimension) and B (`usage_hourly`) are not built; + awake-seconds per hour and sampled peak concurrency become Analytics Engine points + (C.2, C.3). **C.1 stands as written**: `at_capacity` is a `usage_daily` counter, and is + also emitted as a `session.start` outcome. D (privacy) stands. +- **ADR-0022** gains a subordinate o11y ceiling and per-script billing rows; + `recordContainerUsage` takes a SKU. +- **ADR-0038**'s WAF exception grows by one path, `/telemetry/*`. +- **ADR-0007**: no longer deviated from — `/grafana/*` gates through the same + Handsontable login broker as every other internal surface (§H). +- **ADR-0020**: more route patterns on the main hostname, still in deploy commands. +- **Sentry** keeps uncaught errors (as §E.1 defines them) and spend alerts; handled + diagnostics move and lose grouping; the per-event `environment` re-homing and the + `demo-runtime` environment disappear once the scope flips. +- **Source maps** are no longer deleted inside `vite build`; CI uploads them to Sentry + and to R2, then deletes them before deploy. +- **We own**: the Faro-to-OTLP converter, the scrubber, the symbolicator, the drain and + its ledger, and the alert evaluator. That is more code than revision 2, in exchange for + synchronous acknowledgements, event-time timestamps, and one path for every record. +- **New operational surface**: one image (Loki + Grafana), three EU buckets, two Durable + Object classes, provisioning in git, a fixture set, one Slack webhook, one + `O11Y_SESSION_SECRET` (no Access application; `/grafana/*` gates through the + existing login broker instead). +- **Accepted limits**: no browser-to-worker trace join and no traces in Grafana until + `spanContext()`; no alert rules or durable UI state in Grafana; the first visit after a + sleep waits behind the waking page; an unclean stop costs a replay and duplicate storage + until lifecycle; embeds have no docs page attribution; handled errors have no issue + grouping; the non-EU items listed in §H. +- **Cost**: ≈ $5–8/month target, $10 exit ceiling, reported separately, capped + separately, summed under the same product ceiling. **Measured (real platform, + §A/§L.7): $0.21/month at 1× traffic, $0.33/month at 10×** — both far under target. +- **Volume**: Analytics Engine and the raw Workers Logs pool both pass exit criterion 8 at + 10× with real margin; the exported-logs allotment does not, measured (§D) — + `docs/run-and-deploy.md`'s Launch plan carries the pre-launch action (a real Tier-2 + stdout measurement to confirm or refine the projection, and the `head_sampling_rate` + fallback if it holds). diff --git a/runner/docs/adr/0042-example-analytics.md b/runner/docs/adr/0042-example-analytics.md new file mode 100644 index 0000000000..14c00d94a4 --- /dev/null +++ b/runner/docs/adr/0042-example-analytics.md @@ -0,0 +1,112 @@ +# ADR-0042: Count which examples people open, by docs guide and starter + +**Status:** Proposed — design approved 2026-09-23 (revision 2), implemented. +Ships with ADR-0041 and depends on its ingest path and Grafana; stays at the same status +(Proposed) until ADR-0041 flips to Accepted, for the same two pending items (see +ADR-0041's own status line and Implementation deltas below). + +## Context + +Nobody can say which Handsontable features people reach for a live example of. +Opening an example on `/` — `?docs=<content-path>` for a documentation-guide example, +`?example=<starter>` for a starter — never reaches the API worker: the SPA and the +example JSON come from the assets worker, and `notePageView` counts only `/d`, `/embed` +and `/share`. + +What already exists: + +- **The taxonomy.** The docs-example JSON under + `apps/authoring/public/docs-examples/<bucket>/` holds 1,482 entries across 130 guides, + each with `guide`, `exampleId`, `docPermalink`, `breadcrumb` (its first element is the + area: Columns, Rows, Formulas, Accessibility, Recipes…), `framework` and `lang`. + Starters are the 19 keys of `config/frameworks.json`. +- **Attribution.** `demos.forked_from` already records each saved demo's origin + lineage (`catalog:<framework>`, `mcp:<framework>`, and docs, import and payload + prefixes), so saved demos and their `/d` and `/embed` views can be traced to an + example without a migration. The implementing task confirms the exact docs format and + the first date from which every save carries it (review: 2026-07-17). **Confirmed + (from git history, not production data):** the docs format is + `docs:<bucket>:<content-path>` (three segments), landed in commit `3277a52a4`, + **2026-07-17 13:38:35 +0200** — before that commit, a docs save carried only the + 2-segment `docs:<content-path>` (no bucket), whose `ref` cannot be resolved back to a + guide. This ADR's own §"No migration for attribution" rule uses this exact date as the + cutoff for "unknown." No code in this repository performs the `forked_from` ↔ `/d`/ + `/embed`-view join this Context section describes — it is groundwork for ADR-0043, not + a deliverable of this ADR. +- **No referrer.** The docs site's buttons use `rel="noopener noreferrer"` and the site + sends `strict-origin-when-cross-origin`, so a docs deep link cannot be told apart from a + direct visit by its referrer, and no embed request carries the docs page path. + +Constraint: anonymous by construction. Counts only, no user id, no per-request rows. + +## Decision + +1. **`example.open`**, fired once per resolved example (not per render) at the + example-resolve path in `App.tsx`, sent through the ADR-0041 facade and turned into an + Analytics Engine point at ingest. It is **never written to the inbox or Loki**, so it + carries no page-load id anywhere it is stored. Attributes: `kind` (`docs`, `starter`, + `saved`, `import`, `payload`), `ref` (defined per `kind`, `exampleAnalytics.ts#exampleTaxonomy`: + the docs guide path for `docs`; the starter's `config/frameworks.json` key for `starter`; + the saved demo's id for `saved`; the import provider for `import`; the payload source for + `payload`), `area` (the + loaded entry's first breadcrumb element; it is not derivable from `ref`), `framework` + (for docs examples this already distinguishes JavaScript from TypeScript), `ht_major`, + `bucket`, and + `entry` — `deep-link` when the example came from the URL at page load, `picker`, + `switch`, `version-switch` or `fork` otherwise. `entry` is known inside the app, so it + needs no referrer. **Implementation note:** distinguishing a post-fork landing + from an ordinary deep link needs a one-shot URL marker (`?fork=1`, stripped on read via + `history.replaceState`, never `localStorage`/`sessionStorage`), because `onFork` + navigates with a full page reload — the same hard-navigation pattern the rest of the + app already uses for every route change, which destroys any in-memory alternative. A + bare `/` with no `?example=`/`?docs=` at page load (the silent starter default) is not + counted as a `deep-link` open — that default is not the visitor reaching for anything. +2. **Engagement**, same shape and same storage rule: `example.engaged` (first code edit, + or preview ready plus 30 s), `example.forked`, `example.saved`, `example.shared`, + `example.downloaded`. Engaged opens rank features; raw opens rank curiosity. + `example.saved` is the one the API worker writes, when an editor Save's rebuild + succeeds, because the rebuild can take longer than the visitor stays on the page. The + editor passes the open example's `ht_major` in the Save request, so the row has the + same values the browser would have sent (contract §5). +3. **No migration for attribution**: rollups join `demos.forked_from` against `/d` and + `/embed` view counts; demos saved before the confirmed date are reported as `unknown`. +4. **Analytics Engine layout**: `kind`, `ref` and `area` take three of the blob slots the + observability contract left unassigned (`blob17`–`blob19`); one stays free. + **Implementation note:** these three keys, and ADR-0041 §F's own AE-only + `bucket`/`reason`/`fingerprint` keys, share one transport problem — none has a dotted + `hot.*` resource-attribute form, so the browser's own scrub allowlist silently dropped + them before this was found and fixed (ADR-0041 §M). `kind` is read from + `hot.metric_kind`, not `hot.kind`, which is reserved for the Faro item's own kind. +5. **Permanent record**: a nightly step in the reconcile cron recomputes **the previous + full UTC day** from Analytics Engine into D1 `example_daily(day, kind, ref, area, + framework, ht_major, opens, engaged, forked, saved, shared, downloaded)` with primary + key `(day, kind, ref, framework, ht_major)` (`area` is a function of `ref`), written + with `INSERT OR REPLACE`. Re-running it for a day replaces that day's rows; events + arriving after the day closed are not counted. Counts use + `SUM(_sample_interval * count)`. **Follow-up:** `example_daily` shipped + without a `downloaded` column — this decision's own six-metric list above + included `example.downloaded`, but the table and rollup only ever carried the other + five. Migration `0009_example_daily_downloaded.sql` (additive `ALTER TABLE ... ADD + COLUMN downloaded INTEGER NOT NULL DEFAULT 0`) and the matching `reconcile.ts` rollup + change close that gap; existing rows backfill to `downloaded = 0` (their true count for + already-rolled days is unrecoverable from D1 alone, and 0 never overcounts). +6. **Dashboard** "Examples & features" reads Analytics Engine, so it covers the last three + months without D1: top guides by opens and engaged opens, area breakdown, framework + split per guide, starter ranking, `ht_major` distribution, the funnel open → engaged → + saved → shared per area, deep-link share of opens, zero-open examples over 90 days. A + long-range view from `example_daily` waits for ADR-0043's D1 path. **Not implemented:** + the zero-open-examples-over-90-days panel — Analytics Engine only ever holds + points for events that happened, so a "never happened" query needs the full docs-example + taxonomy (every guide × framework that exists, from the docs-examples JSON, not from + `runner_events` or D1) to diff against; the other seven panels shipped. + +## Consequences + +- One event family, one D1 table, one nightly rollup step, one dashboard. No `demos` + migration. +- The data measures who reached for a live example, not who read the docs page; pair it + with the docs site's analytics before calling a feature important. +- Docs pages are not attributed to embed views, and deep links are counted by how the app + was entered, not by where the visitor came from. +- Tests: an `example.open` test that drives the real resolve path, a test that + `example.*` never reaches the inbox, and an idempotency test for the rollup. diff --git a/runner/docs/adr/0043-admin-cutover-to-grafana.md b/runner/docs/adr/0043-admin-cutover-to-grafana.md new file mode 100644 index 0000000000..e9bdbd6a62 --- /dev/null +++ b/runner/docs/adr/0043-admin-cutover-to-grafana.md @@ -0,0 +1,59 @@ +# ADR-0043: `/admin` reads move to Grafana; the writes stay on `/admin/controls` + +**Status:** Proposed — design approved 2026-09-23 (revision 2). Follows ADR-0041 after launch; amends ADR-0022. + +## Context + +`/admin` is one page (`apps/authoring/src/Admin.tsx`) over these API routes: + +| Route | Kind | Used for | +|---|---|---| +| `GET /api/admin/usage?days=N` | read | spend, tier, SKUs, daily spend and activity, demos, AI assistant, audience, first page of live sessions | +| `GET /api/admin/sessions` | read | live-session paging and the 24 h tail | +| `GET /api/admin/settings` | read | effective guardrail settings and whether they are overrides | +| `PUT /api/admin/settings` | write | save guardrail settings | +| `DELETE /api/admin/settings` | write | reset to defaults | +| `DELETE /api/admin/sessions/:ref` | write | kill a live session | + +Every read comes from D1 or KV. Grafana has no D1 datasource and cannot be trusted to +stay read-only on its own: any Viewer can make the Infinity plugin send arbitrary requests, +POST included, to any host the datasource allows. A service-binding HTTP fetch lands in +the same handler as public `/api` traffic, so "internal" cannot be a header. + +## Decision + +1. **Reads over RPC, not HTTP.** The API worker exposes a `WorkerEntrypoint` class + `AdminReads` with one method per read (`usage`, `sessions`, `settings`, + `costLedger`). It is reachable only through a service binding with + `entrypoint: "AdminReads"` from the o11y worker, never from the public fetch handler. +2. **A read-only forwarder.** The o11y worker serves `GET /grafana/_o11y/admin/<name>` + behind the same session check as Grafana (the Handsontable login broker, ADR-0007 + — not Cloudflare Access), maps `<name>` onto exactly those methods and + answers 404 to any other name and 405 to any other method. The Infinity datasource's + allowed-hosts list holds only this forwarder. No admin token exists in Grafana. +3. **Panels**: Cost (moved here from ADR-0041: spend from `cost_ledger`, app / + observability / total, per SKU, estimate vs billing), Audience (unique visitors stay + D1-only — the salted hash table is the honest source and the 180-day rule governs it), + demos inventory, AI assistant, live sessions, and budget-tier history from ADR-0041's + `budget.gauge`. `example_daily` (ADR-0042) gets its long-range panel here. +4. **`/admin/controls`** keeps the writes and the quick look: the settings form with + reset, the live-sessions table with kill, and the three budget numbers. The manual + inbox re-open stays on `/grafana/_o11y/reopen` (ADR-0041 §B.3). +5. **Equivalence instead of a comparison week.** Grafana panels and `/admin` read the same + D1 rows, so the risk is the panel's query and transform, not the data. A test renders + both from one D1 fixture and asserts equal totals per panel (counts to the unit, USD to + the cent); one production spot check then compares the two for the same day. When + both pass, `/admin` redirects to Grafana through the login broker and the read + half of `Admin.tsx` is deleted. + +## Consequences + +- ADR-0022's `/admin` becomes `/admin/controls`; runtime-editable guardrails are + unchanged. +- The API worker gains an RPC entrypoint; the o11y worker gains a second service binding + to it and a forwarder with an allowlist. +- A quick look at spend no longer needs Grafana's cold start: the three numbers stay on + `/admin/controls`. +- Tests: the forwarder refuses unknown names and non-GET methods; the public fetch handler + cannot reach `AdminReads`; the equivalence test per panel; `/admin/controls` with + `VITE_DEV_USER` cleared. diff --git a/runner/docs/adr/README.md b/runner/docs/adr/README.md index 971d1783c8..44fb8f2bf1 100644 --- a/runner/docs/adr/README.md +++ b/runner/docs/adr/README.md @@ -44,4 +44,7 @@ once Accepted. | [0037](0037-persistent-api-tokens.md) | Persistent API tokens, verified in the Worker rather than by the broker | Accepted (DEV-2583; amends 0007) | | [0038](0038-waf-exception-for-source-code-payloads.md) | Source-code payloads need a WAF exception, not an encoding trick | Accepted | | [0039](0039-detached-tier-2-snapshot-builds.md) | Detached tier-2 snapshot builds on the MCP service path | Accepted (amends 0033) | -| [0040](0040-hourly-buckets-and-pool-pressure.md) | Hour-of-day buckets, and measuring pool pressure | Accepted (extends 0022) | +| [0040](0040-hourly-buckets-and-pool-pressure.md) | Hour-of-day buckets, and measuring pool pressure | Accepted (extends 0022; A, B, C.2, C.3 superseded by 0041; C.1 stands) | +| [0041](0041-observability-stack.md) | Observability on Cloudflare — a sleeping Loki + Grafana box, OTLP inward, Sentry for uncaught errors | Proposed, implemented, 13/15 exit criteria pass with evidence (rev. 3; §L "Results"; two items pending — criterion 5's real-isolate measurement, criterion 13's calendar-pending retention; supersedes part of 0040; amends 0022, 0038; deviates from 0007 for the operator UI) | +| [0042](0042-example-analytics.md) | Count which examples people open, by docs guide and starter | Proposed, implemented (rev. 2; ships with 0041, same status) | +| [0043](0043-admin-cutover-to-grafana.md) | `/admin` reads move to Grafana; the writes stay on `/admin/controls` | Proposed, design approved (rev. 2; after 0041's launch; amends 0022) | diff --git a/runner/docs/cloudflare-resources.md b/runner/docs/cloudflare-resources.md index bf060b7802..48ab0e8591 100644 --- a/runner/docs/cloudflare-resources.md +++ b/runner/docs/cloudflare-resources.md @@ -37,7 +37,7 @@ meta-framework — remix, angular, next, next-shadcn, astro, nuxt — plus a `buildersandbox` that runs the static build snapshotter. Deploy with `npx wrangler deploy` from `workers/api/` and `apps/authoring/`, or -let the `deploy-runner-*` workflows do it on merge to `master`. See +let `master.yml`'s path-gated deploy jobs do it on merge to `master`. See [run-and-deploy.md](run-and-deploy.md). ### Preview URLs need a wildcard domain @@ -49,6 +49,40 @@ does not support wildcards — without the custom domain, static shares still wo (they are built in the builder container and served from R2) but live Tier-2 preview does not. +## Observability (o11y worker, Grafana box) + +A third Worker, **`handsontable-demos-o11y`** (`workers/o11y/`), owns the +`/telemetry/*` and `/grafana/*` routes on the same `demos.handsontable.com` +host — `workers_dev: false`, `preview_urls: false` (contract §1: everything +reaches it through the deploy script's `--routes` flags, never a Cloudflare +subdomain). Full setup — buckets, lifecycle, secrets (`/grafana/*` gates through the +Handsontable login broker, ADR-0007 — not Cloudflare Access), the export +destination, deploy ordering — is in +[run-and-deploy.md](run-and-deploy.md#one-time-setup); this is the +resource inventory. + +| Kind | Name | Binding | Notes | +|------|------|---------|-------| +| Durable Object | `InboxWriter` | `INBOX_WRITER` | one instance `main`, EU jurisdiction; ingest dedupe, ledger, fingerprint registry, alert state | +| Durable Object + Container | `GrafanaBox` | `GRAFANA_BOX` | one instance `box`, EU jurisdiction; runs the Loki+Grafana image, woken on demand | +| R2 (EU) | `handsontable-demos-o11y-inbox` | `O11Y_INBOX` | normalised OTLP records awaiting drain; 7-day lifecycle | +| R2 (EU) | `handsontable-demos-o11y-loki` | `O11Y_LOKI_STATE` (read-only in the Worker; the box itself reaches it over S3) | Loki's own chunk/index storage + `state/` clean markers; prefix-scoped lifecycle (T01) | +| R2 (EU) | `handsontable-demos-o11y-maps` | `O11Y_MAPS` | source maps, `sourcemaps/<sha>/<asset path>.map`; 30-day lifecycle | +| Analytics Engine | `runner_events` | `RUNNER_EVENTS` | shared with the API worker's own binding of the same dataset | +| Service binding | `handsontable-demos-api`, entrypoint `O11yUsage` | `API` | o11y usage metering / spend cap RPC, not an HTTP route | +| Rate limiting | namespace `1001` | `RATE_LIMITER` | self-chosen scoping id, provisioned automatically on deploy — no dashboard step | + +The API worker gains one addition of its own: an `O11Y` service binding to +`handsontable-demos-o11y`, entrypoint `O11yHeartbeat` (the watchdog heartbeat +RPC call, not an HTTP route) and a shared `RUNNER_EVENTS` Analytics Engine +binding to the same `runner_events` dataset. + +The Grafana box (`containers/o11y/`) is a single container application — Loki ++ Grafana only, no Tier-2 Sandbox SDK involved — woken by a request to +`/grafana/*` or the `*/10` backlog cron, and stopped again after an idle +window. It is a different container mechanism from the 7 Tier-2 application +images above (`@cloudflare/containers`, not `@cloudflare/sandbox`). + ## Cost guardrails Spend is metered per session into the D1 `cost_ledger` and reconciled nightly by diff --git a/runner/docs/cost-guardrails.md b/runner/docs/cost-guardrails.md index 5d96c29256..32c0673c41 100644 --- a/runner/docs/cost-guardrails.md +++ b/runner/docs/cost-guardrails.md @@ -79,7 +79,17 @@ Cloudflare's cap is also the only thing bounding *concurrency*, not just spend: there is no separate queue in front of the pool, so `Sandbox.max_instances` is exactly the number of visitors who can hold a live preview at once. -### Measuring pool pressure — [ADR-0040](adr/0040-hourly-buckets-and-pool-pressure.md) +### Measuring pool pressure — [ADR-0040](adr/0040-hourly-buckets-and-pool-pressure.md), delivered by [ADR-0041](adr/0041-observability-stack.md) + +> **Implemented, on the ADR-0041 observability branch.** As of this branch, both gaps +> below are closed: the `at_capacity` refusal counter lands in `usage_daily` +> (`recordUsageEvent(env, "at_capacity", …)` beside the 503, `workers/api/src/index.ts`) +> exactly as ADR-0040 C.1 says, and as a `session.start` outcome=`at_capacity` point; +> peak concurrency is a `pool.gauge` Analytics Engine point (`reason: "live"`, +> `telemetry/cron.ts#emitPoolGauge`) sampled every 5 minutes by the API worker's `*/5` +> cron and read by Grafana — not the `usage_hourly` D1 table or the `hour` dimension +> this section originally proposed; ADR-0041 superseded both of those in favor of +> Analytics Engine points, which is what actually shipped. The 5 → 10 decision above was made without the number that should have decided it. Daily totals cannot yield peak concurrency (13,237 sessions over 30 days @@ -309,7 +319,7 @@ day/dimension/value): | `country` | two-letter code from the Cloudflare edge | | `device` / `browser` / `os` | three to six coarse buckets each | | `language` | primary subtag (`en`, `pl`, …) | -| `hour` | hour of day the view landed, `00`–`23` UTC (ADR-0040) | +| `hour` | *planned, not implemented* — hour of day; delivered as an Analytics Engine point per ADR-0041, not as a D1 row | | `bot` | requests identified as bots, excluded from every other bucket | **Never stored:** cookies or any client-side id, IP addresses, user-agent diff --git a/runner/docs/observability-contract.md b/runner/docs/observability-contract.md new file mode 100644 index 0000000000..ccc257a6e3 --- /dev/null +++ b/runner/docs/observability-contract.md @@ -0,0 +1,650 @@ +# Observability contract + +The names, shapes and slot positions every observability component shares +([ADR-0041](adr/0041-observability-stack.md) revision 3, [ADR-0042](adr/0042-example-analytics.md)). +**Permanent document**: it outlives the implementation task board and is what a +dashboard author or a new emitter reads. + +Implemented once, in `packages/runtime/src/telemetry/`, exported as +`@handsontable/demo-runtime/telemetry` — the subpath-export pattern `monitor-inject.ts` +already uses for `@handsontable/demo-runtime/monitor`. The module is pure (no DOM, no +Cloudflare imports), so the API worker, the o11y worker, the authoring app and +`pipeline/` tests import the same definitions. `pipeline/telemetry-contract.test.mjs` +parses the tables below and fails when the module disagrees with this file. + +**Changing this file**: in its own small PR, together with the module, before any code +relies on the change. Metric rows and slots are **append-only**: never reuse, move or +rename a slot or a metric name. Analytics Engine columns are positional, so a moved slot +silently corrupts every stored row and no type checks it. + +## 1. Deployables, routes, ports + +| Name | Path | Notes | +|---|---|---| +| `handsontable-demos-o11y` | `workers/o11y/` | Worker + `InboxWriter` DO + `GrafanaBox` Container class; `workers_dev: false`, `preview_urls: false` | +| Grafana box image | `containers/o11y/` | Loki + Grafana only | + +Routes on `demos.handsontable.com`, owned by the o11y worker, passed as `--routes` flags +in its deploy script (ADR-0020), never in `wrangler.jsonc`: + +| Route | Purpose | Gate (ADR-0041 §B.5) | +|---|---|---| +| `POST /telemetry/collect` | Faro payloads from the authoring app | host/env, bot filter, caps, kind allowlist, rate limit, server scrub | +| `POST /telemetry/lite` | lite beacon from `/d` and `/embed` | same as `collect` | +| `POST /telemetry/v1/logs` | Cloudflare OTLP log export | `x-o11y-secret` | +| `POST /telemetry/deploy` | deploy event from CI | GitHub OIDC token, `x-o11y-secret` fallback | +| `POST /telemetry/hooks/sentry` | Sentry issue-alert webhook | `sentry-hook-signature` HMAC | +| `GET /grafana/_o11y/login` | start a broker sign-in | per-IP rate limit; mints the `__Host-o11y_login` nonce cookie | +| `GET /grafana/_o11y/callback` | broker return; static page, no server-side check of its own | none (fix round M6: the `__Host-o11y_login` cookie + the live `/broker/userinfo` call happen on the `session` row below, not here — this route only serves a hash-pinned static page) | +| `POST /grafana/_o11y/session` | mint the `__Host-o11y_session` cookie | per-IP rate limit; `__Host-o11y_login` cookie (nonce bound), same-origin `Origin`, `@handsontable.com` broker identity | +| `GET /grafana/_o11y/logout` | a same-origin sign-out page (fix round M5) | none; the page itself fires the `POST` below | +| `POST /grafana/_o11y/logout` | clear both cookies | same-origin `Origin` | +| `/grafana/*` | Grafana UI, waking page | the Worker's own session cookie (`O11Y_SESSION_SECRET`, K1 — not Access) | +| `POST /grafana/_o11y/reopen` | manual ledger re-open | session cookie, same-origin `Origin` | +| `GET /grafana/_o11y/admin/<name>` | ADR-0043 read forwarder (after launch) | session cookie, name allowlist, GET only | + +There is no trace route: traces are not exported (ADR-0041 §C.4). + +Ports inside the Grafana box, reached only through `GrafanaBox.containerFetch`: + +| Port | Service | +|---|---| +| 3000 | Grafana (served from sub-path `/grafana/`) | +| 3100 | Loki HTTP (`/otlp/v1/logs`, `/ready`, `/metrics`) | + +## 2. Bindings, variables, secrets + +**o11y worker** (`workers/o11y/wrangler.jsonc`): + +| Name | Kind | Value / purpose | +|---|---|---| +| `INBOX_WRITER` | Durable Object | class `InboxWriter`, one instance `main`, `.jurisdiction("eu")`; owns inbox keys, ledger, dedupe set, fingerprint registry, alert state | +| `GRAFANA_BOX` | Durable Object + Container | class `GrafanaBox`, one instance `box`, `.jurisdiction("eu")`, container `jurisdiction: "eu"` | +| `O11Y_INBOX` | R2 | bucket `handsontable-demos-o11y-inbox` (EU) | +| `O11Y_LOKI_STATE` | R2 | bucket `handsontable-demos-o11y-loki` (EU); the Worker reads only `state/wakes/<wakeId>/clean` markers | +| `O11Y_MAPS` | R2 | bucket `handsontable-demos-o11y-maps` (EU) | +| `RUNNER_EVENTS` | Analytics Engine | dataset `runner_events` | +| `API` | service binding | `handsontable-demos-api` (o11y usage metering, o11y spend, later `AdminReads`) | +| `RATE_LIMITER` | Rate Limiting binding | gates `POST /telemetry/collect` and `POST /telemetry/lite` (ADR §B.5): 100 requests / 60 s per `cf-connecting-ip`, 429 with `Retry-After: 60`, which the browser's Faro transport waits out (runbook "Ingest rate limit") | +| `O11Y_ENV` | var | `production` \| `local` | +| `LOGIN_BROKER_URL` | var | Handsontable login broker base URL (ADR-0007, K1) — same value as `workers/api/wrangler.jsonc`'s own `LOGIN_BROKER_URL` | +| `GITHUB_OIDC_REPOSITORY` | var | `handsontable/examples` | +| `GITHUB_OIDC_WORKFLOW_REF` | var | `<owner>/<repo>/<workflow file path>@<ref>`, exact match — the deploy webhook's OIDC `workflow_ref` claim (ADR §B.5's "issuer, audience, repository, **workflow**") | +| `CLOUDFLARE_ACCOUNT_ID` | var | duplicates `wrangler.jsonc`'s top-level `account_id` — a Worker has no runtime way to read its own account id, and `GrafanaBox` needs it to build the Loki bucket's R2 S3 endpoint and the Analytics Engine SQL API URL | +| `SERVICE_VERSION` | `--var` in the deploy script | full `GITHUB_SHA`, same pattern as the API worker's own row below; falls back to `"dev"` when unset (`wrangler dev`) | +| `LOKI_S3_BUCKET` | var, optional | probe-only override of the Loki bucket name (COMMON.md probe rules); falls back to the production bucket name when unset, so `wrangler.jsonc` need not set it at all | +| `O11Y_EXPORT_SECRET` | secret | `x-o11y-secret` on the export destination and the deploy fallback | +| `SENTRY_HOOK_SECRET` | secret | Sentry internal-integration client secret | +| `AE_SQL_TOKEN` | secret | Analytics Engine SQL API (alert cron; passed to the box for Grafana) | +| `LOKI_S3_ACCESS_KEY_ID`, `LOKI_S3_SECRET_ACCESS_KEY` | secrets | R2 S3 credentials, passed to the box as `envVars` | +| `SLACK_WEBHOOK_URL` | secret | alert channel; never passed to the box | +| `O11Y_SESSION_SECRET` | secret | HMAC key for the Worker's own `__Host-o11y_session`/`__Host-o11y_login` cookies (K1); rotating it logs every signed-in person out at once | +| `RUNNER_EVENTS_CLICKHOUSE_URL` | `.dev.vars` only | local-mode stand-in for the Analytics Engine SQL API's URL (alert queries, §10); defaults to `http://localhost:8123` when absent | +| `O11Y_LOCAL_MINIO_PORT`, `O11Y_LOCAL_CLICKHOUSE_PORT` | `.dev.vars` only | host ports `containers/o11y/compose.yml`'s `minio`/`clickhouse` are published on, reached from the box's Container via `host.docker.internal`; never set in production | +| `O11Y_LOCAL_PUBLIC_ORIGIN` | `.dev.vars` only | the origin `wrangler dev` is actually reachable on, for Grafana's own `GF_SERVER_ROOT_URL` | +| `DEV_ADMIN` | `.dev.vars` only | fail-closed local bypass of the session check | + +The box reaches the Loki bucket over S3 at +`https://<account-id>.eu.r2.cloudflarestorage.com` with `LOKI_S3_*`, scoped to that bucket +only; it writes Loki data and the clean markers there. Lifecycle rules: `browser/` chunks +30 d, `worker/` chunks 90 d, index 90 d, `state/` 30 d. + +**API worker** additions (`workers/api/wrangler.jsonc`): + +| Name | Kind | Value / purpose | +|---|---|---| +| `RUNNER_EVENTS` | Analytics Engine | dataset `runner_events` | +| `O11Y` | service binding, entrypoint `O11yHeartbeat` | `handsontable-demos-o11y`, `heartbeat()` RPC for the watchdog (not an HTTP route — A-C1) | +| `SERVICE_VERSION` | `--var` in the deploy script | full `GITHUB_SHA` | +| `SENTRY_SCOPE` | var | `full` \| `uncaught` (§11) | +| `CF_ACCOUNT_ID` | var | GraphQL Analytics API account tag; also scopes the Analytics Engine SQL API read below | +| `AE_SQL_TOKEN` | secret | Analytics Engine SQL API (Account Analytics Read) — production read side of the nightly `example_daily` rollup (ADR-0042 §5, C-I1). Unset means `reconcile.ts#queryExampleEventTotals` throws instead of rolling up an empty day | +| `o11y-logs` | export destination name | referenced from `observability.logs.destinations` | +| `*/5 * * * *` | cron | `pool.gauge`, `budget.gauge`, o11y heartbeat check | + +**Authoring app**: `VITE_SENTRY_RELEASE` (the `GITHUB_SHA` define) doubles as +`service.version`; `VITE_TELEMETRY_LOCAL=1` enables the local path (§10) and is never set +for a production build; `VITE_SENTRY_SCOPE` = `full` | `uncaught` (§11). + +**Headers**: `x-hot-session` (page-load id, browser → API worker), `x-o11y-secret`, +`sentry-hook-signature`, the `__Host-o11y_session`/`__Host-o11y_login` cookies +(`/grafana/*`'s own session, never forwarded to the container). Both cookies use `Path=/` +(the `__Host-` prefix requires it), so the browser also sends them to `/api`, `/d` and +the authoring app — none of those read them (the API worker reads only +`Authorization`/`X-MCP-Secret`; authoring is static), but it means the cookie header +itself must be stripped before proxying to the box (see `/grafana/*`'s gate row above). +`x-o11y-grafana-user` (set by the o11y worker for Grafana `auth.proxy`; stripped from +every client request), `X-Scope-OrgID` (Loki tenant: `browser` \| `worker`). + +## 3. Attributes + +All of these are **OTLP resource attributes** on every record, so Loki's `otlp_config` +can promote them to labels. + +| Key | Values | Loki label | AE slot | +|---|---|---|---| +| `service.name` | `demos-authoring`, `demos-api`, `demos-o11y`, `demos-embed` | `service_name` | `blob1` | +| `service.version` | full git SHA | no | `blob2` | +| `deployment.environment.name` | `production`, `local` | `deployment_environment_name` | `blob3` | +| `hot.surface` | `authoring`, `share`, `embed`, `d`, `api`, `demo-runtime`, `o11y` | `hot_surface` | `blob4` | +| `hot.tier` | `1`, `2`, `static`, `none` | `hot_tier` | `blob5` | +| `hot.framework` | a key of `config/frameworks.json` (every docs-example framework is one), or `none` | `hot_framework` | `blob6` | +| `hot.ht_major` | `15`…`19`, `next`, `none` | `hot_ht_major` | `blob7` | +| `hot.outcome` | per metric, see §5; `none` on a record no metric describes | `hot_outcome` | `blob8` | + +Ingest bounds both open labels, since each distinct label tuple is a Loki stream +(5000 per tenant): a `hot.framework` outside the list above, or a `hot.outcome` +outside the set of the item's metric (`none` when there is none), becomes `other`. +A stored browser record (exception, log, event) always carries `hot.outcome` = +`none`; only a measurement's AE point keeps its metric outcome. That caps the +browser tenant at 4410 label tuples (collect 7 × 4 × 21 × 7, lite 2 × 1 × 21 × 7). +The box's Loki sets `max_global_streams_per_user` to 20000, over 4× that worst case; +`pipeline/o11y-label-cardinality.test.mjs` reads the limit from the config and fails +when the reachable tuples cross it. +Ingester memory grows with the streams that actually receive lines and the bytes +pushed, not with the limit, and a stream costs kilobytes (labels, index entry, head +block), so 20000 fits easily in the box's 4 GiB `standard-1` container. +`packages/runtime/src/telemetry/attrs.ts#KNOWN_FRAMEWORKS` mirrors +`config/frameworks.json`. + +Structured metadata only — never a Loki label, never an Analytics Engine index: +`hot.demo_id`, `session.id` (an in-memory page-load id), `cf.ray`, `hot.kind` (the Faro item +kind: `exception`, `log`, `event`, `measurement`). + +Diagnostic tags — flat, non-dotted, never a Loki label, never an Analytics Engine +index, but hoisted to structured metadata alongside the dotted keys above (§6, +`convert.ts#hoistAttributes`'s `STRUCTURED_KEY_SET`): `handled`, `context`, +`sentry_event_id`, `versions_fetch_attempts`, `versions_fetch_outcome`, +`versions_fetch_elapsed_bucket`, `versions_fetch_online`, `api_base_origin`, +`net_effective_type`. Each is a boolean flag, an enum-like/bucketed value, an +opaque platform id, or the reporting call site's own name — never user or +request content. + +AE-only transport keys — never a Loki label, never structured metadata, never +hoisted by `hoistAttributes` at all: `hot.bucket`, `hot.reason`, `hot.fingerprint`, +`hot.metric_kind`, `hot.ref`, `hot.area`. Survive the same browser/ingest attribute +allowlist as every key above (so the browser can transmit them at all), but exist +only to carry an Analytics Engine column (§4/§5) through a Faro item's raw +`context`/`attributes`, read directly from the wire body by +`normalise/browser-attrs.ts#readAeOnlyAttrs` before `hoistAttributes` ever runs — +for a metric such as `example.open` that is never written to the inbox or Loki at +all (§6, ADR-0042). + +**Never sent to the o11y stack**: the user pseudonym, an email, an IP, a user-agent +string, a query string or fragment, authored code (including Babel code frames), chat +text, console output, `url.full`, geo or ASN attributes. + +One bounded exception, demo-runtime records (§6): a relayed preview message is sent +only as its §7 fingerprint shape (`fingerprintShape`: code frame stripped; quoted +strings, numbers, URLs, timestamps and keystroke-ladder identifiers replaced; ≤200 +chars), never raw, never with its stack or URL. Unquoted prose a demo itself passes to +`new Error(…)` or `console.error(…)` survives that normalisation. + +## 4. Analytics Engine layout (`runner_events`) + +`index1` = metric name (the sampling key); queries filter on `index1` directly. + +| Slot | Column | Meaning | +|---|---|---| +| `index1` | `metric` | metric name (§5) | +| `blob1` | `service_name` | §3 | +| `blob2` | `service_version` | §3 | +| `blob3` | `environment` | §3 | +| `blob4` | `surface` | §3 | +| `blob5` | `tier` | §3 | +| `blob6` | `framework` | §3 | +| `blob7` | `ht_major` | §3 | +| `blob8` | `outcome` | §3, §5 | +| `blob9` | `reason` | metric-specific qualifier (§5) | +| `blob10` | `route_class` | API route class, e.g. `api/versions` | +| `blob11` | `fingerprint` | §7 | +| `blob12` | `demo_id` | demo id where the metric concerns one demo | +| `blob13` | `model` | LLM model id | +| `blob14` | `provider` | upstream or import provider | +| `blob15` | `device` | `desktop`, `mobile`, `tablet` | +| `blob16` | `bucket` | docs or starter bucket (`18.1`, `next`) | +| `blob17` | `kind` | ADR-0042: `docs`, `starter`, `saved`, `import`, `payload` | +| `blob18` | `ref` | ADR-0042: guide path or starter id | +| `blob19` | `area` | ADR-0042: first breadcrumb element of a docs example | +| `blob20` | — | unassigned | +| `double1` | `count` | 1 per point unless pre-aggregated | +| `double2` | `duration_ms` | latency | +| `double3` | `value` | generic measurement (web-vital value, gauge level, seconds, percent) | +| `double4` | `usd` | cost | +| `double5` | `tokens_in` | LLM input tokens | +| `double6` | `tokens_out` | LLM output tokens | +| `double7` | `bytes` | payload or artifact size | +| `double8` | `cap` | the limit a gauge is measured against | +| `double9`–`double20` | — | unassigned | + +**Reading rule**: Analytics Engine samples at write and read time. Every count is +`SUM(_sample_interval * double1)`, every percentile a weighted quantile, never `COUNT()`. +Queries go through one helper that allowlists Analytics Engine's documented functions; +the local ClickHouse shim accepts more. + +## 5. Metric registry + +Outcome values are the only strings allowed in `blob8` for that metric. + +| Metric | Emitted by | Blobs used | Doubles | Outcomes / reason | +|---|---|---|---|---| +| `preview.ready_ms` | browser | surface, tier, framework, ht_major, outcome, bucket | duration_ms | `ready`, `error`, `timeout`, `abandoned` | +| `sandpack.compile_ms` | browser; the settled (last) compile of each edit burst, closed by 2 s without a compile or compile error. The mount's compile is not sent (`preview.ready_ms` covers first load) | tier, framework, ht_major, outcome | duration_ms | `ok`, `error` | +| `sandpack.compile_error` | browser | framework, ht_major, fingerprint | count | — | +| `sandpack.bundler_unreachable` | browser | ht_major | count, duration_ms | — | +| `preview.runtime_error` | browser | surface=`demo-runtime`, tier, framework, ht_major, fingerprint, reason | count | reason: `uncaught`, `console`, `network`, `stderr` | +| `version.switch` | browser | framework, ht_major (to), reason (from), bucket | count | — | +| `bucket.resolve_ms` | browser | bucket, outcome | duration_ms | `ok`, `error` | +| `session.start_ms` | browser | framework, ht_major, outcome, reason | duration_ms | outcomes as `session.start`; reason `cold`, `warm` | +| `hmr.roundtrip_ms` | browser | framework, ht_major | duration_ms | — | +| `web_vital` | browser, beacon | surface, framework, ht_major, reason, device, demo_id | value | reason `LCP`, `INP`, `CLS`, `TTFB` | +| `error.uncaught` | browser, beacon | surface, fingerprint, demo_id | count | — | +| `error.handled` | browser, API worker | surface, route_class, fingerprint | count | — | +| `example.open` | browser (ADR-0042) | kind, ref, area, framework, ht_major, bucket, reason (`entry`) | count | reason `deep-link`, `picker`, `switch`, `version-switch`, `fork` | +| `example.engaged`, `example.forked`, `example.shared`, `example.downloaded` | browser (ADR-0042) | kind, ref, area, framework, ht_major, bucket | count | — | +| `example.saved` | API worker (ADR-0042) | kind, ref, area, framework, ht_major, bucket | count | — | +| `api.request` | API worker | route_class, outcome | count, duration_ms | `2xx`, `3xx`, `4xx`, `5xx` | +| `session.start` | API worker | framework, ht_major, outcome | count, duration_ms | `ready`, `at_capacity`, `container_starting`, `boot_timeout`, `budget_denied`, `error` | +| `session.end` | API worker | framework, reason | count, value (awake s) | reason `pagehide`, `sleep_after`, `teardown_failed`, `budget_closed` | +| `container.boot_ms` | API worker | framework, outcome, reason | duration_ms | `ready`, `window_exceeded`, `error`; reason `cold`, `warm` | +| `pool.gauge` | API worker `*/5` | reason (`live`, `builder`) | value (awake), cap | — | +| `budget.gauge` | API worker `*/5` | reason (tier) | value (percent of ceiling), usd | — | +| `snapshot.build` | API worker | framework, outcome, reason | count, duration_ms, bytes | `ok`, `failed`; reason `inline`, `detached` | +| `serve.share`, `serve.d`, `serve.embed` | API worker | outcome, demo_id | count, bytes | `2xx`, `304`, `4xx`, `5xx` | +| `chat.answer` | API worker | model, outcome | count, duration_ms, usd, tokens_in, tokens_out | `answered`, `denied`, `error` | +| `chat.edit` | API worker | outcome | count | `proposed`, `applied`, `undone` | +| `theme.ai` | API worker | model, outcome | count, duration_ms, usd | `answered`, `denied`, `error` | +| `import.url` | API worker | provider, outcome, reason | count, duration_ms | `ok`, `refused`, `error` | +| `payload.boot` | API worker | framework, outcome | count | `ok`, `error` | +| `reconcile.run` | API worker cron | outcome | count, duration_ms, usd (billing total WRITTEN this run — not a delta against the estimate; D-M15 fix round) | `ok`, `skipped`, `error` | +| `o11y.ingest` | o11y worker | reason, outcome | count, bytes | `accepted`, `dropped`, `duplicate`; reason = gate | +| `o11y.drain` | o11y worker | reason, outcome | count (objects), duration_ms, bytes, value (records dropped for being too old — `reject_old_samples_max_age`) | `ok`, `partial`, `error`; reason `backlog`, `visit`, `reopen` (this batch replayed reopened keys) | +| `o11y.wake` | o11y worker | reason, outcome | count, duration_ms (to ready) | reason `backlog`, `visit`; outcome `clean`, `unclean` | +| `o11y.backlog` | o11y worker cron | — | value (oldest age s), bytes | — | +| `o11y.alert` | o11y worker cron | reason (rule id), outcome | count | `fired`, `resolved` | + +`o11y.wake`'s `duration_ms` is wake-to-ready time — from the wake starting to the +box's first successful `isReady()`, sourced from `wake:<wakeId>.readyMs` (§8) — on both +the `clean` and `unclean` outcome. `duration_ms = 0` means the box never became ready +during that wake, not a genuinely instant boot. + +The point's own **timestamp** is a different moment: it is written at +**resolution** — the next `backlog()` call that runs `resolveWakes()` (§8) and finds this +wake `over`, not when the wake itself started. Locally, with no cron firing, that can lag +the actual wake by an hour or more; in production it lags by at most one `*/10` cron +period. Two wakes resolved in the same `resolveWakes()` call get timestamps identical to +the millisecond even though their wakes started at different times — expected, not a bug. + +`sandpack.compile_error` counts Tier-1 compile failures from +both places they occur: +- a bundler diagnostic (`show-error` with no frames — the module never evaluated); +- the parcel pre-transpile's own babel parse failure (every starter except `vue-cli`). + That source never reaches the bundler and the last good render stays on screen, so + `SandpackRuntime` reports it from the transpile catch itself (`isTranspileFailure`): + at mount (a saved/shared/`?payload=` demo that does not parse — counted at once), and + on the edit path for the newest push only. + +The signal is the transpile catch, never the message shape: a runtime `SyntaxError` +(`JSON.parse`, `new Function`) is relayed by the preview like any other throw and stays +`preview.runtime_error`. Compile errors go through the same edit-burst collapse as +`preview.runtime_error` (below), keyed by kind, so a burst counts at most one — from its +final state. A compile failure of the burst's newest edit also **replaces the run**: the +preview never ran that code, so what it relays for the rest of the burst (a +keystroke-prefix rung still in flight, a re-render warning) is from code already typed +past and is dropped. One typed broken line = one `sandpack.compile_error`, no +`preview.runtime_error`. The next edit re-arms runtime reports. No error card and no +Sentry capture is added for the edit-path failure; the mount-path Sentry capture +(`Tier1CompileError`) is unchanged. + +`preview.runtime_error` counts broken preview states, not relays. Reason `console` is a +`console.error`; a console warning is not counted (§6). The preview +re-runs on every keystroke, so one typed line relays a whole keystroke-prefix ladder +(`s is not defined`, `se is not defined`, …, then the line's real error). The browser +collapses it (`apps/authoring/src/demoEventCollapse.ts`) before the facade: +- an edit that re-runs the preview (a non-quiet workspace write, a file add, delete or + rename) opens or extends a burst, and discards what the previous run reported; +- on Tier 1, an edit whose transpiled sandbox matches the running one (a closing `;`, + whitespace, a trailing comma) re-runs nothing, so the burst ends with the running + sandbox's reports that have not been counted yet. If the bundler rejected that + sandbox (a frameless `show-error`), its `sandpack.compile_error` is that result and + replaces the run; a pre-transpile failure never ran, so it is not the running + sandbox's; +- on Tier 1, the bundler runs one compile at a time, so a run's reports can still arrive + after the next edit has been dispatched. A new run starts at the bundler's `start` + message for a pushed compile (`onPushOutcome("rerun")`), not at dispatch, and what the burst held until + then came from the run it replaces and is dropped. A pre-transpile failure of the + newest edit is kept, because no run of that edit will start. A run that never starts + (a stalled or unreachable bundler) drops nothing, and the burst closes on its quiet + window as usual; +- 2 s (`DEMO_EDIT_SETTLE_MS`) after the last edit the burst closes, and the last run's + reports are emitted, one per §7 fingerprint; +- outside a burst (first load, a user interaction, a Tier-2 rebuild that reports after + the burst closed) a report is emitted at once; +- a fingerprint counts once until the next edit or preview mount, and at most 50 + (`DEMO_COLLAPSE_CEILING`) points per page load. +An async report of the previous run (a timer, a rejected promise, a failed request) that +fires after the next run started can still add one point. On Tier 2, where a rebuild +outlasts the 2 s window, a superseded rebuild's report can land after the burst closed +and count on its own. The Sentry side is not behind this collapse; its relay budgets are +unchanged. + +`example.saved` is written by the API worker when an editor Save (`PATCH /api/demos/:id` +with `files`) finishes its rebuild, because the rebuild can outlast the visitor's stay on +the page. It carries the same values the browser's other `example.*` events do for a saved +demo: `kind=saved`, `ref` = the demo id, `framework` = the demo row's, and `ht_major` = the +body's `exampleHtMajor`, which is the major the editor opened the demo at. `area` and +`bucket` stay empty. `blob1`/`blob2` name `demos-api`. + +What gates the count is a successful rebuild whose request carries a valid +`exampleHtMajor` (one of §3's `ht_major` values), whoever sends it. The editor sends the +field only while its own telemetry gate is open (§10). The rebuild and the point are +registered with `ctx.waitUntil`, so a client disconnect within the 30 s grace does not +cancel them. Every rebuild response carries `exampleSaved` (whether the point was +written). The editor emits the browser `example.saved` only when a response lacks that +key, i.e. an API that does not count saves; that fallback can be removed once every +deployed API sends the marker. + +A build that fails on the demo's own input is client input (`isUserBuildError` in +`workers/api/src/share.ts`), on any route that builds inline: `POST /api/demos`, `PATCH +/api/demos/:id`, `POST /api/mcp/demos` and `PATCH /api/mcp/demos/:id`. That means one of: +- the build command exited with a code from 1 to 125; +- the install failed with `ERR_PNPM_NO_MATCHING_VERSION`, `ERR_PNPM_FETCH_404`, + `ERR_PNPM_SPEC_NOT_SUPPORTED_BY_ANY_RESOLVER` or `ERR_PNPM_BAD_PM_VERSION` (a + dependency the author named). + +It answers `422 {"error":"<phase> failed: <detail>","code":"build_failed","detail":<the +build error, one line>}`. `error` carries the diagnostic because MCP clients read only +that field; the editor keys on `code`. `api.request` records it as `4xx`, so `api-5xx-rate` +does not count it. `snapshot.build` still records `failed`, and the +`snapshot-build-failed-rate` alert is the backstop: it fires when one framework has more +than 50 % failed builds over 30 min with at least 10 failed. The stored demo is +unchanged, because the build runs before anything is written. + +Everything else stays `5xx`: exit code 126, 127 or 128 and above (not executable, not +found, killed by a signal), a result without an exit code, any other install failure +(`ERR_PNPM_FETCH_5xx`, a reset or timed-out connection), and any other throw. So does +any failure whose output names infrastructure, whatever the exit code: `ENOTFOUND`, +`ECONNRESET`, `ETIMEDOUT`, `EAI_AGAIN`, `fetch failed`, `Failed to fetch`, `heap out of +memory`, `signal SIGKILL`, `worker exited`, and Next's `` `next/font` error `` or a +`fonts.googleapis.com` fetch. This last rule is a match on message text: a tool that +rewords these lines moves its failure into the 422 class. + +`serve.share` locally: under `vite dev` (what `pnpm dev:full` serves), React +StrictMode runs the share page's load effect twice, so one `/share/<id>` view gives 2 +points. A production build gives 1 (measured on `vite preview`). + +`payload.boot` records the Theme Builder hand-off at both ends: +- `ok` and `error` at `POST /api/payload`, when the link is minted; +- `error` (`framework=other`) at the playground boot, `GET /api/payload/:id`, when the + link cannot boot: a miss (expired or never minted), a malformed id, or a KV failure. + A link that boots adds no point, so each hand-off counts one `ok` at most. + +The server-side `error` point is the record of a `?payload=` boot that failed. The +browser shows the "expired" message for the 404 without reporting it. `error.handled +context=payload-boot` is only the browser failing to reach the API at all (a network +error), which the server never sees. + +## 6. Browser facade and Faro + +The app never calls Faro directly; it calls one facade, implemented with Faro: + +```ts +interface Telemetry { + metric(name: MetricName, values: { duration_ms?: number; value?: number; count?: number }, attrs: HotAttrs): void; + event(name: EventName, attrs: HotAttrs & Record<string, string>): void; + error(err: unknown, context: string, attrs?: HotAttrs): void; // handled errors + pageLoadId(): string; // minted in memory at page load +} +``` + +Faro configuration: session tracking disabled; only the errors and web-vitals +instrumentations; no `user` meta; the facade sets `session.id` = page-load id on every +item; transport to same-origin `/telemetry/collect`; `beforeSend` = `scrubTelemetry` then +the shared noise gates. + +Faro's global `dedupe` stays on for `pushError`: consecutive identical +exceptions (same type, message, stack, context) collapse to one item, with no time +window. So `error.uncaught`/`error.handled` counts — including the dashboard panels +built on them — are **reports**, not occurrences: a burst of identical errors counts as +one. Sentry, not this pipeline, is the occurrence counter (ADR §F.3). + +What the o11y worker does with each Faro item at ingest: + +| Faro item | Analytics Engine | Inbox (Loki `browser` tenant) | +|---|---|---| +| measurement whose `type` is a browser metric in §5 | one point | **none** | +| `web-vitals` measurement | one `web_vital` point per vital | **none** | +| exception with `context.handled = "true"` | `error.handled` | one log record, symbolicated at drain | +| other exception | `error.uncaught` | one log record, symbolicated at drain | +| event named `example.*` | one point | **none** | +| other event, log | — | one log record | + +**Demo-runtime records.** Each report that survives the collapse (§5) +is also one handled Faro exception, through `Telemetry.error` with context +`demo-runtime`. Its `type` is `DemoError`, `DemoUnhandledRejection`, `DemoConsoleError`, +`DemoNetworkError` or `DemoStderr`. Its value is the §7 fingerprint shape (§3). It has +no stack, and the §3 labels include `hot.surface = demo-runtime`. At ingest it becomes +an `error.handled` point (surface `demo-runtime`, never feeding the new-fingerprint +alert) and one Loki line, which the "Recent demo-runtime errors" panels read with +`{hot_surface="demo-runtime"} | hot_kind="exception"`. A `console-warn` report is +neither counted nor recorded: a warning is context, not a fault (DEV-2539), and +Handsontable's own load-time notices would otherwise count on every preview load. Faro's +`pushError` dedupe applies, so an identical record in two consecutive bursts is sent +once. The `preview.runtime_error` metric, not the line count, is the counter. + +A Faro measurement or web-vitals item (this table's scope — the item +kinds `normalise/faro.ts` handles) is AE-only; Loki holds logs, events and exceptions. +This was a contract/ADR mismatch, not an implementation bug — ADR §F.1 ("Counts and +latencies go to Analytics Engine; Loki holds the text") already said this; requiring a +stored record for every measurement too would make measurements +~99% of the browser Loki tenant's lines and drained bytes for no reader: +no dashboard panel parses a measurement's `{"duration_ms":N}`-shaped body, so the AE +point was always the only consumer. A measurement/web-vitals item still gets a +hash-only `ingestItem` (no `record`) so a retried/redelivered batch cannot double-write +its Analytics Engine point — the same dedupe-only shape an `example.*` event already +used above. + +**Scope note, not yet fixed**: §9's lite-beacon path (`workers/o11y/src/lite.ts`, +`POST /telemetry/lite`) is a separate converter and still stores a vital beacon's record +today (the triage's own `LCP=172` inbox-record example) — this ruling was not extended +there. A future consistency pass may want to. + +## 7. Fingerprint + +`fingerprint(context, message)` = `<context>:<16 hex chars of FNV-1a 64 over the +normalised message>`, synchronous and identical in browser and Worker; the normalisation +is `normalizeMonitorMessage` plus `stripCodeFrame`. For `hot.surface = demo-runtime`, +keystroke-ladder shapes (`"<identifier> is not defined"` and similar) collapse to one +fingerprint per shape. Demo-runtime fingerprints never feed the new-fingerprint alert. + +`context` is caller-chosen (typically `hot.surface` or a metric name) and MAY itself +contain further `:`-separated segments, e.g. `docs-example-load:fetch` or +`npm-registry:version-exists` — a call-site path, not always a single flat token. A +client-supplied fingerprint (Faro's own `payload.fingerprint` wire field, or +`context["hot.fingerprint"]`) is trusted only when it passes the ONE shared validator +(`isValidFingerprint`, `packages/runtime/src/telemetry/fingerprint.ts` — also used by +`normalise/faro.ts`'s `resolveFingerprint` and `normalise/otlp.ts`'s +`apiFingerprintFeed`, never a second, independently drifting copy of the shape): +anchor on the LAST `:`, followed by exactly 16 lowercase hex characters, with zero or +more earlier `:`-separated segments in `context`, each drawn from `[a-z][a-z0-9._-]*`; +the whole `context` half is capped at 128 characters. A value that does not match is +discarded, never stored or forwarded to Slack verbatim. + +## 8. Inbox + +Normalised records are OTLP JSON log records (`resourceLogs` shape), one tenant per +object: + +```text +inbox/<tenant>/<yyyy-mm-dd>/<hh>/<seq:012d>.ndjson.gz # inbox bucket; tenant = browser | worker +state/wakes/<wakeId>/clean # Loki bucket; written by the box on a clean stop +``` + +`<seq>` is a counter in `InboxWriter` storage, incremented in the transaction that +records the key. Each NDJSON line is one OTLP `ResourceLogs` object. The arrival time is +**not** part of the record (it would break dedupe); it lives on the storage row. The +dedupe hash is computed over the decoded, scrubbed record before timestamps are stamped. + +`InboxWriter` storage: + +| Key | Value | +|---|---| +| `seq` | last issued sequence | +| `row:<n:012d>` | pending records with their arrival time, ≤ 1 MB per row. `<n>` is zero-padded to 12 digits so native ascending key order equals arrival order (`pack.ts#collectRowBatch`) | +| `key:<inbox key>` | `written` \| `provisional:<wakeId>` \| `rejected:<reason>` — **never `committed`** (see `done:`, below) | +| `done:<inbox key>` | `1` — a **committed** key, moved OUT of `key:` on commit (same write that deletes `key:<inbox key>`) | +| `hash:<yyyymmdd>:<sha256>` | first-seen epoch ms; 24 h window, checked across the current and previous UTC-day bucket | +| `fp:<fingerprint>` | first-seen epoch ms (exact registry for the new-fingerprint alert) | +| `fpts:<firstSeenMs:015d>:<fingerprint>` | same first-seen epoch ms as its `fp:` twin — a time-ordered secondary index (G1 fix round, B-C1/A-I1 remainder) so the new-fingerprint alert can do a bounded `start`/`end` range read instead of listing the whole (alphabetically, not chronologically, ordered) `fp:` prefix every tick. Written/deleted together with its `fp:` twin, always | +| `alert:<rule>` | `{ state: firing \| resolved, since, lastNotified }` | +| `alertMeta:newFingerprintCursorKey` / `alertMeta:newFingerprintAnnouncedKeys` | the new-fingerprint alert's keyset cursor (an `fpts:` key) and a JSON array of the `fpts:` keys it already announced past that cursor. The cursor lags 2 minutes behind the tick, so the next tick reads recent entries again; the announced set makes sure each fingerprint is announced exactly once (F35) | +| `wake:<wakeId>` | `{ startedAt, reason, over: boolean, readyMs? }` — over when a newer wake started or the container is not running; **deleted once fully resolved** (see below). `readyMs` (F8) is wake-to-ready time in ms, written once by `InboxWriterApi.recordWakeReady(wakeId, readyMs)` — called by `GrafanaBox` on the wake's first successful `isReady()`, first call wins, a no-op for an already-resolved (deleted) wake — and copied onto the resolved `o11y.wake` point as `duration_ms` (§5); absent while the box has not yet become ready | +| `rejectedEvent:<ms:015d>:<inbox key>` | rejection reason (string) — a chronological audit/alert log (G1 fix round, row 19 / B-C1/A-I1 remainder), written by both a full rejection (`ledger.ts#rejectKey`) and a **partial** one (`ledger.ts#recordPartialReject`, see below). The `rejected-inbox-key` alert fires on a RECENT (last hour) count here, not on `rejectedKeyCount()`'s never-pruned total, so it resolves once rejections stop instead of firing forever after the first one ever seen | +| `drainsPaused` | boolean (o11y spend cap). Set on every alert tick from the `o11y-spend-cap` result, so it follows the runtime budget override. The cron decides its backlog wake only after the alerts ran (F37). While it is set, the cron never wakes the box for the backlog, and `GrafanaBox.drainStep` pushes nothing, whatever woke the box. A visit wake still starts the box and serves Grafana (ADR §G) | +| `heartbeat` | `{ lastCron, lastIngest }` | + +Limits: records over 256 KB are dropped; requests to Loki carry at most 1 MB +decompressed. Every free-text string (Faro/OTLP body, +message, attribute value) is truncated to this same 256 KB before any scrub/redact +regex runs over it (`SCRUB_TEXT_MAX_CHARS`, `packages/runtime/src/telemetry/scrub.ts`) +— a ReDoS defense-in-depth independent of each pattern also being made linear-time. +The fingerprint normaliser (`normalizeMonitorMessage`, §7) is bounded separately, to a +much smaller 4096 chars, since its own output is always sliced to 200 chars regardless. + +The pre-scrub truncation above shares the same 256 KB limit as the +record-size drop, so a field that actually gets truncated still leaves the record over +the drop cap — an oversize record is genuinely **dropped**, never truncated down to fit. +Only the inbox/Loki record and its first-seen `fp:` entry are lost this way; the item's +Analytics Engine point (e.g. `error.uncaught`/`error.handled`) is still written — dropping +is a size decision at the inbox/Loki layer only, not an ingest-wide refusal. + +**Bounded storage** (the resolve/drain/backlog paths must never scan committed history): +- A `key:` entry only ever holds a **live** state (`written`, `provisional:<wakeId>`, or + a genuine `rejected:<reason>`). The moment a key is confirmed clean-committed, its + `key:<inbox key>` entry is deleted and a `done:<inbox key>` marker takes its place in + the same write — `key:` therefore never grows with committed history, only with what is + currently open or in flight. `done:` entries are pruned once their embedded date is + older than the 7-day inbox-object retention (the ten-minute cron path, via + `InboxWriter.backlog()`), using a bounded `start`/`end` range delete (`done:inbox/<tenant>/` + through the cutoff date), never a full-prefix scan. +- A `wake:<wakeId>` entry is deleted as soon as `resolveOverWakes` fully resolves it (every + provisional key under it moved to `written` or `done:`) — not merely flagged. The + `wake:` prefix therefore only ever holds the (at most one) currently-active wake plus + any wake whose resolution crashed mid-way, never all-time history. +- `hash:` is bucketed by UTC calendar day (`hash:<yyyymmdd>:<sha256>`) instead of one flat + set; a dedupe check reads exactly the current and previous day's buckets (the 24 h window + can never span more than those two), and stale buckets (2+ days old) are pruned with a + bounded range delete on the same cron path. +- `fp:` keeps its flat shape (nothing reads it by date range), but is swept by a bounded, + cursor-paginated TTL prune (default 90 days) on the same cron path, so it does not grow + forever either. +- `POST /grafana/_o11y/reopen`'s window WIDTH (`toMs - fromMs`) is capped to the same + 7-day retention (`ledger.ts#reopenWindowExceedsRetention`) — a wider request is + refused up front. This does not require the window itself to be recent: a `[from, to)` + pair from long ago, 7 days wide or narrower, is accepted too, it just finds nothing to + reopen, since `done:`/`hash:` past the retention window are already pruned (Loki's own + `reject_old_samples_max_age` is 7d too). The route also accepts a request whose + `content-type` merely contains `application/json` as a parameter (e.g. with a charset), + not only an exact match — still enough to force a CORS preflight for any cross-origin + caller (the CSRF hardening this exists for). +- A manual reopen of a **committed** key reads `done:`, moves it back to `key:<inbox + key> = written`, and deletes the `done:` entry. + +**Additional bounded-storage invariants:** +- **Pack alarm, bounded.** The alarm pages `row:` in small chunks rather than loading + every pending row into memory before packing, accumulating up to one packed object's + own ~4 MB budget per round, looping until the backlog is drained or + `MAX_OBJECTS_PER_ALARM` (25) objects have been packed this invocation. See + `pack.ts#collectRowBatch`. +- **DO storage 128-key batch limit.** Cloudflare's SQLite-backed Durable Object + storage API caps `get`/`put`/`delete` at 128 keys/pairs per call + (<https://developers.cloudflare.com/durable-objects/api/storage-api/>, fetched + 2026-09-24: "Supports up to 128 keys at a time" / "up to 128 key-value pairs at a + time"). Every multi-key call in `InboxWriter` chunks through `storage.ts`'s + `getManyChunked`/`putChunked`/`deleteChunked` — `checkDuplicates`, `newFingerprintWrites`, + `pruneLedger`/`pruneHashBuckets`/`pruneFingerprintRegistry`, `finalizeWakeResolution`, + `markKeysProvisional`, `reopenWindow`, `commitPackedObject`, and `ingest`'s own + transaction `put`. `finalizeWakeResolution`/`reopenWindow` run their whole put+delete + sequence inside one `storage.transaction()` — chunking alone, without that, would let + a crash between chunks leave a partial write. +- **Prune throughput.** `hash:`/`done:` prune batch sizes are 5,000 rows/tick (still + chunked to 128 per actual `delete()` call) — ADR §D's own 10× headroom projects + ~220,000 worker records/day, which 500/tick × 144 ten-minute ticks/day (72,000/day) + falls behind at roughly 3× today's traffic. `rejected:` `key:` entries are pruned too, + past the same 7-day retention (filtered by value, since `key:` mixes live and rejected + states chronologically — see `ledger.ts#pruneLedger`). +- **Drain partial-400 durability.** A key with at least one chunk + accepted (2xx) and at least one chunk permanently rejected (400) now stays + `provisional` (not `rejected`) — its accepted content follows the normal + written→provisional→committed path, so an unclean stop before Loki's local flush + still triggers an automatic replay instead of being silently unrecoverable except by + manual reopen. Only a key with ZERO accepted chunks stays `rejected`. See + `drain.ts#drainKey`'s own doc comment. +- **Drain refusals.** A `429` whose body names Loki's stream limit (`Maximum active + stream limit exceeded`) is never retried (a retry can answer 204 with the excess + streams dropped) and never rejects: the table may have been filled by earlier keys, + so that key is deferred and stays `written`, with no `rejectedEvent`. Its tenant is + then excluded for the rest of the wake (`streamLimitedTenants` in the box's storage; + `nextWrittenKeys` pages past it), so the batch fills with the other tenant's keys, and + one `o11y.drain.stream_limit` warning line names the tenant and Loki's message. Any + other `429`, and a `5xx`, stays transient and stops the batch. An inbox read that + throws defers only that key; the rest of the batch still pushes and commits. A batch + of only deferred keys ends the wake's drain once no un-excluded tenant has keys left. +- **Symbolication read caps.** One inbox object reads at most 32 distinct maps + (`MAX_MAP_KEYS_PER_CALL`, first-seen order), one body adds at most 8 of them + (`MAX_NEW_MAP_KEYS_PER_BODY`), and at most 128 frames are looked up per body + (`MAX_FRAMES_PER_BODY`), so a drain step of 10 objects stays at a few hundred of the + Workers limit of 10,000 subrequests. Frames past a cap stay byte-for-byte and are + reported as `o11y.symbolicate.skip` with reason `over_cap`, plus one aggregate + `over_cap` line with the call's capped `frames` and `keys` that `MAX_SKIP_REPORTS` + never suppresses. + +## 9. Lite beacon payload + +`POST /telemetry/lite`, a JSON body, **≤ 2 KB**, sent with `navigator.sendBeacon(url, json)` — +a plain string, not a `Blob`, so the browser sends its own default +`Content-Type: text/plain;charset=UTF-8`, never `application/json` (the +route itself does not check or require a content type either way, so this is a fact about +what ships on the wire, not a gate). The body itself is still JSON: + +```json +{"v":1,"t":"err","s":"embed","demo":"r-react-18-0-0","ht":"18","fw":"react","n":"TypeError","m":"<normalized, ≤500 chars>","st":"<stack, ≤2000 chars>","val":null,"dev":"desktop","ts":1695463200000,"id":"a1b2c3d4"} +``` + +`t` = `err` | `vital`; `s` = `embed` | `d`; for `vital`, `n` is `LCP` | `INP` | `CLS` | +`TTFB` and `val` carries the value. No page path: docs pages send no referrer. Vitals are +sampled at 10 % per page view, decided once per page; errors are sent up to the +`monitor.ts` event ceiling. The o11y worker converts beacons with the same converter as +Faro items, clamping `ts` to the receive time ± 5 minutes. + +The dedupe hash covers the whole converted record (body with message and stack, attributes) +plus the raw `ts` and, when present, `id`: a per-beacon random value, used only in the +dedupe hash. + +## 10. Local mode + +| Piece | Local stand-in | +|---|---| +| `RUNNER_EVENTS` | ClickHouse at `http://localhost:8123`, table `runner_events` with the §4 columns plus `timestamp` and `_sample_interval` (always 1), DDL in `containers/o11y/local/clickhouse-init.sql` | +| Loki S3 | Miniflare's local S3 endpoint for R2, or MinIO from `containers/o11y/compose.yml` | +| Grafana session (K1) | `DEV_ADMIN` in `workers/o11y/.dev.vars`; a real broker login also works locally (`LOGIN_BROKER_URL` defaults to the production broker) | +| Cloudflare OTLP export | fixtures in `pipeline/fixtures/otlp/` (scrubbed sandbox-probe captures plus hand-built edge cases), replayed by `scripts/o11y-replay-fixtures.mjs` | +| Slack | `scripts/o11y-slack-capture.mjs`, a local HTTP capture server started by `pnpm dev:full` (NOT by `pnpm o11y:dev`, which doesn't start it — point `SLACK_WEBHOOK_URL` at your own instance if you need one from the standalone o11y-only command); prints and keeps the last 50 posts, `GET /_captured` to inspect | + +`deployment.environment.name = local`; the production o11y worker drops `local` data. + +**Local telemetry gate in the browser.** Faro runs on the local path only when the build +was made with `VITE_TELEMETRY_LOCAL=1` **and** the host is `localhost` or `127.0.0.1`. It +checks neither `import.meta.env.DEV` nor `navigator.webdriver`, because Playwright serves +a production `vite preview` under automation. The production gate (`resolveReporting`) +is unchanged and stays closed under automation. A production build never sets the flag, +and the post-build leak check fails if the local path survives into it. + +## 11. Sentry scope + +`SENTRY_SCOPE` / `VITE_SENTRY_SCOPE` = `full` (default): handled diagnostic reports go to +Sentry **and** the new stack. `uncaught`: they go only to the new stack; Sentry keeps +errors that escape a handler (browser `onerror`, `unhandledrejection`, +`Sentry.ErrorBoundary`; Worker fetch catch-all, DO alarms, cron, snapshot-job failures) +and the budget-alert `captureMessage`. The launch plan flips to `uncaught` after the +pipeline is seen working in production. diff --git a/runner/docs/run-and-deploy.md b/runner/docs/run-and-deploy.md index 310b835991..2110d34e05 100644 --- a/runner/docs/run-and-deploy.md +++ b/runner/docs/run-and-deploy.md @@ -29,50 +29,422 @@ bundles; the UI lazy-fetches artifacts from the selected version's bucket. ## Run locally +Three commands (`runner/scripts/dev.mjs --tier=1|2|full`, called by +`runner/package.json`'s `dev`/`dev:live`/`dev:full` scripts — one +orchestrator, not three separate scripts) cover every level: + ```bash -# Tier-1 authoring only (no containers needed): -pnpm --filter @handsontable/demo-runtime build -pnpm --filter @handsontable/demo-authoring dev # http://localhost:5173 - -# Full stack (Tier-2 containers + sharing): needs Docker. -cd workers/api && printf 'DEV_AUTH_EMAIL="dev@handsontable.com"\nPREVIEW_HOST="localhost:8787"\n' > .dev.vars -npx wrangler d1 execute handsontable-demos --local --file=migrations/0001_init.sql -y -npx wrangler d1 execute handsontable-demos --local --file=migrations/0002_buildkey_nonunique.sql -y -npx wrangler d1 execute handsontable-demos --local --file=migrations/0003_cost_ledger.sql -y -npx wrangler d1 execute handsontable-demos --local --file=migrations/0004_settings_and_analytics.sql -y -npx wrangler d1 execute handsontable-demos --local --file=migrations/0005_profiles.sql -y -npx wrangler d1 execute handsontable-demos --local --file=migrations/0006_api_tokens.sql -y -npx wrangler dev --port 8787 # builds the container images -# then run the authoring app pointing at it: -cd ../../apps/authoring -# VITE_API_BASE points at this dev server, NOT at :8787 — vite.config.ts proxies -# /api, /d and /embed to the worker, and `?mode=full` needs one origin (AGENTS.md). -printf 'VITE_DEV_USER=dev@handsontable.com\nVITE_API_BASE=http://localhost:5173\n' > .env.local -npx vite --port 5173 +pnpm dev # Tier 1: rebuilds @handsontable/demo-runtime if its dist is + # stale vs src, then runs the authoring app — http://localhost:5173 +pnpm dev:live # Tier 1 + Tier 2: + the API worker (wrangler dev, Docker + # containers, local D1 migrations) — http://localhost:8787 +pnpm dev:full # + the o11y worker, docker compose (minio/clickhouse — the + # box itself runs through wrangler dev's own container + # orchestration, same as `pnpm o11y:dev`), telemetry wiring, + # and a local Slack capture server ``` -Migrations are listed one file at a time on purpose. Do **not** substitute -`wrangler d1 migrations apply --local`: local bookkeeping starts empty, so an -apply re-runs every file, and `0003_cost_ledger.sql` ends in a bare -`ALTER TABLE demos ADD COLUMN artifacts_purged_at` with no `IF NOT EXISTS` — -which fails the second time. (Remote is a different story: CI has applied -migrations through the framework since before `0003` landed, so its bookkeeping -is populated and `deploy-runner-api.yml` applies new files automatically.) - -`.dev.vars` and `.env.local` are gitignored dev-only bypasses — never used in prod. -`PREVIEW_HOST="localhost:8787"` overrides the `wrangler.jsonc` default -(`demos.handsontable.com`, a real public wildcard that routes to the *deployed* -worker) so container preview URLs come out as `*.localhost:8787`, which browsers -treat as `127.0.0.1` (RFC 6761) and reach your local `wrangler dev`. It must be a -real host value — wrangler silently ignores empty-string `.dev.vars` overrides. -Without it, Tier-2/container sessions boot fine but the preview iframe fails with -`INVALID_TOKEN` — the token is only known to your local session, not to prod. - -This only works because `wrangler.jsonc` declares **no `routes`**: when routes -are present, `wrangler dev` simulates the first route's host on every request, -destroying the preview subdomain before `proxyToSandbox()` can route on it. -That's why the production routes live in the `deploy` script instead — don't -move them back into `wrangler.jsonc`. +`node scripts/dev.mjs --help` prints the full option list. All three need +Docker running for anything past Tier 1 — `dev:live`/`dev:full` fail fast +with a clear message (not a hung/opaque container-build error) if +`docker info` doesn't succeed. Ctrl-C (also SIGTERM, or SIGHUP from closing +the terminal) tears down every spawned `wrangler`/`vite`/capture-server +process and (for `dev:full`) runs `docker compose ... down` for the +minio/clickhouse stack this run itself started. It does **not** guarantee no +orphaned Tier-2 Sandbox containers — see "What Ctrl-C actually cleans up" +below for why, and for what it prints instead. + +**Container base-image pre-pull.** Before starting any worker, `dev:live`/ +`dev:full` read every `FROM` base image the tier's Dockerfiles declare +(`containers/live/Dockerfile` + `containers/builder/Dockerfile` for the API +worker, plus `containers/o11y/Dockerfile` for `dev:full`'s o11y worker — +looked up from each worker's own `wrangler.jsonc` `containers[].image`, not +hardcoded), runs `docker image inspect` on each, and `docker pull`s (3 +attempts, with backoff) any that's missing, printing `[images]` progress +lines. This is what `wrangler dev`'s own container build otherwise skips +silently: without it, a missing base image (e.g. a Docker Hub timeout +pulling `cloudflare/sandbox:0.12.3`) can leave `wrangler dev` running with a +broken container build, surfacing only later as an opaque Tier-2 +session-start failure. If a pull still fails after every retry, `dev.mjs` +prints which image, the Docker error's last line, and the exact +`docker pull ...` command to retry by hand, then exits before starting any +worker. Pass `--skip-image-check` to skip this check entirely (e.g. offline, +with the images already built locally). + +Every port is overridable by env var, defaulting to what's below; two +workers under `wrangler dev` always get their own, distinct `--port` and +`--inspector-port` so two dev sessions on the same machine never collide on +wrangler's inspector default (9229): + +| Var | Default | Used by | +|---|---|---| +| `AUTHORING_DEV_PORT` | 5173 | the authoring app (`vite`) — every tier | +| `API_DEV_PORT` | 8787 | the API worker (`wrangler dev`) — tier 2, full | +| `API_DEV_INSPECTOR_PORT` | 9230 | the API worker's inspector — tier 2, full | +| `O11Y_DEV_PORT` | 4200 | the o11y worker (`wrangler dev`) — tier full, `o11y:dev` | +| `O11Y_DEV_INSPECTOR_PORT` | 4201 | the o11y worker's inspector — tier full, `o11y:dev` | +| `O11Y_MINIO_PORT` | 9000 | compose's MinIO (Loki S3 stand-in) — tier full | +| `O11Y_MINIO_CONSOLE_PORT` | 9001 | compose's MinIO console — tier full | +| `O11Y_CLICKHOUSE_PORT` | 8123 | compose's ClickHouse HTTP (Analytics Engine stand-in) — tier full | +| `O11Y_CLICKHOUSE_NATIVE_PORT` | 9009 | compose's ClickHouse native protocol — tier full | +| `O11Y_SLACK_CAPTURE_PORT` | 4210 | the local Slack capture server — tier full | + +Plus `COMPOSE_PROJECT_NAME` (tier full's `docker compose` project; default is +derived PER WORKTREE — `o11y-dev-<hash of this worktree's absolute path>`, +via `scripts/dev-lib.mjs`'s `defaultComposeProjectName()` — so two worktrees +running `pnpm dev:full` never resolve to the same compose project, containers +or named volumes; set `COMPOSE_PROJECT_NAME` explicitly to still share one +on purpose) and `WRANGLER_REGISTRY_PATH` (forwarded as-is to every spawned +`wrangler dev`, for isolating one worktree's service-binding registry from +another's — several worktrees on this machine routinely run `wrangler dev` +at once, and without this, one worktree's API/o11y service binding can +resolve to another worktree's Worker instead of its own). + +If you already ran `pnpm dev:full` before this per-worktree default existed, +your MinIO/ClickHouse named volumes were under the old shared `o11y-dev` +project; docker does not rename them — your next run starts this worktree's +new derived project on fresh, empty volumes instead, and the old +`o11y-dev_minio-data`/`o11y-dev_clickhouse-data` are left orphaned. If +`workers/o11y/.wrangler/state`'s ledger has committed keys from before (the +usual case), that first run also prints the "committed key(s) ... but this +project's MinIO volume doesn't exist" divergence warning — `--fresh`'s own +advice there is the right fix (wipes the o11y worker state so the ledger +agrees with the new, empty volumes again). To reclaim the old volumes' +disk space instead of leaving them orphaned, remove them explicitly by the +old project name: `docker compose -p o11y-dev -f containers/o11y/compose.yml +down -v`. + +**`.dev.vars` bootstrap.** `workers/api/.dev.vars.example` and +`workers/o11y/.dev.vars.example` are committed, non-secret templates. +`dev.mjs`/`o11y-dev.mjs` copy either one to its gitignored `.dev.vars` +**only when `.dev.vars` doesn't already exist** — an existing file (your own +edits, real secret values) is never touched. On a *fresh* o11y bootstrap +only, a few known-inert local placeholders are filled in with real, +non-secret working values (matching `containers/o11y/compose.yml`'s own +documented local defaults): `DEV_ADMIN=dev@handsontable.com` (the local +session bypass, contract §10, `workers/o11y/src/gates/session.ts#verifySession` +— honoured only when `O11Y_ENV === "local"`, fail-closed everywhere else), +`AE_SQL_TOKEN=local-dev-token`, `LOKI_S3_ACCESS_KEY_ID`/ +`LOKI_S3_SECRET_ACCESS_KEY=minioadmin` (MinIO's own default root +credential), and `SLACK_WEBHOOK_URL` pointed at the local capture server +(`http://localhost:4210/slack` by default). + +`O11Y_EXPORT_SECRET` and `SENTRY_HOOK_SECRET` are filled +in separately, on **every** `dev.mjs --tier=full` run, not only a fresh +bootstrap: whichever of the two lines is still declared empty gets a fresh, +local-only random value (`ephemeralSecret()` — a 32-byte hex string, the same +kind of value `O11Y_SESSION_SECRET` already uses, never a pasted-in +production credential), written into `.dev.vars` and left alone on every +later run once filled. This is what makes `node scripts/o11y-replay-fixtures.mjs` +work locally without any manual setup: it reads both secrets from the +environment first, then falls back to reading them straight out of +`workers/o11y/.dev.vars`, so both the standalone command above and +`dev.mjs --tier=full --replay` succeed instead of 401ing on the OTLP/deploy/ +Sentry fixtures. Only the two key NAMES are ever printed to `dev.mjs`'s own +log — never the generated value. A `.dev.vars` you already pasted a real +value into is never touched (only an empty declared line is filled). + +**Why not just `--var`?** Wrangler's `.dev.vars` always wins over a +same-named `--var`, even when the `.dev.vars` line is empty (confirmed +against wrangler 4.108's `getVarsForDev`) — so for a key `.dev.vars.example` +already declares, `dev.mjs` bakes the working value into the bootstrapped +file instead of passing `--var` (which would be silently ignored). If a +`.dev.vars` value's port (`PREVIEW_HOST`, `SLACK_WEBHOOK_URL`) disagrees with +what this run actually resolved, and the corresponding port env var +(`API_DEV_PORT`, `O11Y_SLACK_CAPTURE_PORT`) was **not** explicitly set for +this run, `dev.mjs` **adopts** the `.dev.vars` port instead — `.dev.vars` +was always going to win for that key, so this makes the worker's own +`--port`, the vite proxy target, and the printed URLs agree with reality +instead of silently pointing at the wrong port. Only when the port env var +**was** explicitly set and disagrees does it warn instead, naming the file +to edit — an explicit choice is never silently overridden. For a genuinely +per-run secret (`O11Y_SESSION_SECRET`, the Grafana session-cookie signing +key — see the o11y auth runbook step for the deployed equivalent), the +bootstrap strips that line from a *freshly created* `.dev.vars` instead, so +the key stays undeclared and `dev.mjs`'s own ephemeral `--var` (a fresh +random value every run, never written to disk) is the only source. + +**o11y worker local-mode config (contract §10).** `dev:full`/`o11y:dev` run +the o11y worker with `O11Y_ENV` set to `local` (bootstrapped in +`.dev.vars`), which is what turns on `DEV_ADMIN` (the local session bypass — +`workers/o11y/src/gates/session.ts#verifySession`) and every +`O11Y_LOCAL_*`-prefixed override below. `dev.mjs` injects the rest as +`--var` (never `.dev.vars` — none of these are declared there, so there's no +precedence conflict to work around): `RUNNER_EVENTS_CLICKHOUSE_URL` (points +the Analytics Engine stand-in sink at compose's ClickHouse — +`O11Y_CLICKHOUSE_PORT`), `O11Y_LOCAL_MINIO_PORT`/`O11Y_LOCAL_CLICKHOUSE_PORT` +(how the box's own container, reached via Docker's `host.docker.internal`, +finds compose's MinIO/ClickHouse), and `O11Y_LOCAL_PUBLIC_ORIGIN` — the +origin `gates/session.ts#publicOrigin` builds the broker login's +`return_to` against and binds every locally-minted session token's `aud` +claim to; `dev.mjs` always sets it to `http://localhost:<O11Y_DEV_PORT>` +(Grafana is served from the o11y worker's own origin, not proxied through +the authoring app), which matters once you override `O11Y_DEV_PORT` away +from its default — `publicOrigin`'s own built-in fallback assumes the +default port. + +**Migrations.** `dev.mjs` applies every `workers/api/migrations/NNNN_*.sql` +file (currently `0001` through `0008`) one `wrangler d1 execute --local +--file=` call at a time — never `wrangler d1 migrations apply --local`, +because local bookkeeping starts empty and `0003_cost_ledger.sql` ends in a +bare `ALTER TABLE demos ADD COLUMN artifacts_purged_at` with no +`IF NOT EXISTS`, which fails the second time an apply re-runs it. Unlike the +old by-hand recipe, this is now **idempotent**: `dev.mjs` records each +applied file in `workers/api/.wrangler/state/dev-migrations-applied.json` +(gitignored, next to the local D1 state itself — wiping one wipes the +other) and only applies files not yet in that record, so a second run of +`pnpm dev:live`/`dev:full` applies nothing. (Remote is a different story: CI +has applied migrations through the framework since before `0003` landed, so +its bookkeeping is populated and `master.yml`'s `deploy-api` job applies new +files automatically.) + +**Adopting a local D1 with no record.** A local D1 migrated before this +record existed (or by hand, matching the exact bug this fixed) has none of +this bookkeeping, so every file looks "pending" — re-running an already +non-idempotent `ALTER TABLE ... ADD COLUMN` (0003, 0007) would otherwise die +with a raw `duplicate column name` failure. Before running each pending +file, `dev.mjs` takes a schema snapshot of the local D1 (table/index names +from `sqlite_master`, columns from `PRAGMA table_info`) and, if every target +the file declares (its `CREATE TABLE`/`CREATE INDEX`/`ALTER TABLE ... ADD +COLUMN` statements) already exists, **adopts** it — records it as applied +without running it, with a clear log line — instead of re-running it. A file +whose effect can't be probed generically this way (any other statement +shape) is never adopted; it always runs, relying on its own idempotency +(`IF NOT EXISTS`/`IF EXISTS`). Any migration failure — a real SQL error, or a +probe-query failure — prints ONE clean line (the file, the SQLite message, +and how to recover: the record path, or `node scripts/dev.mjs --tier=<N> +--reset-local-db` to wipe local D1 state and the record and start fresh) and +exits non-zero with nothing left running, never a raw stack trace. +`--reset-local-db` deletes `workers/api/.wrangler/state/v3/d1` and the +applied-migrations record; passing the flag is itself the confirmation (no +interactive prompt), and it prints exactly what it deleted. + +**Fixture replay (`dev:full`).** Once the o11y worker reports ready, +`dev.mjs` prints the replay command +(`node scripts/o11y-replay-fixtures.mjs --base http://localhost:<O11Y_DEV_PORT>`). +Pass `--replay` to run it automatically instead of just printing it. + +**Local rate limit is shared across browsers.** `checkRateLimit`'s key is +`cf-connecting-ip ?? "unknown"` (`gates/browser.ts`) — every local browser +resolves the same fallback, so `/telemetry/collect`/`/telemetry/lite` share +one 100-per-60s bucket across the whole machine. Several parallel test +browsers can burst past it and get `429`s. A Faro batch is then sent again +about 60 s later, but a lite beacon dropped at the gate never reaches Loki. +Expect delays and gaps, not a bug, when running multi-browser local traffic. +Production keys on each visitor's own `cf-connecting-ip`, so this is +local-only (see "Ingest rate limit" below for the per-IP budget). + +**Local Slack alerts.** `dev:full` starts +`node scripts/o11y-slack-capture.mjs --port <O11Y_SLACK_CAPTURE_PORT>` — a +tiny local HTTP server (no real Slack workspace involved) that prints and +keeps the last 50 alert posts (`GET http://localhost:4210/_captured`). The +o11y worker's local `SLACK_WEBHOOK_URL` points at it (see the bootstrap +section above), so a fired alert (ADR-0041 §F.3 — trigger the `*/10` cron by +hand with `curl "http://localhost:<O11Y_DEV_PORT>/cdn-cgi/local/scheduled?cron=*/10+*+*+*+*"`, +or replay the fixtures, which trips the new-fingerprint rule on first run) +shows up locally instead of needing a real Slack webhook. + +**Crons never fire on their own under `wrangler dev`** — neither worker's, +and this is by design, not a bug (both print "Scheduled Workers are not +automatically triggered during local development" on startup when they have +any). Trigger a specific one by hand with the pattern above for the o11y +worker (its `*/10 * * * *` backlog/alert tick), or, for the API worker's two +triggers (`workers/api/wrangler.jsonc`'s `triggers.crons`): +`curl "http://localhost:<API_DEV_PORT>/cdn-cgi/handler/scheduled?cron=17+4+*+*+*"` +for the nightly job (reconciliation, spend alerts, GC, analytics prune), and +`curl "http://localhost:<API_DEV_PORT>/cdn-cgi/handler/scheduled?cron=*/5+*+*+*+*"` +for the `*/5` pool/budget-gauge tick. The two workers' local trigger paths +differ (`/cdn-cgi/local/scheduled` vs `/cdn-cgi/handler/scheduled`) because +they pin different wrangler versions (o11y 4.136.3, API 4.108.0) whose +Miniflare internals name this differently; wrangler itself prints the +correct path and port for whichever worker you're running, so treat that +printed line as the source of truth if it ever disagrees with this doc. +(`wrangler dev --test-scheduled` + `/__scheduled` still exists as an older, +separate opt-in path, but needs the flag and does not let you pick which of +several triggers fires, so the `curl` forms above are simpler.) + +**Browsing logs.** Sign into Grafana (`http://localhost:<O11Y_DEV_PORT>/grafana/` — +`DEV_ADMIN` logs you in automatically in local mode) and open the **Logs** +dashboard for API-worker lines, authoring/embed browser errors, and a +free-text/`cf.ray`/`session.id`/demo-id search across every service in one +place — it's linked from the Runner overview and Observability self +dashboards too. (The same dashboard's `hot_surface="demo-runtime"` +panel carries no message text. `reportDemoEvent` files the preview relay as +a `preview.runtime_error` count metric (a Faro measurement, `toFacade` → +`telemetry.metric` → `pushMeasurement`) — per the ruling above (§6), +every measurement is Analytics-Engine-only, so this produces no Loki line +at all any more, not even a count-only one. The one demo-runtime Loki line +that still exists is the Tier-1 compile-failure branch, whose message is +always the constant `"Tier-1 compile failed"` — the real diagnostic detail +goes to `extra` there, and for everything else lives in Sentry, when +`MONITOR_DEMOS`/`VITE_MONITOR_DEMOS` is on.) `/admin`'s +header also has an **Open Grafana** link +(otherwise nothing in the app points at it): `href={GRAFANA_URL}` in +`Admin.tsx`, which reads `import.meta.env.VITE_GRAFANA_URL` and falls back to +`/grafana/`. On the deployed zone that fallback is what actually runs (no env +override needed) because the authoring app and the o11y worker share one +origin. Locally they don't — `apps/authoring/vite.config.ts`'s dev proxy has +no `/grafana` entry (only `/api`, `/d`, `/embed`, `/telemetry`), and a proxy +wouldn't be the right fix anyway: Grafana's own `GF_SERVER_ROOT_URL` is the +o11y worker's origin, so its login redirect would bounce off a proxied +origin. `pnpm dev:full` (`scripts/dev-lib.mjs`'s `--tier=full` plan) instead +sets `VITE_GRAFANA_URL=http://localhost:<O11Y_DEV_PORT>/grafana/` as process +env for the app's dev server, so the same link on `:<AUTHORING_PORT>/admin` +opens Grafana directly there too, with the `DEV_ADMIN` local login bypass +landing correctly. Every signed-in user is a Grafana Viewer, but Viewers now also get +**Explore** (`/grafana/explore`): pick the `Loki (browser)` or +`Loki (worker)` datasource and run a LogQL query directly against either +tenant, without needing a dashboard panel for it. Neither capability lets a +Viewer save a change back to a provisioned dashboard or datasource — those +stay read-only, and Grafana's state is disposable anyway (a fresh DB on +every wake). + +**Logs are only as fresh as the last wake.** The box drains its packed +objects into Loki once, right after it wakes, and nothing re-arms that +drain while it stays awake (ADR-0041 §B.3's out-of-order window assumes the +drain replays into an empty ingester, which only holds true at wake start). +So any ingest that arrives *while* the box is already up sits undrained and +invisible in Grafana/Explore until the *next* wake. There is no staleness +indicator on the dashboards for this; treat Logs/Explore as "as of the last +wake started", not live — a design change may follow. + +**Bot traffic is filtered locally too.** The o11y worker's bot gate drops +any request whose user agent matches `HeadlessChrome` — including local +requests, by design. This repo's own Playwright config already avoids it +(`playwright.config.ts` uses `devices["Desktop Chrome"]`, which does not +send a `HeadlessChrome` UA even when headless — confirmed live), so it's +only a risk for a bare `chromium.launch()` with no device preset. That kind +of scripted local traffic is silently dropped before it reaches Analytics +Engine/Loki, with no client-side signal that it happened; use a +`devices[...]` preset (confirmed live: `devices["Desktop Chrome"]` does not +send `HeadlessChrome` even headless) or an explicit non-`HeadlessChrome` UA +override if you need it to actually show up in local dashboards — +`channel: "chrome"` alone does **not** fix it: confirmed live, a headless +`chromium.launch({ channel: "chrome" })` still sends `HeadlessChrome/...`. + +**Local o11y data persists across a restart.** `containers/o11y/compose.yml` +gives MinIO and ClickHouse named volumes (Grafana itself stays ephemeral by +design), and `workers/o11y/.wrangler/state` (the InboxWriter ledger, dedupe +hashes, local R2 inbox objects) was already kept across a restart before +this. So a plain Ctrl-C + `pnpm dev:full` again keeps your local +logs/metrics AND the ledger that tracks them, together — `dev.mjs` prints +one line at startup either way: `o11y local data: kept (MinIO/ClickHouse +volumes + o11y worker state)`, or `o11y local data: fresh` when you passed +`--fresh`. + +**`--fresh`** wipes all of that local o11y state together in one shot: +`docker compose ... down -v` for this project's minio/clickhouse volumes, +AND `workers/o11y/.wrangler/state`. It prints exactly what it removed. +Wiping only one half (e.g. `docker volume rm` by hand) is what causes the +stack to look "broken" after a restart: a `done:` (committed) ledger key +whose MinIO data is gone is never re-drained on its own, and a fixture +replay's dedupe hashes can then block the same data from ever refilling the +now-empty store. In practice, only that second half — the dedupe-hash +hazard — applies to a wake that pushed data: see "Local clean markers never +commit" below for why a pushed key can't reach `done:` locally at all, so +neither the "committed key whose MinIO data is gone" case nor `dev.mjs`'s +own divergence warning (below) ever fires for one. If `dev.mjs` finds that +mismatch (the MinIO volume is gone but the ledger still has committed keys — +in practice, only a wake that drained nothing) it prints a warning +recommending `--fresh` — or, if you'd rather keep what R2 still has +(7-day retention), `POST /grafana/_o11y/reopen` once the worker is up. +`--fresh` never touches `workers/api`'s local D1 — that's `--reset-local-db`, a +different flag for a different store. `pnpm o11y:dev` also accepts +`--fresh`, for just its own half (workers/o11y's worker state) — it never +runs `docker compose` itself, so it can't wipe the compose volumes; see that +command's own startup log for the divergence risk if you're also running +`dev:full`'s compose stack. + +**Local clean markers never commit.** A wake resolves clean or unclean, and +it's the keys it **drained** (packing is `InboxWriter`'s job, not the +wake's) that become `done:` on a clean resolution — or get reopened and +replayed on the next wake otherwise. In production, the box writes each +wake's `state/wakes/<wakeId>/clean` marker and the worker checks for it +through the same R2 bucket (`handsontable-demos-o11y-loki`), so a clean stop +resolves those drained keys `done:` and they are never replayed again. +Locally those are two *different* stores: the container writes the marker +straight to MinIO over S3, but the o11y worker checks for it through its +`O11Y_LOKI_STATE` R2 binding, which under `wrangler dev` is Miniflare's own +separate local R2 — not MinIO. The worker never finds the marker, so a local +wake that drained any data always resolves `unclean` and reopens every key +it drained, which then replays again on the next wake, indefinitely (within +Loki's 7-day retention) — not only after a crash. This is a local-dev-only +gap with no bridge today; treat `o11y.wake outcome=unclean` as expected +locally rather than a sign something broke, and don't rely on local +`done:`/clean-shutdown testing as evidence for the production path. + +**What Ctrl-C actually cleans up.** Every `wrangler dev`/`vite`/capture-server +child is spawned in its own process group and signalled as a group on +Ctrl-C (SIGINT — also SIGTERM and SIGHUP), with an 8s grace period before +escalating to SIGKILL, and `dev:full` also runs `docker compose ... down` +(never `-v` — see above) for the minio/clickhouse stack it started, keeping +its named volumes for next time. + +What it does **not** do: stop a Tier-2 Sandbox/GrafanaBox container on your +behalf. Measured for this task: Ctrl-C does not make wrangler's own +Sandbox-container orchestration tear itself down synchronously — a +session's `workerd-handsontable-demos-api-Sandbox-*`(-proxy) container can +still be `Up` several seconds after `dev.mjs` has already exited. An +earlier version of this script tried to sweep those up itself (stop any +container that was "new since this run started" and name-matched +`handsontable-demos-(api|o11y)`), but that signal cannot tell this run's own +container apart from one a DIFFERENT worktree's concurrent `wrangler dev` +session started — several worktrees running `wrangler dev` on this same +machine at once is the normal case here (see `WRANGLER_REGISTRY_PATH` +above), not a rare race, and the sweep's window was this run's entire +session, not a narrow few seconds. Stopping the wrong worktree's container +silently kills its session. So `dev.mjs` now only **reports** containers +that look like they might be leftovers — it prints their names, a +`docker ps` filter, and the exact manual `docker stop` command — and never +runs `docker stop`/`docker rm` on anything itself. If you see that report, +confirm what a container actually is (e.g. `docker inspect` its ports) +before stopping it by hand. + +**Standalone o11y worker.** `pnpm o11y:dev` (unchanged as its own command) +starts just the o11y worker under `wrangler dev`, sharing the same +`.dev.vars` bootstrap/port-resolution code as `dev.mjs` — for working a pure +o11y bug without the API worker, Docker compose, or the Slack capture +server. It does **not** also start `containers/o11y/compose.yml` — +`wrangler dev` manages its own container instance via the same Dockerfile, +and running both would fight over the same image/ports for no benefit. + +**Debugging one piece in isolation.** The three commands above cover normal +development; to run a single worker by hand (e.g. with a debugger attached +outside the orchestrator), the underlying commands are still just +`wrangler dev` from that worker's own directory and `vite` from +`apps/authoring` — `dev.mjs --help` prints every flag and env var this +script itself understands if you want to replicate its exact invocation. + +`.dev.vars` and `.env.local` are gitignored dev-only bypasses — never used in +prod. `PREVIEW_HOST="localhost:8787"` (the API worker's bootstrapped +default) overrides the `wrangler.jsonc` default (`demos.handsontable.com`, a +real public wildcard that routes to the *deployed* worker) so container +preview URLs come out as `*.localhost:8787`, which browsers treat as +`127.0.0.1` (RFC 6761) and reach your local `wrangler dev`. It must be a real +host value — wrangler silently ignores empty-string `.dev.vars` overrides. +Without it, Tier-2/container sessions boot fine but the preview iframe fails +with `INVALID_TOKEN` — the token is only known to your local session, not to +prod. `VITE_DEV_USER`/`VITE_API_BASE` for the authoring app are injected as +process env by `dev.mjs`, never written to an `.env.local` file — nothing +committed to disk can leak the dev-login bypass into a later "real" build. + +This only works because `wrangler.jsonc` declares **no `routes`**: when +routes are present, `wrangler dev` simulates the first route's host on every +request, destroying the preview subdomain before `proxyToSandbox()` can +route on it. That's why the production routes live in the `deploy` script +instead — don't move them back into `wrangler.jsonc`. + +**Docker memory.** A Tier-2 load test that fills the real pool (up to 10 containers) plus +ClickHouse needs more than Docker Desktop's default 8 GB — ClickHouse itself was OOM-killed +at 8 GB during one such test. Raise Docker's memory limit before running one. + +**Host sleep.** Restart `dev:full` if the laptop has been asleep for a while — `workerd`'s +own alarms fire late once the host wakes, and the o11y worker can stop answering (every +route times out, no error) until the stack is restarted. + +**`at_capacity`.** Local `wrangler dev` does not enforce `containers[].max_instances`, so +`at_capacity` and its alert can only be produced and checked on the sandbox/production +account — never expect them to fire locally, however many sessions you open. ## Deploy (main Handsontable account) @@ -124,14 +496,22 @@ in [ADR-0038](adr/0038-waf-exception-for-source-code-payloads.md). On the `handsontable.com` zone → **Security → WAF → Managed rules → Cloudflare Managed Ruleset → Add exception**. Requires *Zone WAF: Edit* on the zone. -- Skip **only** rule `9c8dda9708cc4452ac76e7be7b58420b` (ruleset - `efb7b8c949ac4650a09736fc376e9aee`), not the whole ruleset. +- Skip **only** managed-ruleset rule `9c8dda9708cc4452ac76e7be7b58420b` (in the + Cloudflare Managed Ruleset, id `efb7b8c949ac4650a09736fc376e9aee`), not the + whole ruleset. - Expression: `http.host eq "demos.handsontable.com" and starts_with(http.request.uri.path, "/api/")` Scoped to `/api/*` on purpose: `/d/:id` and `/embed/:id` are the paths that serve HTML to a browser, and they stay behind the full ruleset. +"Add exception" creates its own custom rule in the zone's +`http_request_firewall_managed` entrypoint — the skip/override rule, distinct +from the managed rule it skips: `97dde65c10e049d5821a7c3643a0382d` ("demos +runner: source-code payloads trip Script Tag XSS (DEV-2631, ADR-0038)"). "WAF +exception for `/telemetry/*`" below extends **this** rule, not +`9c8dda9708cc4452ac76e7be7b58420b`. + Verify — a body the Worker itself would refuse, so `401` proves the request arrived and `403` proves the edge ate it: @@ -149,6 +529,15 @@ cd workers/api # Account -> Account Analytics -> Read. Nothing else. npx wrangler secret put CF_ANALYTICS_TOKEN +# Analytics Engine SQL API token (same token SHAPE as CF_ANALYTICS_TOKEN +# above — Account -> Account Analytics -> Read — but a SEPARATE credential: +# this one is the production read side of the nightly `example_daily` +# rollup (ADR-0042 §5, contract §2, `reconcile.ts#queryExampleEventTotals`), +# not the billing GraphQL reconciliation CF_ANALYTICS_TOKEN feeds. Also set +# on the o11y worker (step 6) for Grafana's own ClickHouse datasource — the +# two workers need their own copies, they do not share a binding. +npx wrangler secret put AE_SQL_TOKEN + # Example chat (DEV-2047) — see docs/example-chat.md: npx wrangler secret put LITELLM_API_KEY # LiteLLM virtual key; absent -> /api/chat 503s npx wrangler secret put ALGOLIA_API_KEY # Algolia search key; absent -> no doc page links @@ -160,70 +549,789 @@ ceiling. They are informational; the enforced ceiling is the Worker's own, shipp observe-only and switched on from **/admin → Guardrail settings**. Full detail in [cost-guardrails.md](cost-guardrails.md). -`wrangler dev --test-scheduled` + `curl localhost:8787/__scheduled` runs the -nightly job (reconciliation, spend alerts, GC, analytics prune) on demand. +Crons never fire on their own under `wrangler dev` — see "Crons never fire +on their own under `wrangler dev`" in the local-dev section above for the +exact `curl` commands, including the one that runs this nightly job +on demand. ## Continuous deployment -Merges to `master` deploy automatically via two path-gated GitHub Actions -workflows in `.github/workflows/`, both authenticating with the single repo -secret **`CLOUDFLARE_API_TOKEN`** (account id is read from each `wrangler.jsonc`). +Merges to `master` deploy automatically from **`.github/workflows/master.yml`**, +a single path-gated workflow with three independent deploy jobs (authoring, API, +o11y), all authenticating with the single repo secret **`CLOUDFLARE_API_TOKEN`** +(account id is read from each `wrangler.jsonc`). `.github/workflows/ci.yml` is +the separate PR-gate workflow (below); `master.yml` does not run it — a master +push has already passed it on the PR, and verifies PRODUCTION afterwards +instead (the `smoke` job). + +> History: an earlier pair of workflows, `deploy-runner-authoring.yml` and +> `deploy-runner-api.yml`, did the same two deploys separately; they were +> merged into `master.yml` so a single push range's `changes` job can gate a +> third deploy (o11y) off the same diff without a third redundant +> checkout+diff. There is no dashboard Git integration (Cloudflare Workers +> Builds) — that requires one-time setup by someone with Cloudflare access and +> silently deploys nothing until then; GitHub Actions needs only the existing +> repo secret. ### Tests (CI) -`.github/workflows/ci.yml` runs on every PR + on `master`: typecheck, unit + -catalog-smoke tests (`pnpm test` → `node --test pipeline/*.test.mjs`, validating -the wrapper output and that every committed `docs-examples` artifact is runnable), -an authoring build, and Playwright **e2e** (`pnpm e2e`) covering the picker, -cascader drill-down, framework switching, and the "See in documentation" link. +`.github/workflows/ci.yml` runs on every PR: typecheck (`pnpm typecheck` → +`pnpm -r run typecheck`, which already reaches `workers/o11y` — it is a normal +workspace package, no extra wiring needed), unit + catalog-smoke tests +(`pnpm test` → builds `@handsontable/demo-runtime` then +`node --test pipeline/*.test.mjs`, which already runs every `pipeline/o11y-*` +and `pipeline/telemetry-*` file glob-matched the same way as every other +pipeline test — validating the wrapper output, that every committed +`docs-examples` artifact is runnable, and the o11y worker's own gates/normalise/ +inbox/drain/alert logic), an authoring build, and Playwright **e2e** +(`pnpm e2e`) covering the picker, cascader drill-down, framework switching, and +the "See in documentation" link. - Live-render e2e (needs the external Sandpack bundler) is gated behind `E2E_LIVE=1`, kept off in PR CI to stay deterministic. - Run e2e against production (real live render): `E2E_BASE_URL=https://demos.handsontable.com E2E_LIVE=1 pnpm e2e`. -- The API deploy workflow also does a post-deploy smoke (`GET /api/health` on - `demos.handsontable.com` must return 200). +- `master.yml`'s `deploy-api` job does a post-deploy smoke (`GET /api/health` on + `demos.handsontable.com` must return 200); `deploy-authoring` checks the + served bundle hash; the shared `smoke` job then runs a `@smoke`-tagged e2e + subset against production once either deploy succeeds. - A separate, opt-in starter compatibility matrix (`pnpm e2e:matrix`, gated behind `E2E_STARTER_MATRIX=1`) boots every starter at every supported Handsontable major against a live instance — not part of CI, run manually. See `docs/starter-compat-matrix.md`. - -### Authoring app (frontend) → GitHub Actions - -`.github/workflows/deploy-runner-authoring.yml` runs on push to `master` -touching `runner/apps/authoring/**`, `runner/packages/**`, `runner/config/**`, -or `runner/catalog.json` (the authoring build imports `catalog.json` at compile -time, so a catalog-only change — e.g. after `pnpm import` — must redeploy the -app). It gates on the CI workflow, builds the workspace packages + the app, and -`wrangler deploy`s `handsontable-demos-authoring` (Workers Assets, no Docker). -`VITE_API_BASE` is read from committed `.env.production`. A post-deploy smoke -check verifies `demos.handsontable.com` serves the freshly built bundle. -`workflow_dispatch` allows manual runs. - -> History: this briefly moved to Cloudflare Workers Builds (dashboard Git -> integration), but that requires one-time dashboard setup by someone with -> Cloudflare access and silently deploys nothing until then — prod served a -> stale frontend. GitHub Actions needs only the existing repo secret. - -### API worker + Tier-2 image → GitHub Actions (Docker required) - -`.github/workflows/deploy-runner-api.yml` runs on push to `master` touching -`runner/workers/api/**`, `runner/containers/**`, `runner/scripts/**`, etc. On the -Docker-capable runner, `wrangler deploy` builds + pushes the `containers/live` -image (Vue baked) to the Cloudflare registry and deploys `handsontable-demos-api`. -`workflow_dispatch` allows manual runs. +- **`e2e-telemetry`**: builds the authoring app a SECOND time, with + `VITE_TELEMETRY_LOCAL=1` (contract §10), and runs + `E2E_TELEMETRY=1 pnpm e2e e2e/telemetry-faro.spec.ts e2e/example-analytics.spec.ts` + — both specs are self-contained (their own preview server, `page.route` + interception of `/telemetry/collect`, no o11y worker or API worker needed), + so they fit the deterministic PR suite. +- **`e2e-o11y-local.yml`**: `e2e/telemetry-metrics.spec.ts` + (`E2E_LIVE=1` + `E2E_TELEMETRY=1`) and `e2e/o11y-local.spec.ts` + (`E2E_O11Y_LOCAL=1`) both need infrastructure the per-PR `ci.yml` suite + should not own on every PR — a real local API worker with a live Tier-2 + container (Docker) for the first, that plus a real o11y worker, local + ClickHouse/MinIO (Docker compose) and applied D1 migrations for the second. + Rather than leaving them unhomed (`docs/TESTING.md`'s "every gate needs a + workflow home" rule), they get their own workflow, run directly on + `ubuntu-latest` (not the shared Playwright container image — Docker-in-Docker + can't reach a sibling container's `localhost`, and `wrangler dev` needs a + real Docker daemon to build the Tier-2 container image, which the bare + runner already ships, same as `master.yml`'s `deploy-api` job relies on): + `workflow_dispatch`, nightly (02:30 UTC), and on any PR touching + `workers/o11y/**`, `containers/o11y/**`, `apps/authoring/src/telemetry/**`, + or either spec file. Each job's own guard step (`scripts/ci/ + assert-e2e-ran.mjs`) fails if the gate ran zero tests or skipped any — a + mistyped env var must not read as a green, empty run. Run both specs + locally, by hand, before any change that touches the ingest path and before + every launch too — `e2e/o11y-local.spec.ts`'s own file header has the exact + setup commands. + +### Authoring app (frontend) + +`master.yml`'s `deploy-authoring` job runs when the push touches +`runner/apps/authoring/**`, `runner/packages/**`, `runner/config/**`, +`runner/catalog.json` (the authoring build imports it at compile time, so a +catalog-only change — e.g. after `pnpm import` — must redeploy the app), or +either workflow file. It downloads the `authoring-dist` artifact the shared +`build` job already produced and `wrangler deploy`s +`handsontable-demos-authoring` (Workers Assets, no Docker). `VITE_API_BASE` is +read from committed `.env.production`. A post-deploy smoke check verifies +`demos.handsontable.com` serves the freshly built bundle. +`workflow_dispatch`'s `deploy_authoring` checkbox allows a manual run. + +**Source maps (ADR §C.3).** The `build` job's authoring build step passes +`SENTRY_AUTH_TOKEN`/`SENTRY_ORG`/`SENTRY_PROJECT` (repo secret + vars) and +`VITE_SENTRY_SCOPE: full` — those three secrets present is what +`apps/authoring/vite.config.ts` reads as "this is the real production build", +which is what turns maps on (`sourcemap: "hidden"`, no `sourceMappingURL` +comment in the served JS — Workers Assets' SPA fallback, DEV-2569, answers any +path it does not recognise with `200 text/html`, which a browser trying to +follow a real map pointer would choke on) and enables the `sentryVitePlugin`'s +own upload (it now injects Sentry debug IDs but no longer deletes the maps +itself). The next step in `build` walks `apps/authoring/dist/**/*.map`, +uploads each one to R2 bucket `handsontable-demos-o11y-maps` at key +`sourcemaps/<sha>/<original asset path>.map` (matching +`workers/o11y/src/drain/symbolicate.ts`'s own `mapKeyFor`, `<sha>` = the full +`GITHUB_SHA`, same value as `VITE_SENTRY_RELEASE`/`SERVICE_VERSION`), then +deletes it from `dist/`. **This step authenticates with the dedicated, +maps-bucket-only S3 credential (`R2_MAPS_ACCESS_KEY_ID`/`R2_MAPS_SECRET_ACCESS_KEY`, +one-time setup step 2 below), through the S3 API (`aws s3 cp`), never +`CLOUDFLARE_API_TOKEN`** — that token is account-wide, and this job otherwise +never needs Cloudflare API access at all; a bucket-scoped credential is the +same principle the Loki-only token (step 3) already uses for the box. Two +leak checks run only after that deletion (a map's +`sourcesContent` embeds `localhost:8787` and `VITE_DEV_USER` literally, which +would false-fire the first check if it ran before the maps were gone): +`grep -rl "localhost:8787\|VITE_DEV_USER\|dev@handsontable.com" apps/authoring/dist` +(AGENTS.md's dev-bypass check, now automated here — see "Prod build config" +there for what each string catches) and `pnpm check:telemetry-leak` +(`scripts/check-telemetry-leak.mjs`, contract §10 — fails if the local +telemetry path's sentinels survive DCE into a production bundle). A PR build +(`ci.yml`) sets none of the three Sentry secrets, so `uploadEnabled` is false +there, no maps are ever written, and both leak-check commands still run +(harmlessly, over an unmapped `dist/`) as a standing regression net. + +### API worker + Tier-2 image (Docker required) + +`master.yml`'s `deploy-api` job runs when the push touches +`runner/workers/api/**`, `runner/containers/**`, `runner/scripts/**`, +`runner/config/**`, `runner/packages/**`, `runner/pnpm-lock.yaml`, or either +workflow file. On the Docker-capable runner, `pnpm run deploy` (never a bare +`wrangler deploy` — see the ⚠️ under "Error monitoring" below) builds + pushes +the `containers/live` image to the Cloudflare registry and deploys +`handsontable-demos-api`, then applies pending D1 migrations first. +`workflow_dispatch`'s `deploy_api` checkbox allows a manual run. Auth: repo secret **`CLOUDFLARE_API_TOKEN`** (account id is read from `wrangler.jsonc`). Create it once: 1. Cloudflare dashboard → **My Profile → API Tokens → Create Token** → start from **"Edit Cloudflare Workers"**, scoped to the **Handsontable Account**; ensure - **Workers Scripts: Edit** and the Containers/registry push permission. + **Workers Scripts: Edit** and the Containers/registry push permission. No + R2 scope needed here — the source-map upload uses its own bucket-scoped S3 + credential (`R2_MAPS_ACCESS_KEY_ID`/`R2_MAPS_SECRET_ACCESS_KEY`, one-time + setup step 2), never this token, precisely so this account-wide token never + has to be able to write to R2 at all. The export destination (step 4) is + created by hand in the dashboard, under the operator's own login — nothing + in CI calls that API, so this token needs no Observability scope either. 2. GitHub → repo **Settings → Secrets and variables → Actions → New repository secret**: name `CLOUDFLARE_API_TOKEN`, value = the token. (Never commit it.) If routes move out of `wrangler.jsonc` into the deploy command (ADR-0020), add -the corresponding `--route` flags to the API workflow's `wrangler deploy` step. +the corresponding `--route` flags to the relevant `deploy` script. + +### Observability worker (o11y + Grafana box) — GitHub Actions + +`master.yml`'s `deploy-o11y` job runs when the push touches +`runner/workers/o11y/**`, `runner/containers/o11y/**`, `runner/packages/**` +(the shared `@handsontable/demo-runtime/telemetry` module lives under +`packages/runtime/src/telemetry/`, but other files under `packages/` reach it +transitively — e.g. `scrub.ts` imports `redactPreviewHosts` from +`packages/runtime/src/monitor.ts` — so the gate is the whole directory, the +same width the authoring/API gates already use), `runner/pnpm-lock.yaml`, or +either workflow file. `pnpm run deploy` (`workers/o11y/package.json`) builds + +pushes the Grafana box container image and attaches the `/telemetry/*` and +`/grafana/*` routes via `--routes` (never in `wrangler.jsonc` — ADR-0020), plus +`--var SERVICE_VERSION:$GITHUB_SHA`. `workflow_dispatch`'s `deploy_o11y` +checkbox allows a manual run. + +**Deploy events (ADR §C.2, contract §1).** Every deploy job that actually ran +(`deploy-authoring`, `deploy-api`, `deploy-o11y`) posts one event to +`POST /telemetry/deploy` after its own `wrangler deploy`/`pnpm run deploy` +step, authenticated with a GitHub OIDC token (job permission `id-token: +write`; requested with the audience `workers/o11y/src/gates/oidc.ts` pins, +`https://demos.handsontable.com/telemetry/deploy` — the route falls back to +`x-o11y-secret` only when no bearer token is presented at all, so a malformed +one is a hard `401`, never a silent fallback): + +```bash +oidc_token=$(curl -sf -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=https://demos.handsontable.com/telemetry/deploy" \ + | jq -r '.value') +curl -sf -o /dev/null -w '%{http_code}\n' -X POST https://demos.handsontable.com/telemetry/deploy \ + -H "Authorization: Bearer $oidc_token" -H 'Content-Type: application/json' \ + --data "{\"event\":\"deploy\",\"service\":\"<worker name>\",\"sha\":\"$GITHUB_SHA\",\"cf_version_id\":\"<from the deploy step's own output>\"}" +``` + +`<worker name>` is the deploying Worker's own `wrangler.jsonc` `name` +(`handsontable-demos-authoring` / `handsontable-demos-api` / +`handsontable-demos-o11y` — this is also the string the Runner-overview +dashboard's Deploys annotation shows verbatim, `containers/o11y/grafana/dashboards/runner-overview.json`'s +`textFormat: "{{__line__}}"`). `<cf_version_id>` comes from the deploy step's +own stdout — `wrangler deploy` prints a trailing `Current Version ID: <uuid>` +line; capture it with `pnpm run deploy | tee deploy.log` (`set -o pipefail` is +on, so a piped deploy failure still fails the job) and +`grep -oE 'Current Version ID:.*' deploy.log | awk '{print $NF}'`. **This step +never fails the job on its own** (`-f` fails the curl on a non-2xx exit, but +its own exit code is deliberately not checked with `set -e` in force — a +warning line is emitted instead): a deploy that shipped correctly must not be +marked red because the *reporting* of it hiccuped, and on the very first +merge of this feature the o11y route may not be reachable yet for the +authoring/API jobs' own deploy events. + +**DAG note.** `deploy-api`'s job also `needs: deploy-o11y` (in addition to +`build`) and proceeds when that job is `success` **or skipped** (`if: always() +&& needs.deploy-o11y.result != 'failure'`) — see "First deploy, in order" +below for why. + +## One-time setup + +Everything in this section is done once, by hand, against the real Cloudflare +account, before the first `master.yml` run that touches `runner/workers/o11y/**` +can work end to end. CI does not run any of it — this is the checklist for +whoever performs the actual production launch. Every `wrangler` command below needs `CLOUDFLARE_API_TOKEN` (or an +authenticated `wrangler login`) and `-J eu`/`--jurisdiction eu` where shown — +the o11y buckets are all EU (contract §2). + +### 1. R2 buckets + lifecycle rules + +```bash +cd workers/o11y +npx wrangler r2 bucket create handsontable-demos-o11y-inbox -J eu +npx wrangler r2 bucket create handsontable-demos-o11y-loki -J eu +npx wrangler r2 bucket create handsontable-demos-o11y-maps -J eu + +# Loki bucket: browser/ 30d, worker/ 90d, index/ 90d, state/ 30d — the +# committed rule file (T01, containers/o11y/r2-lifecycle-rules.json). `set` +# REPLACES the whole rule set, so this is the only command needed for that +# bucket, run once and again whenever the file changes. +npx wrangler r2 bucket lifecycle set handsontable-demos-o11y-loki -J eu \ + --file ../../containers/o11y/r2-lifecycle-rules.json + +# Inbox and maps buckets are flat (no prefix rules) — one rule each. +npx wrangler r2 bucket lifecycle add handsontable-demos-o11y-inbox inbox-7d -J eu --expire-days 7 +npx wrangler r2 bucket lifecycle add handsontable-demos-o11y-maps maps-30d -J eu --expire-days 30 +``` + +### 2. R2 S3 credential scoped to the maps bucket only (CI source-map upload) + +Dashboard → **R2 → Manage R2 API Tokens → Create API Token**, scope +**Object Read & Write**, restricted to the single bucket +`handsontable-demos-o11y-maps` — the same "one bucket, nothing else" shape as +the Loki token in step 3 below, and for the same reason: the only thing that +ever needs to write here is `master.yml`'s own source-map upload step +(`docs/run-and-deploy.md` §"Source maps (ADR §C.3)" above), and it has +no business being able to touch the inbox or Loki buckets, let alone anything +outside this account's o11y resources. This bucket-scoped credential means +`CLOUDFLARE_API_TOKEN` never needs R2 access at all, rather than piggybacking +on that account-wide token with a blanket R2: Edit grant. + +Add the two values as **repository** secrets (GitHub → repo **Settings → +Secrets and variables → Actions → New repository secret**), not Worker +secrets — the o11y Worker itself never reads them; only the CI job's `aws s3 +cp` step does, as `AWS_ACCESS_KEY_ID`/`AWS_SECRET_ACCESS_KEY`: + +- `R2_MAPS_ACCESS_KEY_ID` +- `R2_MAPS_SECRET_ACCESS_KEY` + +### 3. Loki S3 token (`LOKI_S3_ACCESS_KEY_ID` / `LOKI_S3_SECRET_ACCESS_KEY`) + +Dashboard → **R2 → Manage R2 API Tokens → Create API Token**, scope +**Object Read & Write**, restricted to the single bucket +`handsontable-demos-o11y-loki` (contract §2: "the box writes Loki data and the +`state/` clean markers there" — nothing else needs S3 access to this bucket; +the Worker's own `O11Y_LOKI_STATE` R2 binding is a separate, narrower path that +only ever *reads* `state/wakes/<wakeId>/clean` markers, never the S3 +credential). `LOKI_S3_BUCKET` itself is **not** set in `wrangler.jsonc` — +`box.ts` falls back to `handsontable-demos-o11y-loki` when it is absent, which +is exactly the bucket this token is scoped to; only a throwaway sandbox-probe +config would ever need to override it, and if it ever is overridden the token +must be re-scoped to match, or every Loki write comes back `403`. + +```bash +cd workers/o11y +npx wrangler secret put LOKI_S3_ACCESS_KEY_ID +npx wrangler secret put LOKI_S3_SECRET_ACCESS_KEY +``` + +### 4. Export destination (`o11y-logs`) + `O11Y_EXPORT_SECRET` + +**Do this after the first deploy of both Workers, not before** — creating the +destination runs a pre-flight `POST` to +`https://demos.handsontable.com/telemetry/v1/logs`, which answers 405 until the +o11y worker's `/telemetry/*` routes are actually deployed. `workers/api/wrangler.jsonc` +ships with `observability.logs.destinations: []` for exactly this reason; a +follow-up deploy adds `"o11y-logs"` back once the destination exists (see "First +deploy, in order" below). Create it once, in the dashboard: **Workers & Pages → +Observability → Telemetry → Add destination**. + +- Destination Name: `o11y-logs` +- Destination Type: **Logs** +- OTLP Endpoint: `https://demos.handsontable.com/telemetry/v1/logs` +- Custom Headers: `x-o11y-secret: <the same value as the O11Y_EXPORT_SECRET + secret below>` — Cloudflare's own export sends **no** OIDC token, so this + header is not optional the way it is on the CI deploy-event route (ADR §B.5: + "x-o11y-secret" is the *only* gate on `/telemetry/v1/logs`). + +Generate the secret first, then paste the same value into both places: + +```bash +cd workers/o11y +npx wrangler secret put O11Y_EXPORT_SECRET # generate with `openssl rand -hex 32`, never print it +``` + +**Facts pinned by a real sandbox-probe capture, not +assumed:** the export is always `Content-Type: application/json`, +`Content-Encoding: gzip` — Cloudflare has never been observed sending +protobuf, so the ingest route does not need to handle it. `service.version` is +**absent** from the export (every worker-origin record instead falls back to +`env.SERVICE_VERSION ?? "unknown"` inside the o11y worker's own normalisation, +`workers/o11y/src/normalise/points.ts`). The ray id arrives as the attribute +`cloudflare.ray_id`, not a resource attribute. Do **not** enable a trace +destination — contract §1: "There is no trace route" (ADR §C.4). + +**One more fact, pinned by a real captured export:** a Worker's own `console.log(JSON.stringify(...))` +line (`workers/api/src/telemetry/lines.ts`'s structured request/error lines) +arrives through this export as **opaque body text** — `body.stringValue` is +the raw JSON string, and the record's own `attributes` carry only +Cloudflare's generic wrapper fields, never one of the app's own JSON keys. +The o11y worker's normaliser (`normalise/otlp.ts#tryParseJsonBodyAttrs`) +parses a JSON-object body and merges its keys into the same attribute bag a +real OTLP attribute would land in — with every §3 resource-attribute key +stripped from the parsed body and given the lowest merge priority, so a +crafted body cannot spoof a real label. Nothing to configure here; recorded +so a future change to `lines.ts`'s own JSON shape does not accidentally +reintroduce a field this parser does not expect. + +> ⚠️ The dashboard's create/patch response for a destination has, in a real +> probe session, twice echoed the export secret back in plaintext inside +> `configuration.destination_conf` (not `configuration.headers`, which IS +> redacted) — never paste that response into a shared terminal, log, or +> screenshot. If it happens, rotate `O11Y_EXPORT_SECRET` immediately (a fresh +> `wrangler secret put` + re-editing the destination's header with the new +> value) and only then continue. + +### 5. `O11Y_SESSION_SECRET` for `/grafana/*` + +**No Cloudflare Access application is needed.** The gate is the +Handsontable login broker (ADR-0007) — the +same broker `/admin` and every other internal surface already sign in +through. A callback page under `/grafana/_o11y/` reads the broker's +fragment token once, and the o11y worker mints its own signed session +cookie from it (`workers/o11y/src/gates/session.ts`, `grafana/login.ts`). +There is nothing to create in the Zero Trust dashboard. + +`/admin`'s header has an **Open Grafana** link, opened in a new tab, that +goes through this same broker login — straight to `/grafana/` on the +deployed zone, and to the o11y worker's own local origin under `pnpm +dev:full` (see "Browsing logs" above for the `VITE_GRAFANA_URL` wiring that +makes the local case work too). + +Set nothing in Cloudflare beyond this one secret: + +```bash +cd workers/o11y +npx wrangler secret put O11Y_SESSION_SECRET # generate with `openssl rand -hex 32`, never print it +``` + +`LOGIN_BROKER_URL` needs no dashboard step either — it is a public var, +already the real broker URL in `wrangler.jsonc`'s `vars` block +(`https://mcp-auth-proxy-j0tb.onrender.com`, the same value +`workers/api/wrangler.jsonc` uses). A real production probe by hand — +`curl -s -o /dev/null -w '%{http_code} %{redirect_url}\n' 'https://mcp-auth-proxy-j0tb.onrender.com/broker/login?return_to=https%3A%2F%2Fdemos.handsontable.com%2Fgrafana%2F_o11y%2Fcallback%3Fn%3Dx'` +— confirmed a `302` to Google (2026-09-24), so the callback host is allowed today; a +separate local round trip against a *stubbed* broker proves the Worker's own code, +not the real broker's live +configuration. If the production behaviour ever changes, re-run the curl command above +before assuming it still holds, and ask the broker's owners (`handsontable/hot-mcp`) to +add `demos.handsontable.com` back to `BROKER_ALLOWED_RETURN_HOSTS` if it does not. + +**The broker-wide risk this gate inherits, not fixes (DEV-3088).** The broker's +`return_to` allowlist is host-suffix-only, so it also admits anonymous Tier-2 preview +hosts (`*.demos.handsontable.com`) — anyone can harvest another team member's 1h broker +token by sending them a crafted login link. Without this gate, a stolen token could not +reach Grafana at all (`ACCESS_AUD` was `""`, so Access refused everything); this gate +widens DEV-3088's blast radius: a stolen token can now be exchanged for a Grafana +session. The session is capped at +`min(now + 12h, brokerTokenExp)` instead of a flat 12h, so the exposure a stolen token +buys is close to the token's own 1h lifetime, not 11 hours longer — but does not close +it: DEV-3088 itself remains open and is tracked separately, not by this gate. + +### 6. Every o11y worker secret (contract §2) + +```bash +cd workers/o11y +npx wrangler secret put O11Y_EXPORT_SECRET # step 4 above +npx wrangler secret put SENTRY_HOOK_SECRET # step 8 below +npx wrangler secret put AE_SQL_TOKEN # step below +npx wrangler secret put LOKI_S3_ACCESS_KEY_ID # step 3 above +npx wrangler secret put LOKI_S3_SECRET_ACCESS_KEY # step 3 above +npx wrangler secret put SLACK_WEBHOOK_URL # step 7 below +npx wrangler secret put O11Y_SESSION_SECRET # step 5 above +``` + +`AE_SQL_TOKEN` is the Analytics Engine SQL API token — same token shape as the +API worker's own `CF_ANALYTICS_TOKEN` (Account → Account Analytics → Read), +passed to the box as `GrafanaBox`'s ClickHouse datasource credential. + +`RATE_LIMITER` needs no dashboard step — a Workers rate-limiting binding's +`namespace_id` (`1001`, already in `wrangler.jsonc`) is a self-chosen scoping +id, not a Cloudflare-provisioned resource; it is created the moment +the Worker deploys with that binding present. `O11Y_STOP_GRACE_SECONDS` also +needs no setup here — it is not a Worker var at all, but a hardcoded container +`envVars` value in `box.ts` (120s in production). + +**Ingest rate limit: 100 requests / 60 s per `cf-connecting-ip`.** +`/telemetry/collect` and `/telemetry/lite` share this one bucket. The limit is +sized from what one IP actually sends: + +- **Authoring tab:** about 12 POSTs a minute at most. Faro flushes every 5 s + (`telemetry/faroConfig.ts`) and each time the tab is hidden; a flush whose + items span a URL change goes out as one POST per page URL. + `sandpack.compile_ms` is sent once per edit burst, not once per keystroke, + and not for the mount. Measured (a Tier-1 `javascript` example, typing for 70 s, + `/telemetry/collect` stubbed): 5 POSTs in the busiest 60 s, page load + included, for a comment or a string literal at 150 or 250 ms per key, and for + statements at 150 ms. +- **Embed or `/d` page view:** on average 0.4 lite beacons (four vitals on the + 10 % of views that sample them), plus at most 20 error beacons + (`MONITOR_EVENT_CEILING`) from a demo that throws. +- **One IP:** 100/60 s covers 8 authoring tabs flushing at the 12/min ceiling, + 20 at the measured 5, or about 250 embed views a minute. That is + enough headroom for an office NAT, so the limit stays at 100. + +A 429 carries `Retry-After: 60`. Faro's transport retries up to +`maxBackoffMs: 75 000` (the window plus Faro's 20 % jitter), so a 429'd batch +is sent again after the window, not dropped, unless the tab is left first +(the `pagehide` drain's one keepalive attempt lands in the same window). It keeps at most 30 batches +queued (`bufferSize`) and makes 3 attempts per batch. A lite beacon has no +retry, so a 429 drops it. + +### 7. Slack webhook + +Slack → an **Incoming Webhook** app pointed at the alert channel. Paste the +webhook URL into `SLACK_WEBHOOK_URL` (step 6). The o11y worker posts one line +per alert-rule fire/resolve transition (`slackPoster`) and no-ops +silently without this secret — alerts still land as InboxWriter state and +Grafana annotations either way, just without the Slack ping. + +### 8. Sentry internal integration (issue-alert webhook) + +Sentry → project settings → **Integrations → Internal Integrations → New +Internal Integration**. No scopes are needed (this integration only *receives* +a webhook, it never calls the Sentry API back) — just enable **Alert Rule +Action**, add a **Webhook URL** of `https://demos.handsontable.com/telemetry/hooks/sentry`, +save, and copy the generated **Client Secret** into `SENTRY_HOOK_SECRET` (step +6). Then, in the Sentry project's own alert rules, add this internal +integration as an action on whichever issue alerts should mirror into o11y. +The route verifies Sentry's `sentry-hook-signature` header, an HMAC-SHA256 of +the raw request body under this same secret (`workers/o11y/src/gates/sentry.ts`). + +### 9. GitHub OIDC trust + +Nothing to configure on GitHub's side beyond `id-token: write` on the deploying +jobs (already in `master.yml`) — GitHub's OIDC provider issues a token for its +own workflow run to any job that requests one; there is no separate "trust" +relationship to establish, unlike a cloud provider's IAM OIDC federation. The +whole trust boundary lives on the **o11y worker's** side, and is already +committed: `GITHUB_OIDC_REPOSITORY` (`handsontable/examples`) and +`GITHUB_OIDC_WORKFLOW_REF` +(`handsontable/examples/.github/workflows/master.yml@refs/heads/master`) in +`workers/o11y/wrangler.jsonc`'s `vars` block. **If `master.yml` is ever renamed +or moved, or the default branch changes, `GITHUB_OIDC_WORKFLOW_REF` must be +updated in the same PR** — `workers/o11y/src/gates/oidc.ts` checks the OIDC +token's `workflow_ref` claim against it with an exact string match, +and a stale value makes every CI deploy event fall through to the +`O11Y_EXPORT_SECRET` fallback (harmless, since that secret is also configured, +but worth knowing rather than discovering silently). + +### 10. WAF exception for `/telemetry/*` + +Extends the exception rule created by "WAF exception for `/api/*` (one-time)" +above — `97dde65c10e049d5821a7c3643a0382d` ("demos runner: source-code +payloads trip Script Tag XSS (DEV-2631, ADR-0038)"), in the zone's +`http_request_firewall_managed` entrypoint. **Not** the managed-ruleset rule it +skips (`9c8dda9708cc4452ac76e7be7b58420b`, in the Cloudflare Managed Ruleset, +id `efb7b8c949ac4650a09736fc376e9aee`) — that one is Cloudflare's, read-only. +Faro payloads (`/telemetry/collect`) and the Cloudflare OTLP export +(`/telemetry/v1/logs`) both carry arbitrary JSON bodies that can contain a +`<script` substring (a stack trace frame, a console message) exactly the way +an authored demo's HTML entry does (ADR-0038). Edit the existing exception +rule's expression to: + +``` +http.host eq "demos.handsontable.com" and (starts_with(http.request.uri.path, "/api/") or starts_with(http.request.uri.path, "/telemetry/")) +``` + +Verify the same way as the `/api/*` exception — a body the Worker itself +refuses, so `401`/`400` proves the request arrived and `403` proves the edge +still ate it: + +```bash +curl -s -o /dev/null -w '%{http_code}\n' -X POST https://demos.handsontable.com/telemetry/collect \ + -H 'Content-Type: application/json' --data '{"malformed": "<script>should 400, not 403</script>"}' +``` + +### First deploy, in order + +The o11y worker's `services` binding (`API`, entrypoint `O11yUsage`) and the +API worker's `O11Y` binding (entrypoint `O11yHeartbeat`) are **mutual** — +each names the other's Worker's named RPC entrypoint. Deploy the o11y worker +**first**: its own binding resolves lazily (a Workers service binding is not +validated against the target actually existing at *deploy* time), but +`O11yUsage.recordAwakeSeconds`/`o11ySpend` calls from `GrafanaBox` will fail +until the API worker is deployed too, and the API worker's own +`env.O11Y.heartbeat()` RPC calls (the watchdog heartbeat) fail the same way +in the other direction until the o11y worker exists. Deploying o11y first +means there is only ever one direction of "the other side isn't up yet" instead of +two. `master.yml` encodes this ordering automatically — `deploy-api` needs +`deploy-o11y` and proceeds once it is `success` or was skipped (unrelated +push) — so from the first merge onward this is handled without a manual step. +The same order applies to a throwaway sandbox probe of either worker: stand +up the probe o11y worker (or a stub) before the probe API worker if the +probe exercises the mutual binding at all. + +**The `o11y-logs` export destination follows one deploy later, for the same +reason.** Creating it also runs a pre-flight `POST` to `/telemetry/v1/logs`, +which 405s until the o11y worker's routes are live, so it cannot exist before +either Worker's first deploy: + +1. This merge ships the API worker with `observability.logs.destinations: []` + — no destination named yet. +2. Once `master.yml` has deployed both Workers, an operator creates + `o11y-logs` in the dashboard ("One-time setup" step 4 above). +3. A follow-up, one-line PR adds `"destinations": ["o11y-logs"]` back to + `workers/api/wrangler.jsonc`; its deploy turns the export on. + +Until step 3 lands, the API worker's structured log lines still reach Workers +Logs (`persist: true`), just not Loki — the worker-tenant Grafana panels +(§F.2) stay empty until then. + +## Launch plan (ADR-0041 §L) + +The order above ("First deploy, in order") is the mechanical dependency; this +section is the gate around it — what must be true before deploying at all, +what to check right after, and the two decisions ("flip the Sentry scope", +"roll back") that come later, not at deploy time. + +### Pre-conditions — confirm every one before the first real deploy + +These are carried from the tasks that found them, not newly discovered here: + +- **`O11Y_SESSION_SECRET` must be set (at least 32 bytes, `openssl rand -hex 32`) before + the first real deploy** ("One-time setup" step 5 above). Until it is, + `verifySession`/`verifyLoginCookie` both fail closed on every `/grafana/*` request + (a navigation redirects to `/grafana/_o11y/login`, everything else gets 401) — but that + login page itself answers a plain `500` rather than completing (`grafana/login.ts`'s own + pre-flight check), so Grafana is simply unreachable, not silently degraded. There is no + Access application to create; Cloudflare Access was removed from this gate entirely. +- **The Grafana session is capped at the broker token's own lifetime, not a flat 12h.** + A stolen 1h broker token + (DEV-3088, the broker-wide `return_to` suffix-allowlist risk) could otherwise be turned + into an unrevocable 12h Grafana session — 11 extra hours of exposure per stolen token, + on top of DEV-3088's existing blast radius. `gates/session.ts#computeSessionTtlSeconds` + caps every session at `min(now + 12h, brokerTokenExp)`, falling back to 1h when the + token carries no readable `exp` — this narrows, but does not eliminate, DEV-3088's + blast radius, which is still open and tracked separately. +- **Accepted risk: logout does not revoke a session.** Sessions are stateless, so + logout only clears the browser's cookie; a copied `__Host-o11y_session` stays valid + until its `exp` (at most 12h). The remedy is rotating `O11Y_SESSION_SECRET`, which + signs everyone out. +- **The "no per-panel ClickHouse `database` field" decision has not been checked + against the real Analytics Engine SQL API** — only local ClickHouse and AE's documented + SQL surface were checked. If a query returns "unknown table" in production Grafana where + it worked locally, this is the first thing to check. +- **The `aws s3 cp` step for the R2 source-map upload (`master.yml`'s `build` job) has never + run against a real R2 credential** — only simulated with `wrangler` + replaced by `echo`. Watch the first real `build` job's logs for this step specifically. +- **The deploy-event steps' `Current Version ID:` grep has never run against a real + (non-dry-run) deploy** — an empty capture degrades to an empty `cf_version_id` + rather than failing the job. `master.yml` emits `::warning::` when the parse comes + back empty, and the o11y worker's ingest path (`normalise/deploy.ts`) marks the record + with `cf_version_id_missing: true` in its body (never rejecting it — the deploy already + shipped) plus a `console.warn`, so the gap is visible in the Actions run and queryable in + Loki even if nobody is watching CI logs in real time. Spot-check the first real deploy's + `/telemetry/deploy` payload (visible as a Runner-overview annotation, or in `o11y worker + log stream`) for a real, non-empty `cf_version_id` regardless. +- **The export destination's forced-timeout behaviour is unmeasured** (the probe so far + exercised a forced-500, not a hang) — if Cloudflare's log export ever stops making + progress rather than erroring cleanly, that failure mode has no prior data point. +- **`smoke`'s job (`master.yml`) has no `/telemetry/*` or `/grafana/*` coverage** — it only + ever checked `/api/health` and the authoring bundle hash. The post-deploy smoke list below + is what stands in for that until (if ever) a task adds real `@smoke`-tagged coverage. +- **Workers Observability pricing terms must be accepted on the account before the + first real deploy.** Until they are, the account records only 1% of Workers + events instead of the full sample this task assumes, which makes the + exported-log volume check (post-deploy smoke item 6, ADR §L criterion 8) read + artificially low — a false "under half the allotment" that does not reflect + what the account records once the terms are accepted. + +### Post-deploy smoke (run once, right after the first real deploy of all three Workers) + +Everything below is either a criterion this task could only test locally, or a criterion +this task could not test at all (calendar time, real Cloudflare Analytics Engine +credentials). None of it blocks the deploy — it confirms the deploy did what the local +walkthrough already showed. + +1. **Retired — needs a new probe.** A malformed `POST /api/session` or + `PATCH /api/demos/:id` request used to reach the fetch catch-all as an uncaught 500 + (exit criterion 11's Worker leg), but every `request.json()` call site in `index.ts` + now answers 400 by design before the body is parsed. The only remaining call sites + that still throw on a malformed body sit behind `authenticateService()` (the MCP + routes) — usable only with the MCP's shared service secret, not an operator's own + bearer token, so they are not a drop-in replacement. Needs a new Worker-tenant probe + (or a change to this smoke check's own expectation) before exit criterion 11 can be + re-verified this way again. +2. **Exit criterion 15, worker tenant, against a real Cloudflare export** — a sandbox probe + already did this once; repeat once against the production o11y worker's + real Workers Logs export destination and confirm the same 7 labels + `cloudflare.ray_id` + remap + `service.version` default. +3. **Exit criterion 9 (idle tab)** — open `/grafana/*`, leave the tab genuinely idle for + 16 minutes, confirm the box stops. Every provisioned dashboard ships with + `"refresh": ""`, so a forgotten dashboard tab is idle by default and does not itself + keep the box awake; a viewer who turns refresh back on (the time picker's refresh + options are still there) accepts that their own tab now keeps the box awake, up to + the 4-hour hard cap, for as long as it stays open. +4. **Exit criterion 13 (retention)** — check the sandbox account's 1-day retention-clock + test (`t03-retention-clock-test/` prefix, `o11y-probe-t03-loki`): the two + objects should be gone and the lifecycle rule should still be listed. If more than a few + days have passed since that probe ran, this has almost certainly already resolved either + way — check R2's own lifecycle-rule application/audit log rather than re-deriving timing. +5. **`alert-eval-error` never fires** in the real Observability-self dashboard for the first + several `*/10` ticks — this is the production detector for an Analytics Engine SQL + incompatibility (a query the AE SQL API rejects that local ClickHouse happily accepts, + ADR §L's own named trap). If it fires, treat it as a real incompatibility, not noise. +6. **Tier-2 container stdout volume, confirm against the real measurement.** Locally + (a real Tier-2 session under `wrangler dev`): a Vite-family starter (`react-js`) + logs 12 lines at boot and 2 lines per 60-second keepalive poll (the Sandbox SDK's own + structured logging of its health checks, not the dev server's own output); a + slower-booting starter (`angular`) logs 22 lines at boot, same 2-per-poll rate + afterward. Projected at the ADR's own required 10× headroom (`docs/adr/ + 0041-observability-stack.md` §D "Measured"), this pushes the **exported-logs** + allotment (not the raw Workers Logs pool, which still passes) over half. The + Observability-self dashboard has no panel for exported-log volume, and + cannot get one cheaply — `o11y.ingest` (the only ingest-side AE point) aggregates one + point per *request*, with no route/tenant dimension to split "Tier-2 container stdout" + out from everything else the export destination carries. Read Cloudflare's own + **Workers → Observability → Usage** view instead (account dashboard, not Grafana): + exported log events for the current billing period, for the account. (Whether that + view can be filtered per export destination — isolating `o11y-logs` from anything + else the account exports — is not confirmed; if it cannot, this is a whole-account + figure, a safe over-estimate for this comparison since `o11y-logs` is presently the + only configured destination.) Compare that number, after a day of real production + traffic, against this projection; + if it confirms the projection, lower `head_sampling_rate` (ADR §D's own named + fallback) before the pool crosses half — do not wait for it to actually breach the + 10M/month allotment. +7. **Exit criterion 5 (symbolication CPU/memory, ADR §L).** Every + measurement so far (§L) is a Node-process proxy — no real Workers + isolate profiling access has been available, which is exactly the "stays Proposed" blocker the ADR's + own header names. Criterion 5's own wording: "an exception from a real `vite build` + resolves to `src/…` file and line using at most 500 ms CPU and 64 MB of isolate + memory; Babel-chunk frames are skipped, not parsed." This is a **Faro/browser** + exception specifically (§C.3: "a Faro exception's stack trace reaches the drain as + V8-shaped text") — item 1's (now-retired) probe was a worker-tenant line and never went + through symbolication anyway, so it never exercised this criterion. + A Playwright `page.evaluate` against the production host does not either: + `reportingEnabled`/Faro's own `productionReportingEnabled` both gate on + `navigator.webdriver !== true` (`reportingGate.ts`), which every automation harness + sets. Throw a real, marked error from a **real browser's devtools console** on the + production host instead (`throw new Error("launch-smoke isolate probe " + + Date.now())`), confirm it lands in the Observability-self dashboard's `o11y.drain` + panel (contract §5: `duration_ms`, wall time, not CPU — it is the closest number + this contract exposes to a per-object cost), then read the SAME drain alarm's own + CPU time from the Cloudflare dashboard's Workers → Observability → Logs view for the + `handsontable-demos-o11y` worker (per-invocation CPU time is a supported field + there; `wrangler tail` does not report it). Record that CPU figure against the 500 ms + budget. **Peak isolate memory has no supported per-invocation reading anywhere in + this stack** (dashboard or `wrangler tail`) — record the 64 MB half of this + criterion as "no exceeded-memory/OOM outcome observed for the probe object," not as + a measured figure, and say so explicitly in §L "Results" rather than implying a + number exists. Flip exit-criterion-5's row from "not yet measured in a real + isolate" to a dated pass/fail on that basis (13 flips separately, from its own + calendar-time check in item 4 above) — this is what unblocks Proposed → Accepted. +8. **Exit criterion 13 (retention) in production, and the AE SQL rollup, both against real + data** — item 4 above only re-checks the sandbox probe's own 1-day retention-clock test; + this is the equivalent check on the real Loki bucket and the real Grafana AE datasource. + + ```bash + # No objects in the production Loki bucket older than their prefix's own rule + # (containers/o11y/r2-lifecycle-rules.json: browser/ 30d, worker/ 90d, index/ 90d, + # state/ 30d) — confirm the rules are actually applied to the bucket first, + npx wrangler r2 bucket lifecycle list handsontable-demos-o11y-loki -J eu + # then check the oldest object per prefix in the R2 dashboard's object browser + # (wrangler has no object-listing command) and confirm it is younger than + # that prefix's day count. + ``` + + Open any Analytics-Engine-datasource panel in the **production** Grafana (`/grafana/`, + not the local stack) and confirm it returns real rows rather than an "unknown table" SQL + error — the "no per-panel ClickHouse `database` field" decision was checked against + local ClickHouse and AE's documented SQL surface only, never a live query (see the "Known + gaps" note above). Separately, confirm the nightly `example_daily` rollup (ADR-0042, + `cron:nightly:rollup`) is actually landing rows in D1: + + ```bash + cd workers/api + npx wrangler d1 execute handsontable-demos --remote \ + --command "SELECT day, framework, count(*) FROM example_daily GROUP BY day, framework ORDER BY day DESC LIMIT 20" + ``` + + `rollupExampleDaily` only writes rows for groups with at least one event, so a quiet day can + legitimately add none — expect rows, not necessarily one per calendar day, once at least one + nightly cron (04:17 UTC) has run since deploy. +9. **`at_capacity` and its alert.** Local `wrangler dev` does not enforce + `containers[].max_instances` — every session request succeeds locally regardless of + how many are already "awake", so `at_capacity`/`AT_CAPACITY_CODE` and the capacity-related + alert can only be produced and confirmed against the real sandbox/production account. + Fill the live pool (real Tier-2 sessions, one per framework, up to `max_instances`) and + confirm the next session gets a 503 `at_capacity` refusal and the alert fires. + +### Flipping `SENTRY_SCOPE` / `VITE_SENTRY_SCOPE` to `uncaught` + +All three conditions below must hold, evidenced the same way this task's own local +walkthrough evidenced them (Grafana dashboards, a fired-and-resolved alert, the volume +projection) — but against real production data, not the local stack: + +1. **Data seen end to end in Grafana** — every §F.2 journey that gets real production + traffic shows real points on its dashboard (not "No data"), for at least a full day. +2. **Alerts have fired at least once** — at least one real alert (any rule) has gone + `fired` → `resolved` in production and posted to the real Slack channel, confirming the + whole cron → rule → notify → Slack path works against real infrastructure, not just this + task's local capture server. +3. **Volume sits inside the projection** — the Observability-self dashboard's real numbers, + after at least a few days of production traffic, are under half of every allotment (§D) + the dashboard covers, matching or beating the sandbox-measured figures (ADR-0041 §L, + criteria 7–8: $0.21/month at 1× traffic, $0.33/month at 10×, both far under the $10 + ceiling). The exported-logs allotment specifically is NOT on that + dashboard (see the post-deploy smoke's item 6 above for why) — read it from + Cloudflare's own Workers → Observability → Usage view instead, same place, same + number, this time "at least a few days" rather than "one day." If real Tier-2 stdout + volume turns out to exceed the measured exported-logs allotment (ADR-0041 §L + criterion 8, the one criterion that stayed Mixed rather than passing), do not flip + the scope until the fallback (lowering `head_sampling_rate`, ADR §D's own named + escape hatch) has brought it back under half. +4. **The API-side new-fingerprint feed is confirmed live in production, not + just correctly gated.** ADR §E.1: the exact new-fingerprint alert (§F.3) is what + replaces Sentry's own "new issue" signal for a handled-error class once the scope + narrows — if this feed is dark, an API-side handled-error class that goes from zero + to happening gets NO signal at all under `uncaught` (Sentry stops seeing it, and + nothing tells the operator a new one started). This is **not** itself gated by the + `SENTRY_SCOPE` flip — the o11y worker's `*/10` new-fingerprint cron runs + unconditionally (ADR §M) — so confirm it separately, before relying on + it as the flip's replacement signal: trigger a real, once-off `reportDiagnostic` call + in production (the `npm-registry:version-exists`/`npm-registry:versions` probe paths + are the ADR's own named example) and confirm its `hot.fingerprint` appears as a new + `fp:` entry and a Slack "new fingerprint" post, not silently dropped. This needs BOTH + the real `service.name` normalised to the contract's `demos-api`, and the shared + fingerprint validator accepting a `:`-joined `context` — either one reverted or + regressed makes this feed a silent no-op again. + Also confirm, separately, that no unrelated Tier-2 SSR authored `console.log` is + producing spurious `fp:` entries of its own (ADR §M's known, accepted, + bounded residual risk — Slack noise only, not a blocker, but worth a quick look at the + Slack channel's actual traffic before trusting this as a clean signal). + +**Who flips it**: whoever owns the o11y stack operationally at launch time (the same person +or team who would triage an `alert-eval-error` or a stale-heartbeat page) — a role, not a +name fixed here; confirm with the user before the first flip. The +mechanism is not a `--var` flag pair — both names are already committed config, edited in +place and redeployed/rebuilt: `SENTRY_SCOPE` is the `"full"` var in +`workers/api/wrangler.jsonc`, flipped to `"uncaught"` and deployed with the API worker; +`VITE_SENTRY_SCOPE` is the `full` build-env value in `.github/workflows/master.yml`'s +authoring build step, flipped to `uncaught` and shipped on the next authoring deploy. Both +currently read `full`/`"full"` in those two committed files. + +### Rollback + +- **Drop the export destinations** (Workers Logs → o11y ingest) if the o11y stack itself is + the problem — this stops new data from reaching Loki/the inbox without touching the app. + **Caveat, unverified:** whether `wrangler deploy` rejects `workers/api/wrangler.jsonc`'s + `observability.logs.destinations` once it still names `o11y-logs` but that destination is + gone has not been checked — remove that entry from `wrangler.jsonc` before, or together + with, deleting the destination. +- **Revert the `observability` block** (`workers/api/wrangler.jsonc`'s + `observability.logs`/`.traces`) to pre-o11y values if the volume itself is the problem — + this is a config-only revert, no code change. +- **The `SENTRY_SCOPE`/`VITE_SENTRY_SCOPE` flip needs no revert plan of its own** (ADR + Consequences, and the "Error monitoring" section below repeats this) — it only ever + narrows what reaches Sentry, never widens it past what the production gates already allow, + so reverting it just means editing the same two committed values (`workers/api/wrangler.jsonc`'s + `SENTRY_SCOPE`, `master.yml`'s `VITE_SENTRY_SCOPE` build env) back to `full` and redeploying. +- The o11y worker and the Grafana box can be torn down entirely (delete the Worker, the + Container application, the three R2 buckets) without touching the API worker or authoring + app at all — they have no hard dependency in that direction (the API worker's own + `env.O11Y` calls degrade to the watchdog's own unreachable-heartbeat path, already + live-tested by this task, not a crash). ## Error monitoring (Sentry) @@ -243,6 +1351,27 @@ and from a deploy-time var respectively), with the production strings unchanged. Anything keying on them Sentry-side — alert rules, saved searches, dashboards — keeps working. +**`SENTRY_SCOPE` / `VITE_SENTRY_SCOPE` — full vs. uncaught (contract §11, ADR +§E.3).** Sentry now sits beside the o11y stack described in "Observability" +above, not in front of it, and this switch controls how much overlap the two +keep. `full` (the default — both vars are set to `full` in every committed config +today: `SENTRY_SCOPE` in `workers/api/wrangler.jsonc`, `VITE_SENTRY_SCOPE` in +`.github/workflows/master.yml`'s authoring build env; `resolveSentryScope`/the API +worker's own fallback also treat an absent var as `full`, for a build/deploy that +predates either being set) sends every explicit diagnostic +report — `reportError`, the Tier-1/Tier-2 branches of `reportRuntimeError`, the +Worker's own handled-error lines — to **both** Sentry and the o11y facade, so +today's dashboards, saved searches and on-call habits keep working unchanged. +`uncaught` narrows Sentry to only what escapes a handler outright (browser +`window.onerror`/`unhandledrejection`/`Sentry.ErrorBoundary`; Worker +fetch-catch-all/DO alarms/cron/snapshot-job failures) plus the budget-alert +`captureMessage` — everything else goes to o11y alone. **Do not flip this +switch as part of CI or any one-time setup step above** — see "Launch plan +(ADR-0041 §L)" above for the exact three conditions and who flips it; it +needs no revert plan of its own either way, since it only ever +narrows Sentry, never widens it beyond what `reportingGate.ts`/`sentry-gate.ts` +already allow. + **The DSN is committed, in two places**, because a DSN is a write-only ingest endpoint that ships inside the JS bundle by construction — hiding it buys nothing, and keeping it out of `wrangler secret` means changing it needs no Cloudflare @@ -285,8 +1414,8 @@ two small import-free modules, `apps/authoring/src/reportingGate.ts` and > `--var SENTRY_ENVIRONMENT:api-production` flag lives in that script, and without > it the deployed Worker comes up with error reporting silently off — nothing > errors, events just stop arriving. That is the fail-closed direction working as -> intended, but it is invisible, so it is worth knowing. `.github/workflows/deploy-runner-api.yml` -> calls `pnpm run deploy`, so CI is fine. Verify a change to the flag with +> intended, but it is invisible, so it is worth knowing. `master.yml`'s `deploy-api` +> job calls `pnpm run deploy`, so CI is fine. Verify a change to the flag with > `pnpm exec wrangler deploy --dry-run --outdir /tmp/x --var SENTRY_ENVIRONMENT:api-production` > and check the binding table; a flag-supplied var prints as `(hidden)`, which is a > display convention, not a broken binding. @@ -460,6 +1589,12 @@ can ignore the reporter and `postMessage` crafted payloads straight at the app, `kind` against a closed set (it becomes a Sentry tag). Treat the in-page cap as advisory and the relay's as the limit. +On Tier 1 the in-page cap is per run, not per page load: the preview document is +re-evaluated in place on every compile, so `SandpackRuntime` posts a reset +(`MONITOR_RESET`) into it before each dispatched run. Otherwise the prefix runs of one +typed line spend all 20 slots before the finished line throws. The warning ceiling and +the relay's own budget stay per page load. + **A per-environment rate limit on `demo-runtime` in the Sentry UI is still the only brake that works without a build** — keep one configured for as long as this is on. @@ -561,10 +1696,11 @@ Slugs, not the numeric ids in the DSN (`o95873` / `4511806997135360`). **Create all three together, or none.** `vite.config.ts` enables the plugin only when all three are present, because a token with no org/project has no upload -target. All three are attached to the authoring build step of -`deploy-runner-authoring.yml` only; the `test` job reuses `ci.yml` and gets none of -them, so PR builds neither emit source maps nor create a release. With upload off, -`build.sourcemap` is off too, so no `.map` files are produced or published. +target. All three are attached to `master.yml`'s `build` job's authoring build +step only; `ci.yml`'s own `authoring` job builds the same app again for PR e2e +and gets none of them, so PR builds neither emit source maps nor create a +release. With upload off, `build.sourcemap` is off too, so no `.map` files are +produced or published. Note that a *failed* upload (bad token, wrong slug) does **not** fail the build — `sentry-cli` logs the error and vite still exits 0. The symptom is unreadable diff --git a/runner/e2e/admin-panel.spec.ts b/runner/e2e/admin-panel.spec.ts index a8b9b74d41..e049e3d856 100644 --- a/runner/e2e/admin-panel.spec.ts +++ b/runner/e2e/admin-panel.spec.ts @@ -82,6 +82,40 @@ function liveSessionsSection(page: Page) { }); } +test("signed in, the header links to Grafana through the o11y worker's own broker login", async ({ page }) => { + // Nothing else in the app links to Grafana (ADR-0043's dashboards are + // otherwise unreachable except by typing the URL), so this pins the one + // discoverable entry point: a same-origin anchor, opened in a new tab — + // `target="_blank"` so a signed-out visit there doesn't lose the + // operator's place in `/admin`, and the o11y worker's own session check + // (not this app) decides whether they land in Grafana or its broker login. + // + // The href comes from `Admin.tsx`'s `import.meta.env.VITE_GRAFANA_URL || + // "/grafana/"`, baked in at BUILD time (this suite runs `vite build` once, + // in playwright.config.ts's webServer, then serves the static `dist/`). + // `E2E_EXPECT_GRAFANA_URL` lets a caller that rebuilt with `VITE_GRAFANA_URL` + // set (see pipeline/dev-script.test.mjs's revert-check, which rebuilds and + // reruns this exact test against a fake local value) assert against that + // value instead of the default; CI never sets it, so CI is checking the + // production fallback, the same value a real production build produces. + const expectedHref = process.env.E2E_EXPECT_GRAFANA_URL ?? "/grafana/"; + + await stubShell(page); + await signIn(page); + + await page.route("**/api/admin/usage**", (route) => + route.fulfill({ json: usageReport(sessionsPage([], { offset: 0, limit: 25, total: 0, awakeCount: 0, meterCount: 0 })) }), + ); + + await page.goto("/admin"); + await expect(page.getByRole("heading", { name: /usage & cost/ })).toBeVisible(); + + const grafanaLink = page.getByRole("link", { name: /Open Grafana/ }); + await expect(grafanaLink).toBeVisible(); + await expect(grafanaLink).toHaveAttribute("href", expectedHref); + await expect(grafanaLink).toHaveAttribute("target", "_blank"); +}); + test("signed out, /admin is a login wall — the panel never renders and no admin data is fetched", async ({ page }) => { // The contrast case. AdminGate answers a null user by calling `login()` // (App.tsx), a top-level `location.href` to the broker. stubShell's abort of diff --git a/runner/e2e/authed-actions.spec.ts b/runner/e2e/authed-actions.spec.ts index 67637b423f..34717c8f5b 100644 --- a/runner/e2e/authed-actions.spec.ts +++ b/runner/e2e/authed-actions.spec.ts @@ -388,6 +388,9 @@ test("the workspace save sends the code only, never the metadata", async ({ page expect(patches[0]).toHaveProperty("files"); expect(patches[0]).not.toHaveProperty("title"); expect(patches[0]).not.toHaveProperty("description"); + // This build's telemetry gate is closed, and the API counts `example.saved` + // only for a save that carries this field. + expect(patches[0]).not.toHaveProperty("exampleHtMajor"); }); test("the Edit info dialog cannot be dismissed mid-save", async ({ page }) => { @@ -592,6 +595,51 @@ test("a Save the server refuses on ownership says so, and is not a session promp expect(await storedToken(page)).toBe("e2e-token"); }); +test("a Save whose code does not build shows the build error and keeps the edit unsaved", async ({ page }) => { + await stubShell(page); + await stubSavedDemo(page); + await signIn(page); + await stubProfile(page); + const detail = 'error during build: src/App.tsx:1:10: ERROR: Unexpected ";"'; + await failWrite(page, "PATCH", 422, { error: `build failed: ${detail}`, code: "build_failed", detail }); + + await page.goto(`/edit/${DEMO_ID}`); + await expect(accountAvatar(page)).toBeVisible(); + await editor(page).click(); + await page.keyboard.type("const X = ;"); + await saveButton(page).click(); + + const dialog = page.getByRole("dialog", { name: "Couldn't save" }); + await expect(dialog).toContainText(detail); + await expect(dialog).toContainText("nothing was saved"); + await expect(page.getByText(/build_failed|build failed:/)).toHaveCount(0); + await expect(saveButton(page)).toHaveText("Save •"); + await dialog.getByRole("button", { name: "OK" }).click(); + await expect(dialog).toHaveCount(0); +}); + +for (const action of ["Fork", "Share"] as const) { + test(`a ${action} whose code does not build shows the build error`, async ({ page }) => { + await stubShell(page); + await signIn(page); + await stubProfile(page); + const detail = 'error during build: src/App.tsx:1:10: ERROR: Unexpected ";"'; + await page.route("**/api/demos", (route) => + route.fulfill({ status: 422, json: { error: `build failed: ${detail}`, code: "build_failed", detail } }), + ); + + await page.goto("/?example=react"); + await expect(accountAvatar(page)).toBeVisible(); + await (action === "Fork" ? forkButton(page) : shareIcon(page)).click(); + + const dialogTitle = action === "Fork" ? "Couldn't fork" : "Couldn't share"; + const dialog = page.getByRole("dialog", { name: dialogTitle }); + await expect(dialog).toContainText(detail); + await expect(page.getByText(/build_failed|build failed:/)).toHaveCount(0); + await expect(forkButton(page)).toBeEnabled(); + }); +} + // The bug the uniform early-return would have introduced. `onFork` has no // `finally` — the success path navigates away, and clearing `forking` first // would flash the button back to idle mid-navigation — so the busy state is diff --git a/runner/e2e/example-analytics.spec.ts b/runner/e2e/example-analytics.spec.ts new file mode 100644 index 0000000000..9ce2d02364 --- /dev/null +++ b/runner/e2e/example-analytics.spec.ts @@ -0,0 +1,519 @@ +import { test, expect, type Page, type Route } from "@playwright/test"; +import { spawn, execSync, type ChildProcess } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { flushFaro, signIn, stubShell } from "./helpers.js"; + +// ADR-0042 example analytics: `example.open` at the App.tsx example-resolve +// path. +// +// Gated: needs a dist built with VITE_TELEMETRY_LOCAL=1 (contract §10), same +// pattern as `e2e/telemetry-faro.spec.ts`. No o11y worker needed: +// `/telemetry/collect` is captured with `page.route`, exactly as that spec +// does. +// +// E2E_TELEMETRY=1 pnpm e2e e2e/example-analytics.spec.ts +// +// (No separate manual build step — unlike telemetry-faro.spec.ts, this spec +// builds itself, into its own `--outDir` below.) +// +// This spec must stay gated (never run ungated under plain `pnpm e2e`) and +// must build into a private `dist-example-analytics`, never rebuild +// `apps/authoring/dist` itself in place: `ci.yml`'s `e2e` job has no +// `@handsontable/demo-runtime` dist and no telemetry build step, so +// `beforeAll` would throw there; `e2e-telemetry` runs this file and +// `telemetry-faro.spec.ts` together (`fullyParallel`), and this spec +// emptying/rebuilding the shared `dist/` mid-run would race +// telemetry-faro's own `:4711` preview into flaky 404s; and a local +// `pnpm e2e` would leave a telemetry-flagged, `.env.local`-poisoned `dist/` +// behind for every later spec (and a manual deploy) to pick up. +// +// The docs catalog itself is fully stubbed (same recipe as +// `e2e/docs-picker.spec.ts#installDocsCatalog`) rather than the real +// `apps/authoring/public/docs-examples/` content — never hard-code a docs +// bucket minor in a spec: the bucket is whatever `stubShell`'s fake +// `/api/versions` resolves to (currently 18.0.0 → bucket "18.0"), read back +// from the same fixture this file defines, never a literal written into an +// assertion. + +test.skip( + process.env.E2E_TELEMETRY !== "1", + "set E2E_TELEMETRY=1 first — this spec builds its own VITE_TELEMETRY_LOCAL=1 dist", +); + +// Never 4173/4711/4712 (other specs' shared/own preview ports) and never +// another worktree's concurrent port block. +const PORT = 5701; +// 127.0.0.1, not "localhost": see telemetry-faro.spec.ts's BASE_URL comment — +// in CI's Playwright container, this spec's own `fetch("http://localhost:…")` +// readiness poll failed with "TypeError: fetch failed" while `vite preview` +// itself bound the default host with no startup error logged. Not +// reproduced locally; pinning bind and poll to the same literal address +// removes a whole axis of ambiguity regardless of the exact mechanism. +const BASE_URL = `http://127.0.0.1:${PORT}`; +const AUTHORING_DIR = fileURLToPath(new URL("../apps/authoring", import.meta.url)); +const OUT_DIR = "dist-example-analytics"; + +/** Node's `fetch` (undici) reports a connection failure as a bare + * `TypeError: fetch failed` — the useful part (ECONNREFUSED vs ENETUNREACH, + * which address/port it actually tried) is one level down in `.cause`, + * which a plain `String(err)` drops. Mirrors telemetry-faro.spec.ts's + * helper — this is exactly the CI failure that motivated it: the logged + * line said nothing more than "fetch failed". */ +function formatFetchFailure(err: unknown): string { + if (err instanceof Error) { + const cause = (err as { cause?: unknown }).cause; + if (cause && typeof cause === "object") { + const c = cause as { code?: string; address?: string; port?: number; message?: string }; + return `${err.message} (cause: ${c.code ?? "?"} ${c.address ?? ""}${c.port ? `:${c.port}` : ""} ${c.message ?? ""})`.trim(); + } + return err.message; + } + return String(err); +} + +function waitForServer(url: string, timeoutMs: number): Promise<void> { + const deadline = Date.now() + timeoutMs; + return new Promise((resolve, reject) => { + const attempt = () => { + fetch(url) + .then(() => resolve()) + .catch((err) => { + if (Date.now() > deadline) reject(err); + else setTimeout(attempt, 200); + }); + }; + attempt(); + }); +} + +/** Wires a spawned preview server's stdout+stderr (and a hard spawn failure, + * which fires on `"error"` rather than either stream) into one string, so a + * `waitForServer` timeout's thrown error explains what happened instead of + * just restating the timeout. Stdout matters as much as stderr: vite's own + * `➜ Local: http://…` bind line goes to stdout. Mirrors + * telemetry-faro.spec.ts's helper. */ +function captureServerDiagnostics(child: ChildProcess): { get(): string } { + let text = ""; + child.stdout?.on("data", (chunk) => { text += String(chunk); }); + child.stderr?.on("data", (chunk) => { text += String(chunk); }); + child.on("error", (err) => { text += `\n[spawn error] ${String(err)}`; }); + child.on("exit", (code, signal) => { + if (code !== 0 && code !== null) text += `\n[exited early with code ${code}]`; + else if (signal) text += `\n[killed by signal ${signal}]`; + }); + return { get: () => text.trim() }; +} + +interface FaroEvent { + name?: string; + attributes?: Record<string, string>; +} +interface FaroBody { + events?: FaroEvent[]; +} + +/** One deterministic docs example — same shape/route recipe as + * `docs-picker.spec.ts#installDocsCatalog`, trimmed to what this spec + * needs: a `guide`/`breadcrumb`/`framework` distinct from `docsPath` (so a + * test that reads `ref`/`area` off `docsPath` or the URL by mistake fails), + * and one entry so both the deep-link and the picker paths resolve it. */ +const DOCS_ENTRY = { + breadcrumb: ["Columns", "Adding and removing columns"], + guide: "guides/columns/column-adding/column-adding.md", + guideTitle: "Adding and removing columns", + docsPath: "guides/columns/column-adding/react/example2.tsx", + exampleId: "example2", + exampleTitle: "Add and remove columns from the context menu", + docPermalink: "/column-adding", +}; + +async function installDocsCatalog(page: Page) { + await stubShell(page); // versions stub -> latest 18.0.0 -> bucket "18.0" + await page.route("**/docs-examples/*/*.json", async (route: Route) => { + const url = new URL(route.request().url()); + const [, bucket, file] = url.pathname.match(/\/docs-examples\/([^/]+)\/(.+)\.json$/) ?? []; + const path = decodeURIComponent(file ?? "").replace(/__/g, "/"); + if (path !== DOCS_ENTRY.docsPath) { + await route.fulfill({ status: 404, body: "not found" }); + return; + } + await route.fulfill({ + json: { + framework: "react", + displayName: "React", + tier: 1, + engine: "sandpack", + sandpackTemplate: "react-ts", + sandpackEnvironment: "parcel", + container: null, + htWrappers: ["@handsontable/react-wrapper"], + entry: "/src/App.tsx", + htmlEntry: "/index.html", + devCommand: null, + buildCommand: "vite build", + outputDir: "dist", + outputGlob: null, + staticExport: false, + spaMode: false, + port: null, + installCommand: "pnpm install", + htCoreRange: "18.0.0", + fileCount: 3, + assets: [], + skipped: [], + docsPath: path, + breadcrumb: [...DOCS_ENTRY.breadcrumb], + guide: DOCS_ENTRY.guide, + guideTitle: DOCS_ENTRY.guideTitle, + exampleId: DOCS_ENTRY.exampleId, + lang: "tsx", + files: { + "/src/App.tsx": `export const fixture = "${bucket}:${path}";\n`, + "/index.html": `<div id="root"></div>`, + "/package.json": JSON.stringify( + { dependencies: { handsontable: "18.0.0", "@handsontable/react-wrapper": "18.0.0" } }, + null, + 2, + ), + }, + }, + }); + }); + await page.route("**/docs-examples/*/manifest.json", async (route: Route) => { + const bucket = new URL(route.request().url()).pathname.split("/").at(-2) ?? ""; + await route.fulfill({ + json: { + bucket, + docsBranch: "e2e-fixture", + generatedFrom: "e2e fixture", + hotVersion: "18.0.0", + count: 1, + examples: [ + { + bucket, + docsPath: DOCS_ENTRY.docsPath, + file: DOCS_ENTRY.docsPath.replace(/\//g, "__") + ".json", + breadcrumb: [...DOCS_ENTRY.breadcrumb], + guide: DOCS_ENTRY.guide, + guideTitle: DOCS_ENTRY.guideTitle, + exampleId: DOCS_ENTRY.exampleId, + exampleTitle: DOCS_ENTRY.exampleTitle, + docPermalink: DOCS_ENTRY.docPermalink, + framework: "react", + displayName: "React", + }, + ], + }, + }); + }); +} + +/** Captures every Faro event `page.route` sees on `/telemetry/collect` (the + * real Faro `TransportBody` shape: one shared `meta` plus separate typed + * arrays). Fulfils with a bare 202 so the SDK's own retry logic never + * kicks in. */ +function captureTelemetryEvents(page: Page): FaroEvent[] { + const events: FaroEvent[] = []; + void page.route(`${BASE_URL}/telemetry/collect`, async (route: Route) => { + let body: FaroBody = {}; + try { + body = (route.request().postDataJSON() as FaroBody) ?? {}; + } catch { + body = {}; + } + events.push(...(body.events ?? [])); + await route.fulfill({ status: 202, body: "" }); + }); + return events; +} + +/** Opens the example-pill cascader (named for whatever is currently open) and + * picks a starter template by its catalog `displayName` — the real static + * `starter-examples/18/*.json` artifacts, no route stub needed (same + * reasoning as blank-starter.spec.ts). The category click is defensive: the + * popover already reveals the open starter's own category on open. */ +async function pickStarter(page: Page, currentLabel: RegExp, starterDisplayName: string) { + await page.getByRole("button", { name: currentLabel }).first().click(); + await page.getByText("Starter templates", { exact: true }).click(); + await page.getByRole("treeitem", { name: starterDisplayName, exact: true }).click(); +} + +test.describe.configure({ mode: "serial" }); + +let server: ChildProcess; + +test.beforeAll(async () => { + // Fail loudly rather than silently reusing whatever already answers on + // this port — same trap AGENTS.md warns about for the shared :4173. + const already = await fetch(BASE_URL).then(() => true).catch(() => false); + if (already) { + throw new Error( + `something is already answering on :${PORT} — kill it first (lsof -ti :${PORT} | xargs kill)`, + ); + } + // Built into its own --outDir, never the shared apps/authoring/dist + // other specs' :4173 webServer (playwright.config.ts) or a manual deploy + // could pick up. + execSync("node_modules/.bin/vite build --outDir " + OUT_DIR, { + cwd: AUTHORING_DIR, + env: { ...process.env, VITE_TELEMETRY_LOCAL: "1" }, + stdio: "pipe", + }); + server = spawn( + "node_modules/.bin/vite", + ["preview", "--outDir", OUT_DIR, "--host", "127.0.0.1", "--port", String(PORT), "--strictPort"], + { cwd: AUTHORING_DIR, stdio: "pipe" }, + ); + const diagnostics = captureServerDiagnostics(server); + try { + await waitForServer(BASE_URL, 30_000); + } catch (err) { + throw new Error(`preview server on :${PORT} never came up: fetch: ${formatFetchFailure(err)} | server output: ${diagnostics.get() || "(none)"}`); + } +}); + +test.afterAll(() => { + server?.kill(); +}); + +test("a docs example opened by ?docs= fires one example.open with entry=deep-link, taxonomy read from the loaded entry", async ({ + page, +}) => { + await installDocsCatalog(page); + const events = captureTelemetryEvents(page); + + await page.goto(`${BASE_URL}/?docs=${encodeURIComponent(DOCS_ENTRY.docsPath)}&v=18.0.0`); + + // The editor showing the fetched artifact's own marker is the same "loaded + // the RIGHT example" oracle docs-picker.spec.ts uses — a regression that + // fires example.open without actually resolving the example would still + // pass a bare "one event exists" check. + await expect(page.locator('[data-pane-active="true"] .cm-content')).toContainText( + `18.0:${DOCS_ENTRY.docsPath}`, + ); + + await expect.poll(() => events.filter((e) => e.name === "example.open").length).toBe(1); + const [open] = events.filter((e) => e.name === "example.open"); + const attrs = open.attributes ?? {}; + expect(attrs["hot.reason"]).toBe("deep-link"); + expect(attrs["hot.metric_kind"]).toBe("docs"); + // ref is the GUIDE, never docsPath or the URL (the task's own Traps). + expect(attrs["hot.ref"]).toBe(DOCS_ENTRY.guide); + expect(attrs["hot.ref"]).not.toBe(DOCS_ENTRY.docsPath); + expect(attrs["hot.area"]).toBe(DOCS_ENTRY.breadcrumb[0]); + expect(attrs["hot.framework"]).toBe("react"); + expect(attrs["hot.ht_major"]).toBe("18"); + expect(attrs["hot.bucket"]).toBe("18.0"); + + // "none on re-render": reloading the same URL is a fresh page load (and so + // a fresh, legitimate second open) but simply letting the page sit idle + // must not add a second one. + await page.waitForTimeout(1000); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(events.filter((e) => e.name === "example.open")).toHaveLength(1); +}); + +test("a docs example opened from the picker fires one example.open with entry=picker", async ({ page }) => { + await installDocsCatalog(page); + const events = captureTelemetryEvents(page); + + // Deliberately no `?example=` in the URL: the bare-`/` default landing on + // the react starter is not itself an `example.open` (the top-bar + // "React" trigger below still renders identically either way), so the + // only `example.open` this test ever sees is the picker's own. + await page.goto(`${BASE_URL}/`); + await page.getByRole("button", { name: /React/ }).first().click(); + const search = page.getByPlaceholder("Search examples…"); + await expect(search).toBeFocused(); + await search.fill("context menu"); + await page.getByRole("listbox", { name: "Search results" }).getByRole("option").first().click(); + + await expect(page).toHaveURL(/docs=guides%2Fcolumns%2Fcolumn-adding%2Freact%2Fexample2\.tsx/); + + await expect.poll(() => events.filter((e) => e.name === "example.open").length).toBe(1); + const [open] = events.filter((e) => e.name === "example.open"); + expect(open.attributes?.["hot.reason"]).toBe("picker"); + expect(open.attributes?.["hot.metric_kind"]).toBe("docs"); + expect(open.attributes?.["hot.ref"]).toBe(DOCS_ENTRY.guide); +}); + +// A post-fork landing on the new demo's own `/edit/:id` must classify as +// `entry=fork`, not `deep-link` — `onFork` does a full `location.href` +// reload (no in-memory flag survives it), so the signal is a one-shot URL +// marker (`?fork=1`) the saved-demo load effect reads and strips +// (`exampleAnalytics.ts#consumeForkMarker`). Stubs the saved-demo +// source/meta pair the same way `e2e/description-markdown.spec.ts#stubSavedDemo` +// does — this spec never actually calls `onFork` itself (that needs a real +// POST /api/demos), it simulates landing on the fork's own destination URL, +// which is the half `consumeForkMarker` is responsible for. +const FORKED_DEMO_ID = "e2efork01"; +const FORKED_DEMO_FILES = { + "/src/App.tsx": "export default function App() { return null; }\n", + "/index.html": '<div id="root"></div>', + "/package.json": JSON.stringify({ dependencies: { handsontable: "18.0.0" } }, null, 2), +}; + +async function stubForkedDemo(page: Page) { + await page.route("**/api/demos/**", (route: Route) => + route.fulfill({ + json: new URL(route.request().url()).pathname.endsWith("/source") + ? { framework: "react", files: FORKED_DEMO_FILES } + : { title: "Fork of React", description: null, ht_version: "18.0.0", created_at: "2026-09-23T00:00:00.000Z" }, + }), + ); +} + +test("landing on /edit/:id?fork=1 (onFork's own destination) fires example.open with entry=fork, and strips the marker", async ({ + page, +}) => { + await stubShell(page); + await signIn(page); + await stubForkedDemo(page); + const events = captureTelemetryEvents(page); + + await page.goto(`${BASE_URL}/edit/${FORKED_DEMO_ID}?fork=1`); + + await expect.poll(() => events.filter((e) => e.name === "example.open").length).toBe(1); + const [open] = events.filter((e) => e.name === "example.open"); + expect(open.attributes?.["hot.reason"]).toBe("fork"); + expect(open.attributes?.["hot.metric_kind"]).toBe("saved"); + expect(open.attributes?.["hot.ref"]).toBe(FORKED_DEMO_ID); + + // One-shot: the marker is gone from the URL once it has been read, so a + // manual reload of this same address is a plain deep-link, not a fork. + await expect(page).toHaveURL(new RegExp(`/edit/${FORKED_DEMO_ID}$`)); +}); + +// `example.saved` is the API worker's (contract §5): the browser hands it the +// open example's `ht_major` in the Save request, and counts the Save itself +// only when the response lacks the API's `exampleSaved` marker. +const SAVED_DEMO_ID = "e2esave01"; + +/** Opens a stubbed saved demo, edits and Saves it against a PATCH answering + * `saveResponse`; returns the PATCH bodies, the Faro events, and the open. */ +async function saveOnce(page: Page, saveResponse: Record<string, unknown>) { + await stubShell(page); + await signIn(page); + const patches: Array<Record<string, unknown>> = []; + await page.route("**/api/demos/**", async (route: Route) => { + if (route.request().method() === "PATCH") { + patches.push(JSON.parse(route.request().postData() ?? "{}")); + return route.fulfill({ json: saveResponse }); + } + return route.fulfill({ + json: new URL(route.request().url()).pathname.endsWith("/source") + ? { framework: "react", files: FORKED_DEMO_FILES } + : { title: "Saved demo", description: null, ht_version: "18.0.0", created_at: "2026-09-23T00:00:00.000Z" }, + }); + }); + const events = captureTelemetryEvents(page); + + await page.goto(`${BASE_URL}/edit/${SAVED_DEMO_ID}`); + await expect.poll(() => events.filter((e) => e.name === "example.open").length).toBe(1); + const [open] = events.filter((e) => e.name === "example.open"); + expect(open.attributes?.["hot.metric_kind"]).toBe("saved"); + + const saveButton = page.getByRole("button", { name: /^Save/ }); + await page.locator('[data-pane-active="true"] .cm-content').click(); + await page.keyboard.type("// edit"); + await expect(saveButton).toHaveText("Save •"); + await saveButton.click(); + await expect(saveButton).toHaveText("Save"); + + // Faro sends in push order, so once a probe pushed after the Save has + // arrived, an `example.saved` pushed by the Save would have arrived too. + const probeRef = `save-probe-${Date.now()}`; + await page.evaluate((ref) => { + const hook = (window as unknown as { + __t06Telemetry?: { event: (name: string, attrs: Record<string, string>) => void }; + }).__t06Telemetry; + hook?.event("example.downloaded", { kind: "saved", ref }); + }, probeRef); + await expect.poll(() => events.some((e) => e.attributes?.["hot.ref"] === probeRef)).toBe(true); + return { patches, events, open }; +} + +test("a Save sends the open example's ht_major to the API and, with the API's marker, emits no example.saved itself", async ({ page }) => { + const { patches, events, open } = await saveOnce(page, { ok: true, htVersion: "18.0.0", exampleSaved: true }); + expect(patches).toHaveLength(1); + expect(patches[0]).toHaveProperty("files"); + expect(patches[0].exampleHtMajor).toBe(open.attributes?.["hot.ht_major"]); + expect(patches[0].exampleHtMajor).toBe("18"); + expect(events.filter((e) => e.name === "example.saved")).toHaveLength(0); +}); + +test("a Save answered without the exampleSaved marker (an API that does not count it) emits one browser example.saved", async ({ page }) => { + const { events } = await saveOnce(page, { ok: true, htVersion: "18.0.0" }); + const saved = events.filter((e) => e.name === "example.saved"); + expect(saved).toHaveLength(1); + expect(saved[0].attributes?.["hot.metric_kind"]).toBe("saved"); + expect(saved[0].attributes?.["hot.ref"]).toBe(SAVED_DEMO_ID); + expect(saved[0].attributes?.["hot.ht_major"]).toBe("18"); +}); + +// A starter picked from the picker is a real, later `example.open` — never +// the page's own silent first load, and never a no-op for the `example.*` +// actions that follow it (ADR-0042). Both entry files below open on boot +// (same oracle blank-starter.spec.ts uses), so each pick is checked against +// the actually loaded artifact, not just an event existing. +const REACT_LABEL = /React \(Vite, TS\)/; +const BLANK_LABEL = "Blank (JavaScript)"; + +test("a bare / visit, then a picker pick of a different starter, fires one example.open with entry=picker whose taxonomy a later example.engaged reuses", async ({ page }) => { + await stubShell(page); + const events = captureTelemetryEvents(page); + const opens = () => events.filter((e) => e.name === "example.open"); + const engaged = () => events.filter((e) => e.name === "example.engaged"); + + // No ?example= in the URL — the default react starter's own landing must + // stay silent (ADR-0042: a bare `/` visit is not a pick). `flushFaro` + // forces the queue out before this check, the same way the deep-link + // test above proves "none on re-render". + await page.goto(`${BASE_URL}/`); + await expect(page.locator('[data-pane-active="true"] .cm-content')).toContainText("createRoot"); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(opens()).toHaveLength(0); + + await pickStarter(page, REACT_LABEL, BLANK_LABEL); + await expect(page.locator('[data-pane-active="true"] .cm-content')).toContainText( + "new Handsontable(container", + ); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(opens()).toHaveLength(1); + const [open] = opens(); + expect(open.attributes?.["hot.reason"]).toBe("picker"); + expect(open.attributes?.["hot.metric_kind"]).toBe("starter"); + expect(open.attributes?.["hot.ref"]).toBe("blank"); + expect(open.attributes?.["hot.framework"]).toBe("blank"); + expect(open.attributes?.["hot.ht_major"]).toBe("18"); + + // A first edit's example.engaged reads the same taxonomy back off this ref. + await page.locator('[data-pane-active="true"] .cm-content').click(); + await page.keyboard.type("// edit"); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(engaged()).toHaveLength(1); + expect(engaged()[0].attributes?.["hot.metric_kind"]).toBe("starter"); + expect(engaged()[0].attributes?.["hot.ref"]).toBe("blank"); +}); + +test("a ?example= deep-link visit, then a picker pick of a different starter, fires entry=picker (not deep-link) for the pick", async ({ page }) => { + await stubShell(page); + const events = captureTelemetryEvents(page); + const opens = () => events.filter((e) => e.name === "example.open"); + + await page.goto(`${BASE_URL}/?example=blank&v=18.0.0`); + await expect(page.locator('[data-pane-active="true"] .cm-content')).toContainText( + "new Handsontable(container", + ); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(opens()).toHaveLength(1); + expect(opens()[0].attributes?.["hot.reason"]).toBe("deep-link"); + + await pickStarter(page, new RegExp(BLANK_LABEL.replace(/[()]/g, "\\$&")), "React (Vite, TS)"); + await expect(page.locator('[data-pane-active="true"] .cm-content')).toContainText("createRoot"); + await flushFaro(page, (ref) => events.some((e) => e.attributes?.["hot.ref"] === ref)); + expect(opens()).toHaveLength(2); + const pick = opens()[1]; + expect(pick.attributes?.["hot.reason"]).toBe("picker"); + expect(pick.attributes?.["hot.ref"]).toBe("react"); +}); diff --git a/runner/e2e/helpers.ts b/runner/e2e/helpers.ts index 49adad557e..5e2d0b05ac 100644 --- a/runner/e2e/helpers.ts +++ b/runner/e2e/helpers.ts @@ -177,3 +177,24 @@ export function currentDocsRelease(): { bucket: string; version: string } { return { bucket, version: hotVersion }; } + +/** + * Sends what Faro holds (a `sendTimeout` of seconds) through its own page-hide + * flush, then waits for a probe event pushed behind it, so an absence check + * that follows sees everything the page had queued. Needs a + * `VITE_TELEMETRY_LOCAL=1` build (`window.__t06Telemetry`); `seen(ref)` reads + * the spec's capture for an event with that `hot.ref`. + */ +export async function flushFaro(page: Page, seen: (ref: string) => boolean): Promise<void> { + const ref = "flush-probe-" + Math.random().toString(36).slice(2, 10); + await page.evaluate((probeRef) => { + const hook = (window as unknown as { + __t06Telemetry?: { event: (name: string, attrs: Record<string, string>) => void }; + }).__t06Telemetry; + hook?.event("example.downloaded", { surface: "authoring", kind: "docs", ref: probeRef }); + Object.defineProperty(document, "visibilityState", { configurable: true, get: () => "hidden" }); + document.dispatchEvent(new Event("visibilitychange")); + delete (document as { visibilityState?: unknown }).visibilityState; + }, ref); + await expect.poll(() => seen(ref), { message: "the flush probe reached /telemetry/collect" }).toBe(true); +} diff --git a/runner/e2e/o11y-local.spec.ts b/runner/e2e/o11y-local.spec.ts new file mode 100644 index 0000000000..4d47aa5985 --- /dev/null +++ b/runner/e2e/o11y-local.spec.ts @@ -0,0 +1,311 @@ +import { test, expect } from "@playwright/test"; +import { spawn, execSync, type ChildProcess } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { previewReady, expectGridRendered } from "./helpers.js"; + +// The automatable slice of the local end-to-end walkthrough (ADR-0041 §L, +// docs/run-and-deploy.md's local-dev section). Unlike `telemetry-faro.spec.ts` +// and `telemetry-metrics.spec.ts`, this spec does not mock `/telemetry/*` +// with `page.route` — the whole point is that a real browser's telemetry +// reaches the real o11y worker and is queryable back out of the real +// (local) Analytics Engine sink, proving the wiring end to end, not just +// each unit test in isolation. +// +// Preconditions (own port block, 5200-5299 — not the shared +// playwright.config.ts webServer on :4173, and not any other task's ports): +// +// 1. Local ClickHouse + MinIO (the compose stand-in for Analytics Engine +// + the Loki bucket), on this spec's own ports: +// COMPOSE_PROJECT_NAME=o11y-t11-e2e \ +// O11Y_MINIO_PORT=5210 O11Y_MINIO_CONSOLE_PORT=5211 \ +// O11Y_CLICKHOUSE_PORT=5212 O11Y_CLICKHOUSE_NATIVE_PORT=5213 \ +// AE_SQL_TOKEN=local-dev-token \ +// docker compose -f containers/o11y/compose.yml up -d --wait minio clickhouse +// (`minio` creates its own `loki` bucket via MINIO_DEFAULT_BUCKETS +// before its healthcheck goes green; `--wait` blocks on that.) +// 2. `workers/o11y/.dev.vars` (copy from `.dev.vars.example`, fill in +// O11Y_ENV=local, DEV_ADMIN, O11Y_EXPORT_SECRET, SENTRY_HOOK_SECRET, +// AE_SQL_TOKEN=local-dev-token, O11Y_LOCAL_MINIO_PORT=5210, +// O11Y_LOCAL_CLICKHOUSE_PORT=5212, RUNNER_EVENTS_CLICKHOUSE_URL= +// http://localhost:5212). +// 3. `workers/api/.dev.vars` (copy from the main checkout, add +// RUNNER_EVENTS_CLICKHOUSE_URL=http://localhost:5212 and +// AE_SQL_TOKEN=local-dev-token; PREVIEW_HOST must stay a non-production +// host, e.g. localhost:8799, or the API worker starts reporting to the +// real Sentry project). +// 4. `apps/authoring/vite.config.ts`'s dev proxy is not used here (this +// spec builds a real dist and serves it standalone) — instead the +// build points `VITE_API_BASE` straight at this spec's API worker, +// the same shape `telemetry-metrics.spec.ts` uses. +// +// This spec starts (and tears down) its own o11y + API `wrangler dev` +// processes and its own authoring dist/preview server; it does not start +// docker or apply D1 migrations — those are one-time local setup. +// +// cd workers/o11y && WRANGLER_REGISTRY_PATH=<your worktree>/.wrangler-registry \ +// npx wrangler dev --port 5220 --inspector-port 5221 +// (or `O11Y_DEV_PORT=5220 O11Y_DEV_INSPECTOR_PORT=5221 node ../../scripts/o11y-dev.mjs`) +// cd workers/api && npx wrangler d1 migrations apply handsontable-demos --local +// E2E_O11Y_LOCAL=1 pnpm e2e e2e/o11y-local.spec.ts +// +// CI home: `.github/workflows/e2e-o11y-local.yml` — on workflow_dispatch, +// nightly, and PRs touching the o11y ingest path, never the per-PR `ci.yml` +// gate: the prerequisite stack is Docker + two `wrangler dev` processes + D1 +// migrations. Also run it locally before every o11y change that touches +// the ingest path, and before a launch. + +// Env-overridable (per-worktree port block). Defaults unchanged. +const AUTHORING_PORT = Number(process.env.E2E_O11Y_LOCAL_AUTHORING_PORT ?? 5290); +const API_PORT = Number(process.env.E2E_O11Y_LOCAL_API_PORT ?? 5280); +const API_INSPECTOR_PORT = Number(process.env.E2E_O11Y_LOCAL_API_INSPECTOR_PORT ?? 5281); +const O11Y_PORT = Number(process.env.O11Y_DEV_PORT ?? 5220); +const CLICKHOUSE_PORT = Number(process.env.O11Y_LOCAL_CLICKHOUSE_PORT ?? 5212); +const BASE_URL = `http://localhost:${AUTHORING_PORT}`; +const API_BASE_URL = `http://localhost:${API_PORT}`; +const O11Y_BASE_URL = `http://localhost:${O11Y_PORT}`; +const AUTHORING_DIR = fileURLToPath(new URL("../apps/authoring", import.meta.url)); +const API_DIR = fileURLToPath(new URL("../workers/api", import.meta.url)); +const OUT_DIR = "dist-o11y-local"; + +function waitForServer(url: string, timeoutMs: number): Promise<void> { + const deadline = Date.now() + timeoutMs; + return new Promise((resolve, reject) => { + const attempt = () => { + fetch(url) + .then(() => resolve()) + .catch((err) => { + if (Date.now() > deadline) reject(err); + else setTimeout(attempt, 200); + }); + }; + attempt(); + }); +} + +/** Queries the local ClickHouse stand-in for Analytics Engine directly — + * the same table/columns `clickhouseSink` writes (contract §10), the same + * sink `workers/o11y/src/normalise/points.ts#aeSink` and + * `workers/api/src/telemetry/resource.ts`'s sink selection write to in + * local mode. Not through Grafana/the AE-query allowlist — this spec + * checks the point landed, not that a dashboard query can read it back, + * which `pipeline/o11y-dashboards.test.mjs` covers deterministically. */ +async function chRows(sql: string): Promise<Record<string, string>[]> { + const res = await fetch(`http://localhost:${CLICKHOUSE_PORT}/?default_format=JSONEachRow`, { + method: "POST", + headers: { "X-ClickHouse-User": "default", "X-ClickHouse-Key": "local-dev-token" }, + body: sql, + }); + if (!res.ok) throw new Error(`ClickHouse query failed (${res.status}): ${await res.text()}`); + const text = await res.text(); + return text + .trim() + .split("\n") + .filter(Boolean) + .map((line) => JSON.parse(line)); +} + +// `timestamp` is `DateTime64(3)`; comparing it against a bare integer +// literal silently uses ClickHouse's seconds interpretation (measured: a +// bare-integer `WHERE timestamp >= <epoch ms>` returned zero rows for data +// inserted seconds earlier). `toUnixTimestamp64Milli` makes the comparison +// explicit and matches the millisecond encoding +// `sink.ts#clickhouseTimestamp` writes. +async function pointsFor(metric: string, sinceMs: number): Promise<Record<string, string>[]> { + return chRows( + `SELECT * FROM runner_events WHERE index1 = '${metric}' AND toUnixTimestamp64Milli(timestamp) >= ${sinceMs} FORMAT JSONEachRow`, + ); +} + +test.describe("o11y local end-to-end", () => { + test.skip(process.env.E2E_O11Y_LOCAL !== "1", "set E2E_O11Y_LOCAL=1 — needs Docker + two wrangler dev processes, see the file header"); + // One shared authoring+API pair for the whole file, the same reasoning + // telemetry-metrics.spec.ts gives (a second Playwright worker would race + // on these `--strictPort` processes). + test.describe.configure({ mode: "serial" }); + test.use({ baseURL: BASE_URL }); + test.setTimeout(180_000); + + let authoringServer: ChildProcess; + let apiServer: ChildProcess; + + test.beforeAll(async () => { + test.setTimeout(360_000); + + const o11yUp = await fetch(O11Y_BASE_URL).then(() => true).catch(() => false); + if (!o11yUp) { + throw new Error( + `no o11y worker answering on ${O11Y_BASE_URL} — start it first (see this file's header): ` + + `cd workers/o11y && WRANGLER_REGISTRY_PATH=<worktree>/.wrangler-registry npx wrangler dev --port ${O11Y_PORT}`, + ); + } + const chUp = await fetch(`http://localhost:${CLICKHOUSE_PORT}/ping`).then((r) => r.ok).catch(() => false); + if (!chUp) { + throw new Error( + `no local ClickHouse answering on :${CLICKHOUSE_PORT} — start compose first (see this file's header)`, + ); + } + + const apiAlready = await fetch(API_BASE_URL).then(() => true).catch(() => false); + if (apiAlready) { + throw new Error(`something is already answering on :${API_PORT} — kill it first (lsof -ti :${API_PORT} | xargs kill)`); + } + + // The API worker keeps containers enabled: the /d test's Fork is + // "fork -> build -> R2", and the build runs in a container. Starting it + // with `--enable-containers=false` made Fork + // fail with no redirect to /edit/. The cold-runner image step is covered + // by the longer hook timeout above instead. + apiServer = spawn( + "node_modules/.bin/wrangler", + ["dev", "--port", String(API_PORT), "--inspector-port", String(API_INSPECTOR_PORT)], + { cwd: API_DIR, stdio: "pipe" }, + ); + let apiStderr = ""; + apiServer.stderr?.on("data", (chunk) => { apiStderr += String(chunk); }); + apiServer.stdout?.on("data", (chunk) => { apiStderr += String(chunk); }); + try { + await waitForServer(API_BASE_URL, 60_000); + } catch (err) { + throw new Error(`local API worker on :${API_PORT} never came up: ${apiStderr || String(err)}`); + } + + // A real build, not a mocked one: VITE_API_BASE/VITE_TELEMETRY_LOCAL/ + // VITE_DEV_USER are build-time `import.meta.env` reads. VITE_DEV_USER + // opens the local auth bypass (`auth.ts`) — this dist must never be + // treated as shippable (same rule `check:telemetry-leak` enforces for + // VITE_TELEMETRY_LOCAL); it exists only for this gated, local-only spec. + execSync(`node_modules/.bin/vite build --outDir ${OUT_DIR}`, { + cwd: AUTHORING_DIR, + env: { + ...process.env, + VITE_TELEMETRY_LOCAL: "1", + VITE_DEV_USER: "t11-e2e@handsontable.com", + VITE_API_BASE: API_BASE_URL, + VITE_SENTRY_SCOPE: "full", + }, + stdio: "pipe", + }); + const apiBaseCompiled = execSync(`grep -rl "localhost:${API_PORT}" ${OUT_DIR}/assets || true`, { + cwd: AUTHORING_DIR, + }).toString().trim(); + if (!apiBaseCompiled) { + throw new Error( + `VITE_API_BASE did not compile to http://localhost:${API_PORT} anywhere in ${OUT_DIR}/assets — ` + + `this run would otherwise talk to the wrong (or production) API`, + ); + } + + // Faro's transport always posts to SAME-ORIGIN `/telemetry/collect`, and + // the `/d`/`/embed` lite beacon posts to same-origin + // `/telemetry/lite` too (contract §6/§9) — there is no + // `VITE_TELEMETRY_BASE` the way there is a `VITE_API_BASE`, so this + // preview server must proxy BOTH `/telemetry/*` (to the real o11y + // worker) and `/d`/`/embed` (to the real API worker) itself, the same + // "one origin stands in for production's one zone" trick + // `vite.config.ts`'s own header comment explains. `vite preview` reads + // `server.proxy` from `vite.config.ts` the same way `vite dev` does + // (measured live, not assumed) — `O11Y_DEV_PORT` and + // `API_DEV_PORT` (not the proxy's otherwise-hardcoded `:8787` target) + // are that config's + // own env vars, read at this process's startup. Navigating straight at + // the API worker's own origin for `/d/:id` (skipping this proxy) was + // tried first and found live to be the wrong move: the lite beacon it + // serves would then post to the API worker's OWN origin, which has no + // `/telemetry/*` route at all — a 404 the production zone's shared + // routing never lets happen. + authoringServer = spawn( + "node_modules/.bin/vite", + ["preview", "--outDir", OUT_DIR, "--port", String(AUTHORING_PORT), "--strictPort"], + { + cwd: AUTHORING_DIR, + stdio: "pipe", + env: { ...process.env, O11Y_DEV_PORT: String(O11Y_PORT), API_DEV_PORT: String(API_PORT) }, + }, + ); + let authoringStderr = ""; + authoringServer.stderr?.on("data", (chunk) => { authoringStderr += String(chunk); }); + try { + await waitForServer(BASE_URL, 30_000); + } catch (err) { + throw new Error(`authoring preview on :${AUTHORING_PORT} never came up: ${authoringStderr || String(err)}`); + } + }); + + test.afterAll(() => { + authoringServer?.kill(); + apiServer?.kill(); + }); + + test("Tier-1 preview.ready_ms reaches the real o11y worker and lands in Analytics Engine", async ({ page }) => { + const sinceMs = Date.now() - 5_000; + const collectRequests: number[] = []; + page.on("response", (res) => { + if (res.url().includes("/telemetry/collect")) collectRequests.push(res.status()); + }); + + await page.goto("/?example=react"); + await previewReady(page, "sandpack"); + await expectGridRendered(page); + + // The request genuinely left the browser and reached the real o11y + // worker (not intercepted) — a 204 from anywhere else (a stale proxy + // target, a 404 from the authoring server itself) would fail this. + await expect.poll(() => collectRequests, { timeout: 15_000 }).toContain(204); + + // The real ingest pipeline wrote a real Analytics Engine point — not a + // captured request body, an actual row read back out of the sink the + // worker writes. `emitPoint`/normalise are fire-and-forget past the 204, + // so this polls rather than asserting once. + const rows = await expect + .poll(async () => pointsFor("preview.ready_ms", sinceMs), { timeout: 15_000, message: "preview.ready_ms never landed in ClickHouse" }) + .not.toHaveLength(0) + .then(() => pointsFor("preview.ready_ms", sinceMs)); + const point = rows[rows.length - 1]; + expect(point.blob5, "hot.tier").toBe("1"); + expect(point.blob6, "hot.framework").toBe("react"); + expect(point.blob8, "outcome").toBe("ready"); + expect(point.blob1, "service.name").toBe("demos-authoring"); + }); + + test("a forced /d error reaches the real o11y worker as a demos-embed lite beacon point", async ({ page, request }) => { + // Fork + save a real demo through the real API worker (exampleAnalytics + // path, D1-backed) — the same real Fork/Save flow the walkthrough exercises + // by hand, not a stubbed fixture, so `/d/:id` really serves through + // `serveDemoAsset`'s injection seam. + await page.goto("/?example=react&v=18.1.1"); + await previewReady(page, "sandpack"); + await page.getByRole("button", { name: "Fork" }).click(); + await expect(page).toHaveURL(/\/edit\//); + const demoId = new URL(page.url()).pathname.split("/edit/")[1]; + await page.getByRole("button", { name: "Save" }).click(); + + const sinceMs = Date.now() - 2_000; + await page.goto(`/d/${demoId}`); + // A real uncaught error on the /d page, caught by the injected lite + // reporter's own `window.addEventListener("error", ...)` (packages/runtime/ + // src/monitor.ts) — 100% sampled, unlike web vitals. + await page.evaluate(() => { + setTimeout(() => { + throw new Error("t11-o11y-local forced /d error"); + }, 0); + }); + + const rows = await expect + .poll( + async () => + chRows( + `SELECT * FROM runner_events WHERE index1 = 'error.uncaught' AND blob1 = 'demos-embed' AND toUnixTimestamp64Milli(timestamp) >= ${sinceMs} FORMAT JSONEachRow`, + ), + { timeout: 15_000, message: "no demos-embed error.uncaught point landed in ClickHouse" }, + ) + .not.toHaveLength(0) + .then(() => + chRows( + `SELECT * FROM runner_events WHERE index1 = 'error.uncaught' AND blob1 = 'demos-embed' AND toUnixTimestamp64Milli(timestamp) >= ${sinceMs} FORMAT JSONEachRow`, + ), + ); + expect(rows[rows.length - 1].blob4, "hot.surface").toBe("d"); + + await request.delete(`${API_BASE_URL}/api/demos/${demoId}`).catch(() => {}); + }); +}); diff --git a/runner/e2e/share-not-found.spec.ts b/runner/e2e/share-not-found.spec.ts new file mode 100644 index 0000000000..644f1c09af --- /dev/null +++ b/runner/e2e/share-not-found.spec.ts @@ -0,0 +1,60 @@ +import { test, expect, type Page } from "@playwright/test"; +import { stubShell } from "./helpers"; + +// `/share/<bad>` must not show "entry file /index.html not found in example +// files" for a demo id that doesn't resolve — confusing, and not even about +// the actual problem (there's no demo, not a bad entry file). On a failed +// `GET /api/demos/:id/source`, the loader (App.tsx) must not flip +// `sourceLoaded` true and un-gate rendering EditorShell without also calling +// `loadWorkspace`: `files`/`entry` would stay at their placeholder value +// (`entry.entry` set, `files: {}`, from `toPlaceholderEntry`), and +// EditorShell's preview-mount effect would then run against that +// still-empty, inconsistent placeholder and throw its own "entry file … +// not found" error, overwriting the friendly message with nothing ever +// having rendered it. +// +// The fix short-circuits the render on a 404/410 (same pattern `docsNotFound` +// uses for the docs-example loader) before EditorShell — and hence its +// mount effect — is ever reached. +// +// Deterministic: `stubShell` (e2e/helpers.ts) covers /api/versions, the two +// Sandpack hosts and the login redirect; `/api/demos/:id/source` and +// `/api/demos/:id` are stubbed here on top, so this needs no real API worker, +// running against the local `vite preview`. + +const DEMO_ID = "e2ezzznope1"; + +async function stubMissingDemo(page: Page, { metaStatus }: { metaStatus: 404 | 410 }) { + await stubShell(page); + await page.route("**/api/demos/**", (route) => { + const isSource = new URL(route.request().url()).pathname.endsWith("/source"); + if (isSource) { + // getDemoSource collapses "never existed" and "revoked" to the same 404 + // (index.ts's own comment: "revoked demos return 404") — this route + // never answers 410, matching production. + return route.fulfill({ status: 404, json: { error: "not found" } }); + } + return route.fulfill({ + status: metaStatus, + json: metaStatus === 410 ? { error: "revoked" } : { error: "not found" }, + }); + }); +} + +test("a share link to a demo id that never existed says so, not 'entry file not found'", async ({ page }) => { + await stubMissingDemo(page, { metaStatus: 404 }); + await page.goto(`/share/${DEMO_ID}`); + + await expect(page.getByText("Demo not found")).toBeVisible(); + await expect(page.getByText(/entry file/i)).toHaveCount(0); + await expect(page.getByText(/not found in example files/i)).toHaveCount(0); +}); + +test("a share link to a revoked demo says it was removed, not 'entry file not found'", async ({ page }) => { + await stubMissingDemo(page, { metaStatus: 410 }); + await page.goto(`/share/${DEMO_ID}`); + + await expect(page.getByText("This demo was removed")).toBeVisible(); + await expect(page.getByText(/entry file/i)).toHaveCount(0); + await expect(page.getByText(/not found in example files/i)).toHaveCount(0); +}); diff --git a/runner/e2e/telemetry-faro.spec.ts b/runner/e2e/telemetry-faro.spec.ts new file mode 100644 index 0000000000..a721f012b8 --- /dev/null +++ b/runner/e2e/telemetry-faro.spec.ts @@ -0,0 +1,1039 @@ +import { test, expect, type Route, type Page } from "@playwright/test"; +import { spawn, execSync, type ChildProcess } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { activeEditor, flushFaro, previewReady, stubShell } from "./helpers.js"; +import { fingerprint } from "../packages/runtime/src/telemetry/fingerprint.js"; + +// Faro in the authoring app. Gated: needs a dist built with +// VITE_TELEMETRY_LOCAL=1 (contract §10), served on its own port (never +// 4173, which another worktree's `vite preview` may already hold). No o11y +// worker needed: `/telemetry/collect` is captured with `page.route`. +// +// VITE_TELEMETRY_LOCAL=1 pnpm --filter @handsontable/demo-authoring build +// E2E_TELEMETRY=1 pnpm e2e e2e/telemetry-faro.spec.ts +// +// This spec manages its own preview server, not the shared +// playwright.config.ts webServer, so it never depends on whatever `dist` +// another spec run left behind. `E2E_TELEMETRY_PORT` / +// `E2E_TELEMETRY_UNCAUGHT_PORT` override the ports. +const PORT = Number(process.env.E2E_TELEMETRY_PORT ?? 4711); +// 127.0.0.1, not "localhost": in CI (the Playwright container job) this +// spec's own `fetch("http://localhost:…")` readiness poll failed outright +// ("TypeError: fetch failed", see `formatFetchFailure` below) while `vite +// preview` itself bound the default, unqualified host with no startup +// error — a likely dual-stack "localhost" resolution mismatch between the +// bind and the poller. Pinning both sides to the same literal IPv4 address +// removes that whole axis of ambiguity. +const BASE_URL = `http://127.0.0.1:${PORT}`; +const AUTHORING_DIR = fileURLToPath(new URL("../apps/authoring", import.meta.url)); + +/** Node's `fetch` (undici) reports a connection failure as a bare + * `TypeError: fetch failed` — the useful part (ECONNREFUSED vs ENETUNREACH, + * which address/port it actually tried) is one level down in `.cause`, + * which a plain `String(err)` drops. This is exactly the CI failure that + * motivated this helper: the logged line said nothing more than "fetch + * failed". */ +function formatFetchFailure(err: unknown): string { + if (err instanceof Error) { + const cause = (err as { cause?: unknown }).cause; + if (cause && typeof cause === "object") { + const c = cause as { code?: string; address?: string; port?: number; message?: string }; + return `${err.message} (cause: ${c.code ?? "?"} ${c.address ?? ""}${c.port ? `:${c.port}` : ""} ${c.message ?? ""})`.trim(); + } + return err.message; + } + return String(err); +} + +function waitForServer(url: string, timeoutMs: number): Promise<void> { + const deadline = Date.now() + timeoutMs; + return new Promise((resolve, reject) => { + const attempt = () => { + fetch(url) + .then(() => resolve()) + .catch((err) => { + if (Date.now() > deadline) reject(err); + else setTimeout(attempt, 200); + }); + }; + attempt(); + }); +} + +/** Wires a spawned preview server's stdout+stderr (and a hard spawn failure, + * which fires on `"error"` rather than either stream — e.g. the vite binary + * missing — plus an early exit) into one string, so a `waitForServer` + * timeout's thrown error explains what happened instead of just restating + * the timeout. Stdout matters as much as stderr here: vite's own + * `➜ Local: http://…` bind line — which address it actually listened on — + * goes to stdout, and that line is exactly what would have told the CI + * failure apart from a genuine startup error. */ +function captureServerDiagnostics(child: ChildProcess): { get(): string } { + let text = ""; + child.stdout?.on("data", (chunk) => { text += String(chunk); }); + child.stderr?.on("data", (chunk) => { text += String(chunk); }); + child.on("error", (err) => { text += `\n[spawn error] ${String(err)}`; }); + child.on("exit", (code, signal) => { + if (code !== 0 && code !== null) text += `\n[exited early with code ${code}]`; + else if (signal) text += `\n[killed by signal ${signal}]`; + }); + return { get: () => text.trim() }; +} + +/** One decoded Faro transport body — the shape `FetchTransport` posts to + * `/telemetry/collect` (`@grafana/faro-core`'s `TransportBody`). */ +interface FaroBody { + meta?: Record<string, unknown>; + exceptions?: Record<string, unknown>[]; + logs?: Record<string, unknown>[]; + measurements?: Record<string, unknown>[]; + events?: Record<string, unknown>[]; +} + +/** `flushFaro`'s probe check over a spec's captured bodies. */ +const eventSeen = (captured: FaroBody[]) => (ref: string) => + captured.flatMap((b) => b.events ?? []).some((e) => (e.attributes as Record<string, unknown> | undefined)?.["hot.ref"] === ref); + +/** + * Reads the e2e-only hooks (`sentry.ts`'s `localTestSentryEnabled()` branch) + * from the browser. Every Sentry envelope is `[EnvelopeHeader, Item[]]`; + * each `Item` is `[ItemHeader, payload]`. Flattens to just the + * `event`-typed payloads. + */ +function readSentryEvents(page: Page): Promise<Record<string, unknown>[]> { + return page.evaluate(() => { + const envelopes = + (window as unknown as { __t06SentryCapture?: unknown[][] }).__t06SentryCapture ?? []; + const events: Record<string, unknown>[] = []; + for (const envelope of envelopes) { + const items = (envelope[1] as unknown[][]) ?? []; + for (const item of items) { + const header = item[0] as { type?: string } | undefined; + if (header?.type === "event") events.push(item[1] as Record<string, unknown>); + } + } + return events; + }); +} + +/** `window.__t06ReportDemoEvent`, called from the browser with a minimal + * `MonitorPayload`/`DemoEventContext` pair — bypasses `monitorDemos` (see + * `sentry.ts#reportDemoEventUnguarded`'s doc comment) so this deterministic + * spec can drive `reportDemoEvent`'s reporting logic without a real + * (E2E_LIVE-gated) preview mount. */ +function callReportDemoEvent(page: Page, message: string): Promise<void> { + return page.evaluate((msg) => { + ( + window as unknown as { + __t06ReportDemoEvent?: ( + payload: { type: string; kind: string; message: string }, + context: { tier: number; framework: string; htMajor: string }, + ) => void; + } + ).__t06ReportDemoEvent?.( + { type: "hot-runner-monitor", kind: "error", message: msg }, + { tier: 1, framework: "react", htMajor: "18" }, + ); + }, message); +} + +test.describe("Faro in the authoring app", () => { + test.skip( + process.env.E2E_TELEMETRY !== "1", + "set E2E_TELEMETRY=1 and build with VITE_TELEMETRY_LOCAL=1 first", + ); + // One worker, one server: `beforeAll`/`afterAll` are per-worker, and this + // suite's `--strictPort` preview process is shared state — under the + // default `fullyParallel` config (playwright.config.ts) each of Playwright's + // parallel workers would spawn its own and race on the port, with whichever + // worker's `afterAll` fires first killing the server underneath the others. + test.describe.configure({ mode: "serial" }); + test.use({ baseURL: BASE_URL }); + + let server: ChildProcess; + + test.beforeAll(async () => { + // Fail loudly rather than silently reusing whatever already answers on + // this port (the exact 4173-reuse trap AGENTS.md warns about, one port + // block over) — a stale server from an earlier interrupted run would + // otherwise serve an unknown build and every assertion below would be + // testing the wrong bits. + const already = await fetch(BASE_URL).then(() => true).catch(() => false); + if (already) { + throw new Error( + `something is already answering on :${PORT} — kill it first (lsof -ti :${PORT} | xargs kill)`, + ); + } + server = spawn( + "node_modules/.bin/vite", + ["preview", "--host", "127.0.0.1", "--port", String(PORT), "--strictPort"], + { cwd: AUTHORING_DIR, stdio: "pipe" }, + ); + const diagnostics = captureServerDiagnostics(server); + try { + await waitForServer(BASE_URL, 30_000); + } catch (err) { + throw new Error(`preview server on :${PORT} never came up: fetch: ${formatFetchFailure(err)} | server output: ${diagnostics.get() || "(none)"}`); + } + }); + + test.afterAll(() => { + server?.kill(); + }); + + /** Every `/telemetry/collect` POST this test has seen, decoded. Registered + * before navigation so nothing is missed. */ + function captureTelemetry(page: import("@playwright/test").Page): FaroBody[] { + const bodies: FaroBody[] = []; + void page.route("**/telemetry/collect", async (route: Route) => { + const body = route.request().postDataJSON() as FaroBody; + bodies.push(body); + await route.fulfill({ status: 200, body: "" }); + }); + return bodies; + } + + // `beforeSend` must apply the shared noise gates (contract §6) to Faro + // exception items, same as Sentry's own `beforeSend` — otherwise a + // benign ResizeObserver-loop warning (or any of the other + // `isUnhandledNoise`/`isForeignUnhandled` shapes) reaches Loki and mints + // a fresh §F.3 `fp:` first-seen entry, paging on noise Sentry filters. + test("an unhandled ResizeObserver-loop warning does NOT reach Faro (shared noise gate)", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const marker = "T06 e2e D-I2 noise probe " + Date.now(); + await page.evaluate((msg) => { + setTimeout(() => { + throw new Error(`ResizeObserver loop completed with undelivered notifications. (${msg})`); + }); + }, marker); + + // A real, non-noise probe right after, on the same page — proves the + // page (and Faro transport) is still alive and would have captured the + // noise probe too if the gate had not dropped it, rather than this + // being a false pass from nothing having run yet. + await page.evaluate((msg) => { + setTimeout(() => { throw new Error(`T06 e2e D-I2 control probe (${msg})`); }); + }, marker); + await expect + .poll(() => captured.flatMap((b) => b.exceptions ?? []).some((e) => String(e.value ?? "").includes("control probe"))) + .toBe(true); + + const noiseHit = captured.flatMap((b) => b.exceptions ?? []).find((e) => String(e.value ?? "").includes(marker) && !String(e.value ?? "").includes("control probe")); + assert(!noiseHit, "a ResizeObserver-loop warning must never reach Faro/telemetry/collect"); + }); + + test("the Outlook/Office safelink scanner's injected rejection does NOT reach Faro (shared noise gate)", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const marker = "T06 e2e D-I2 scanner probe " + Date.now(); + await page.evaluate((msg) => { + setTimeout(() => { + // eventGate.ts's INJECTED_SCANNER_MESSAGES regex, same wording DEMOS-5F + // classified as the Office/Outlook safelink scanner's own injected + // rejection (not our own code's, never authored by this app). + throw new Error( + `Non-Error promise rejection captured with value: Object Not Found Matching Id:12, MethodName:update, ParamCount:4 (${msg})`, + ); + }); + }, marker); + + await page.evaluate((msg) => { + setTimeout(() => { throw new Error(`T06 e2e D-I2 control probe (${msg})`); }); + }, marker); + await expect + .poll(() => captured.flatMap((b) => b.exceptions ?? []).some((e) => String(e.value ?? "").includes("control probe"))) + .toBe(true); + + const scannerHit = captured.flatMap((b) => b.exceptions ?? []).find((e) => String(e.value ?? "").includes(marker) && !String(e.value ?? "").includes("control probe")); + assert(!scannerHit, "the Office scanner's injected rejection must never reach Faro/telemetry/collect"); + }); + + // faro-core's default `dedupe: true` keeps one `lastPayload` per API + // (events/measurements) and silently skips a push that deep-equals the + // previous one, with no time window — a real second `example.downloaded` + // (repeat Download, Share) or a second `example.open` on a guide's + // second example (identical `ref`-keyed attrs) must still leave the + // browser. `faro.ts`'s `event()`/`metric()` pass `skipDedupe: true`; + // `window.__t06Telemetry` (a build+host-gated e2e-only hook, same + // guarantee as `__t06ReportDemoEvent`) calls the real facade methods + // directly so this proves the facade's own behaviour without driving the + // real save/download UI. + test("two identical example.downloaded events both reach Faro (facade skipDedupe)", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const ref = "t06-h1-event-probe-" + Date.now(); + await page.evaluate((probeRef) => { + const hook = ( + window as unknown as { + __t06Telemetry?: { event: (name: string, attrs: Record<string, string>) => void }; + } + ).__t06Telemetry; + hook?.event("example.downloaded", { surface: "authoring", kind: "docs", ref: probeRef }); + hook?.event("example.downloaded", { surface: "authoring", kind: "docs", ref: probeRef }); + }, ref); + + const matching = () => + captured + .flatMap((b) => b.events ?? []) + .filter((e) => e.name === "example.downloaded" && (e.attributes as Record<string, unknown> | undefined)?.["hot.ref"] === ref); + await expect.poll(matching).toHaveLength(2); + }); + + test("two identical bucket.resolve_ms measurements both reach Faro (facade skipDedupe)", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const bucket = "t06-h1-metric-probe-" + Date.now(); + await page.evaluate((probeBucket) => { + const hook = ( + window as unknown as { + __t06Telemetry?: { metric: (name: string, values: Record<string, number>, attrs: Record<string, string>) => void }; + } + ).__t06Telemetry; + hook?.metric("bucket.resolve_ms", { duration_ms: 0 }, { bucket: probeBucket, outcome: "ok" }); + hook?.metric("bucket.resolve_ms", { duration_ms: 0 }, { bucket: probeBucket, outcome: "ok" }); + }, bucket); + + const matching = () => + captured + .flatMap((b) => b.measurements ?? []) + .filter((m) => m.type === "bucket.resolve_ms" && (m.context as Record<string, unknown> | undefined)?.["hot.bucket"] === bucket); + await expect.poll(matching).toHaveLength(2); + }); + + // The ingest gate answers an over-limit IP 429 with `Retry-After: 60` (the + // limiter window). The page clock is advanced only after the 429 has been + // answered, so Faro's own 10 s request timeout never fires under fake time. + test("a batch answered 429 with Retry-After: 60 is sent again after the wait, with the same Idempotency-Key", async ({ page }) => { + await stubShell(page); + await page.clock.install(); + const ref = "retry-probe-" + Date.now(); + const attempts: { key: string; refs: string[]; at: number }[] = []; + let limitedKey: string | null = null; + await page.route("**/telemetry/collect", async (route: Route) => { + const body = route.request().postDataJSON() as FaroBody; + const key = route.request().headers()["idempotency-key"] ?? ""; + const refs = (body.events ?? []).map((e) => String((e.attributes as Record<string, unknown> | undefined)?.["hot.ref"])); + attempts.push({ key, refs, at: await page.evaluate(() => Date.now()) }); + if (limitedKey === null && refs.includes(ref)) { + limitedKey = key; + await route.fulfill({ status: 429, headers: { "retry-after": "60" }, body: "" }); + return; + } + await route.fulfill({ status: 204, body: "" }); + }); + await page.goto("/"); + + await page.evaluate((probeRef) => { + (window as unknown as { + __t06Telemetry?: { event: (name: string, attrs: Record<string, string>) => void }; + }).__t06Telemetry?.event("example.downloaded", { surface: "authoring", kind: "docs", ref: probeRef }); + }, ref); + await expect.poll(() => limitedKey, { timeout: 20_000 }).not.toBeNull(); + const forKey = () => attempts.filter((a) => a.key === limitedKey); + + // Relative to the 429'd attempt: the page clock keeps running while the test waits. + const [first] = forKey(); + const pageNow = await page.evaluate(() => Date.now()); + await page.clock.fastForward(Math.max(0, first!.at + 58_000 - pageNow)); + await page.waitForTimeout(500); + expect(forKey(), "not retried before the Retry-After window").toHaveLength(1); + await page.clock.fastForward(17_000); + await expect.poll(() => forKey().length).toBe(2); + const [, retry] = forKey(); + expect(retry!.refs).toContain(ref); + expect(retry!.at - first!.at).toBeGreaterThanOrEqual(60_000); + }); + + test("an uncaught error reaches Faro (window.onerror, via ErrorsInstrumentation)", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + await page.evaluate(() => { + setTimeout(() => { + throw new Error("T06 e2e uncaught probe " + Date.now()); + }); + }); + + await expect.poll(() => captured.flatMap((b) => b.exceptions ?? []).length).toBeGreaterThan(0); + const exception = captured.flatMap((b) => b.exceptions ?? []).find((e) => + String(e.value ?? "").includes("T06 e2e uncaught probe"), + ); + assert(exception, "no exception item matched the probe message"); + // Uncaught: NOT tagged handled — the facade's own `.error()` always sets + // `context.handled = "true"` (contract §6); Faro's automatic + // ErrorsInstrumentation never does. + assert( + (exception.context as Record<string, unknown> | undefined)?.handled !== "true", + "an uncaught window error must not carry context.handled = 'true'", + ); + }); + + // Faro's gecko-regex stack fallback can turn this message's own trailing + // URL into a fake, lineno-less frame, which `isForeignUnhandled` then + // reads as "foreign" and drops the whole event. This only proves the + // event is kept; it deliberately does not assert anything about the + // IP/email text surviving or being redacted (a separate concern). + test("an uncaught error whose message quotes a foreign URL is kept, not dropped as foreign", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const marker = Date.now(); + await page.evaluate((m) => { + setTimeout(() => { + throw new Error( + `HAIKU1 pii jane.doe@example.com 192.0.2.55 https://x.test/p?token=SECRET123 (${m})`, + ); + }); + }, marker); + + await expect + .poll(() => captured.flatMap((b) => b.exceptions ?? []).some((e) => String(e.value ?? "").includes(`HAIKU1 pii`) && String(e.value ?? "").includes(String(marker)))) + .toBe(true); + }); + + test("a render crash inside the error boundary reaches Faro too, still not handled", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/?__test_crash_boundary=1"); + + // The boundary's own fallback UI is Sentry's half of the tee — it can only + // render if Sentry.ErrorBoundary's componentDidCatch actually ran. + await expect(page.getByText("Something went wrong")).toBeVisible(); + + await expect.poll(() => captured.flatMap((b) => b.exceptions ?? []).length).toBeGreaterThan(0); + const exception = captured.flatMap((b) => b.exceptions ?? []).find((e) => + String(e.value ?? "").includes("T06 e2e render-crash probe"), + ); + assert(exception, "no exception item matched the render-crash probe message"); + assert( + (exception.context as Record<string, unknown> | undefined)?.handled !== "true", + "a render crash must not carry context.handled = 'true' — it is uncaught (ADR §E.1), just relayed manually", + ); + }); + + test("reportError (a handled diagnostic) reaches Faro, fingerprinted by its context, tagged handled=true", async ({ page }) => { + await page.route("**/api/versions", (route) => route.fulfill({ status: 500, body: "boom" })); + await page.route("https://sandpack.codesandbox.io/**", (route) => route.abort()); + await page.route("https://sandpack-bundler.codesandbox.io/**", (route) => route.abort()); + await page.route("**/broker/login**", (route) => route.abort()); + const captured = captureTelemetry(page); + await page.goto("/"); + + await expect.poll(() => captured.flatMap((b) => b.exceptions ?? []).length).toBeGreaterThan(0); + // Found by `fingerprint` (contract §7, `fingerprint(context, message)`) — + // a top-level Faro exception field the scrubber never touches, so it + // alone already proves this reached Faro even before checking `context`. + const exception = captured + .flatMap((b) => b.exceptions ?? []) + .find((e) => String(e.fingerprint ?? "").startsWith("versions-fetch:")); + assert(exception, "no exception item fingerprinted versions-fetch:… — reportError never reached Faro"); + // `handled` is in `attrs.ts#DIAGNOSTIC_TAG_KEYS` + // (`packages/runtime/src/telemetry/attrs.ts`, contract §3 "Diagnostic + // tags"), so the browser-side scrub keeps it. + assert( + (exception.context as Record<string, unknown> | undefined)?.handled === "true", + "reportError must tag every Faro push context.handled = 'true' (contract §6 error.handled split)", + ); + }); + + test("API requests carry x-hot-session, matching the Faro session id", async ({ page }) => { + await stubShell(page); + let versionsHeader: string | undefined; + await page.route("**/api/versions", async (route) => { + versionsHeader = route.request().headers()["x-hot-session"]; + await route.fulfill({ json: { latest: "18.0.0", next: null, versions: ["18.0.0"] } }); + }); + const captured = captureTelemetry(page); + await page.goto("/"); + await page.evaluate(() => { + setTimeout(() => { throw new Error("T06 e2e header probe"); }); + }); + await expect.poll(() => captured.length).toBeGreaterThan(0); + + assert(versionsHeader, "GET /api/versions carried no x-hot-session header"); + const sessionId = captured[0]?.meta?.session as { id?: string } | undefined; + assert(sessionId?.id, "no Faro item carried meta.session.id"); + assert( + versionsHeader === sessionId.id, + `x-hot-session (${versionsHeader}) must equal the Faro page-load id (${sessionId.id}) — same in-memory id, contract §6`, + ); + }); + + test("nothing is written to localStorage or sessionStorage by Faro or the facade", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + await page.evaluate(() => { + setTimeout(() => { throw new Error("T06 e2e storage probe"); }); + }); + await expect.poll(() => captured.length).toBeGreaterThan(0); + // Faro's (disabled) persistent-session write is debounced + // (`STORAGE_UPDATE_DELAY`, 1s in the SDK) — give it the chance to land + // before asserting its absence, or this assertion would pass for the + // wrong reason (too early to have seen a write that will still happen). + await page.waitForTimeout(1_500); + + const keys = await page.evaluate(() => ({ + local: Object.keys(localStorage), + session: Object.keys(sessionStorage), + })); + for (const key of [...keys.local, ...keys.session]) { + assert( + !key.toLowerCase().includes("faro"), + `Faro/facade must write no storage key — found "${key}" (contract §6/§10)`, + ); + } + }); + + test("no captured payload carries a query string, a user-agent string, an email, console text, or a Babel code frame", async ({ page }) => { + await page.route("**/api/versions", (route) => route.fulfill({ status: 500, body: "boom" })); + await page.route("https://sandpack.codesandbox.io/**", (route) => route.abort()); + await page.route("https://sandpack-bundler.codesandbox.io/**", (route) => route.abort()); + await page.route("**/broker/login**", (route) => route.abort()); + const captured = captureTelemetry(page); + // A harmless query param + fragment on the page URL itself — `App.tsx` + // does not recognise `probeleak`, so the app renders its ordinary `/` + // route; the only thing under test is whether Faro's `meta.page.url` + // strips it (contract §3: "strip query strings and fragments from every + // URL-valued field"). + await page.goto("/?probeleak=should-be-stripped#fragment"); + await page.evaluate(() => { + setTimeout(() => { throw new Error("T06 e2e scrub probe"); }); + }); + await expect.poll(() => captured.flatMap((b) => b.exceptions ?? []).length).toBeGreaterThanOrEqual(2); + + const EMAIL_RE = /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/; + // A real Chrome UA always names the engine and the platform together. + const UA_RE = /Mozilla\/5\.0|AppleWebKit|Gecko\)/; + const CODE_FRAME_GUTTER_RE = /^[ \t]*>?[ \t]*\d+[ \t]*\|/m; + const QUERY_STRING_RE = /\?[a-zA-Z0-9_=&%-]+=|probeleak=should-be-stripped/; + + function scan(value: unknown, path: string): void { + if (typeof value === "string") { + assert(!EMAIL_RE.test(value), `email-shaped string at ${path}: ${JSON.stringify(value)}`); + assert(!UA_RE.test(value), `user-agent string at ${path}: ${JSON.stringify(value)}`); + assert(!CODE_FRAME_GUTTER_RE.test(value), `Babel code-frame gutter at ${path}: ${JSON.stringify(value)}`); + assert(!QUERY_STRING_RE.test(value), `query string survived at ${path}: ${JSON.stringify(value)}`); + return; + } + if (Array.isArray(value)) { + value.forEach((v, i) => scan(v, `${path}[${i}]`)); + return; + } + if (value !== null && typeof value === "object") { + for (const [k, v] of Object.entries(value)) scan(v, `${path}.${k}`); + } + } + captured.forEach((body, i) => scan(body, `body[${i}]`)); + }); + + // ---- an uncaught error reaches Sentry (transport spy) ---------------------- + // + // Three cases, all against this describe block's `full`-scope build (the + // default — no VITE_SENTRY_SCOPE set): uncaught always reaches Sentry; + // reportError and demo-runtime reach it too, because full scope keeps + // today's behaviour. The mirror describe block below rebuilds with + // VITE_SENTRY_SCOPE=uncaught and proves the opposite for the latter two. + + test("an uncaught error reaches Sentry (transport spy)", async ({ page }) => { + await stubShell(page); + await page.goto("/"); + await page.evaluate(() => { + setTimeout(() => { throw new Error("T06 e2e I3 uncaught probe " + Date.now()); }); + }); + await expect.poll(() => readSentryEvents(page).then((e) => e.length)).toBeGreaterThan(0); + const events = await readSentryEvents(page); + const hit = events.find((e) => + JSON.stringify((e as { exception?: unknown }).exception ?? "").includes("T06 e2e I3 uncaught probe"), + ); + assert(hit, "no Sentry event matched the uncaught probe message"); + }); + + test("reportError reaches Sentry under full scope", async ({ page }) => { + await page.route("**/api/versions", (route) => route.fulfill({ status: 500, body: "boom" })); + await page.route("https://sandpack.codesandbox.io/**", (route) => route.abort()); + await page.route("https://sandpack-bundler.codesandbox.io/**", (route) => route.abort()); + await page.route("**/broker/login**", (route) => route.abort()); + await page.goto("/"); + await expect.poll(() => readSentryEvents(page).then((e) => e.length)).toBeGreaterThan(0); + const events = await readSentryEvents(page); + const hit = events.find((e) => (e as { tags?: { context?: string } }).tags?.context === "versions-fetch"); + assert(hit, "no Sentry event tagged context=versions-fetch — reportError did not reach Sentry under full scope"); + }); + + test("a demo-runtime event reaches Sentry under full scope, re-homed to the demo-runtime environment", async ({ page }) => { + await stubShell(page); + await page.goto("/"); + await callReportDemoEvent(page, "T06 e2e I3 demo-runtime probe"); + await expect.poll(() => readSentryEvents(page).then((e) => e.length)).toBeGreaterThan(0); + const events = await readSentryEvents(page); + const hit = events.find((e) => (e as { tags?: { surface?: string } }).tags?.surface === "demo-runtime"); + assert(hit, "no Sentry event tagged surface=demo-runtime — reportDemoEvent did not reach Sentry under full scope"); + // The re-homing that beforeSend restores. + assert( + (hit as { environment?: string }).environment === "demo-runtime", + `demo-runtime event must be re-homed to environment "demo-runtime", got ${JSON.stringify((hit as { environment?: string }).environment)}`, + ); + }); + + // ---- reportDemoEvent's own gate (previewMonitoring), not the + // ---- __t06ReportDemoEvent bypass above ----------------------------------- + // + // Every test above drives `reportDemoEventUnguarded` directly (the + // `__t06ReportDemoEvent` hook), which never exercises `reportDemoEvent`'s + // own `monitorDemos` gate. With no `VITE_MONITOR_DEMOS` (so `monitorDemos` + // is false, same as every real `dev:full` run), `reportDemoEvent` itself + // must not be a no-op. `__t06ReportDemoEventGuarded` calls + // `reportDemoEvent` (the real, guarded entry point `App.tsx`'s + // `onPreviewMessage` uses) so this proves the fix without a real + // (E2E_LIVE-gated) preview mount. + test("reportDemoEvent (guarded) reaches Faro under the local leg, and never Sentry", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + + const guardedMarker = "T06 e2e R3 F10 guarded probe " + Date.now(); + await page.evaluate((msg) => { + ( + window as unknown as { + __t06ReportDemoEventGuarded?: ( + payload: { type: string; kind: string; message: string }, + context: { tier: number; framework: string; htMajor: string }, + ) => void; + } + ).__t06ReportDemoEventGuarded?.( + { type: "hot-runner-monitor", kind: "error", message: msg }, + { tier: 1, framework: "react", htMajor: "18" }, + ); + }, guardedMarker); + + // The AE metric (contract §5): `previewMonitoring` + // (`monitorDemos || localTestSentryEnabled()`) is what lets this call + // through `reportDemoEvent`'s gate at all. + await expect + .poll(() => captured.flatMap((b) => b.measurements ?? []).some((m) => m.type === "preview.runtime_error")) + .toBe(true); + + // Control: a second demo-runtime event, fired through the unguarded + // hook (opts.sentry=true default), which does reach Sentry on this + // exact dist (proved by the sibling "I3: a demo-runtime event reaches + // Sentry under full scope" test above). Waiting for this first rules + // out "no Sentry event yet because nothing has flushed" as the reason + // the guarded marker is absent below — Sentry capture/transport is + // provably alive on this page. + const controlMarker = "T06 e2e R3 F10 control probe " + Date.now(); + await callReportDemoEvent(page, controlMarker); + await expect + .poll(() => readSentryEvents(page).then((events) => events.some((e) => JSON.stringify(e).includes(controlMarker)))) + .toBe(true); + + // Never Sentry: `opts.sentry` is `monitorDemos` (false in this build, + // same as every real local run) — independent of `previewMonitoring` and + // of `diagnosticsGoToSentry`, which is true in this build (the control + // above proved it). Matched on the message text, not `tags.surface`: + // the control event also carries `surface: "demo-runtime"`, so a + // surface-only match would pass even if the guarded call had leaked + // through too. + const events = await readSentryEvents(page); + const guardedHit = events.find((e) => JSON.stringify(e).includes(guardedMarker)); + assert(!guardedHit, "reportDemoEvent must never reach Sentry through the R3 F10 local leg"); + }); + // ---- the edit-burst collapse in front of the facade ------------------------ + // + // Typing one throwing line relayed one `preview.runtime_error` per + // half-typed prefix. Drives the same two entry points `App.tsx` uses — the + // guarded `reportDemoEvent` and the edit signal `noteDemoEdit` — through + // their local-only hooks, so no real (E2E_LIVE-gated) preview is needed. + test("a keystroke ladder emits one preview.runtime_error + one Faro record; a first-load error counts at once; Sentry is not collapsed", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + // The stubbed version list remounts the preview once after load, and a + // remount closes the open burst; drive the ladder after it. + await expect(page).toHaveURL(/[?&]v=18\.0\.0\b/); + + const relay = (message: string, sentry = false) => + page.evaluate( + ([msg, viaSentry]) => { + const w = window as unknown as Record<string, (p: unknown, c: unknown) => void>; + const hook = viaSentry ? w.__t06ReportDemoEvent : w.__t06ReportDemoEventGuarded; + hook({ type: "hot-runner-monitor", kind: "error", message: msg }, { tier: 1, framework: "react", htMajor: "18" }); + }, + [message, sentry] as const, + ); + const noteEdit = () => + page.evaluate(() => (window as unknown as { __t06ReportDemoEventNoteEdit: () => void }).__t06ReportDemoEventNoteEdit()); + // Letters only in the markers: digits would be normalised to `<n>` in the shape. + const run = "F" + Math.random().toString(36).replace(/[^a-z]/g, "").slice(0, 8); + // Scoped to this test's own relays: the real (bundler-less) preview on this + // page relays events of its own — a Handsontable "Theme … is already + // registered" console warning, observed — which are real reports, just not + // the ones this test drives. Every relay below is `kind: "error"` + // (reason `uncaught`), and every record it produces carries `run`. + const runtimePoints = () => + captured + .flatMap((b) => b.measurements ?? []) + .filter((m) => m.type === "preview.runtime_error") + .filter((m) => (m.context as Record<string, string> | undefined)?.["hot.reason"] === "uncaught"); + const demoRecords = () => + captured + .flatMap((b) => b.exceptions ?? []) + .filter((e) => (e.context as Record<string, string> | undefined)?.["hot.surface"] === "demo-runtime") + .filter((e) => String(e.value ?? "").includes(run) || String(e.value ?? "").includes("is not defined")); + const ladder = ["s", "se", "set", "setT", "setTi", "setTim", "setTime", "setTimeo"].map((p) => `${p} is not defined`); + ladder.push(`Unexpected token ${run}`, `Unterminated string constant ${run}`); + for (const rung of ladder) { + await noteEdit(); + await relay(rung); + } + await noteEdit(); // the last keystroke + await relay(`ladder final ${run} 'secretLiteral'`); + + // Held back while the burst is open (the settle window is 2 s). + await page.waitForTimeout(700); + await flushFaro(page, eventSeen(captured)); + expect(runtimePoints()).toHaveLength(0); + + await expect.poll(() => runtimePoints().length, { timeout: 10_000 }).toBe(1); + await expect.poll(() => demoRecords().length).toBe(1); + // Nothing else trickles in after the burst closed. + await page.waitForTimeout(1500); + await flushFaro(page, eventSeen(captured)); + expect(runtimePoints()).toHaveLength(1); + expect(demoRecords()).toHaveLength(1); + + // The one record is the final run's, as the §7 shape: handled, no stack, + // quoted text (authored content, contract §3) replaced. + const record = demoRecords()[0]!; + expect(record.type).toBe("DemoError"); + expect(String(record.value)).toBe(`ladder final ${run} <str>`); + expect(record.stacktrace).toBeUndefined(); + expect((record.context as Record<string, string>).handled).toBe("true"); + expect(JSON.stringify(captured)).not.toContain("secretLiteral"); + expect(runtimePoints()[0]!.context).toMatchObject({ "hot.surface": "demo-runtime", "hot.reason": "uncaught" }); + + // A first-load / interaction error (no edit open): counted without the settle wait. + await relay(`first load ${run}`); + await flushFaro(page, eventSeen(captured)); + expect(runtimePoints()).toHaveLength(2); + expect(demoRecords().map((r) => r.value)).toContain(`first load ${run}`); + + // Sentry is NOT behind the collapse: under an open burst, every rung still + // reaches it at once (the unguarded hook is the `opts.sentry` path). + await noteEdit(); + for (const rung of ["a is not defined", "ab is not defined", "abc is not defined"]) { + await relay(`${rung} ${run}`, true); + } + await expect + .poll(() => readSentryEvents(page).then((events) => events.filter((e) => JSON.stringify(e).includes(run)).length)) + .toBe(3); + }); + + // The test above drives the edit signal through its hook; this one proves + // `App.tsx` actually sends it: a real keystroke in the code editor must open + // a burst, so an error relayed right after it is held back until the editor + // goes quiet, instead of counting at once like a first-load error. + test("a code-editor keystroke opens the edit burst (App.tsx wiring)", async ({ page }) => { + await stubShell(page); + // Every bundler host, the versioned one too: with a live bundler the keystroke's run + // starts after the injected relay (a run's start drops what the burst held), and a + // compile error of the typed `x` replaces it, so the outcome would race the bundler. + await page.route(/\.codesandbox\.io\//, (route) => route.abort()); + const captured = captureTelemetry(page); + await page.goto("/"); + await expect(page).toHaveURL(/[?&]v=18\.0\.0\b/); + await expect(activeEditor(page)).toBeVisible(); + + const run = "F" + Math.random().toString(36).replace(/[^a-z]/g, "").slice(0, 8); + const points = () => + captured + .flatMap((b) => b.measurements ?? []) + .filter((m) => m.type === "preview.runtime_error") + .filter((m) => (m.context as Record<string, string> | undefined)?.["hot.reason"] === "uncaught"); + + await activeEditor(page).click(); + await page.keyboard.type("x"); + await page.evaluate((msg) => { + (window as unknown as Record<string, (p: unknown, c: unknown) => void>).__t06ReportDemoEventGuarded( + { type: "hot-runner-monitor", kind: "error", message: msg }, + { tier: 1, framework: "react", htMajor: "18" }, + ); + }, `after keystroke ${run}`); + + await page.waitForTimeout(700); + await flushFaro(page, eventSeen(captured)); + expect(points(), "an error right after a keystroke waits for the burst to settle").toHaveLength(0); + await expect.poll(() => points().length, { timeout: 10_000 }).toBe(1); + }); + + // Handsontable's load-time notices are console warnings (18: the theme + // notice; 17: the `date` deprecation), relayed on every preview load. A + // warning is not a runtime error; a demo's own console.error is. + test("a relayed console warning is not a preview.runtime_error; a console.error is", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + const run = "F" + Math.random().toString(36).replace(/[^a-z]/g, "").slice(0, 8); + const relay = (kind: string, message: string) => + page.evaluate( + ([k, msg]) => + (window as unknown as Record<string, (p: unknown, c: unknown) => void>).__t06ReportDemoEventGuarded( + { type: "hot-runner-monitor", kind: k, message: msg }, + { tier: 1, framework: "react", htMajor: "18" }, + ), + [kind, message] as const, + ); + const themeNotice = 'Theme "main" is already registered. Registration skipped.'; + const consoleError = `a real console.error ${run}`; + const points = (message: string) => + captured + .flatMap((b) => b.measurements ?? []) + .filter((m) => m.type === "preview.runtime_error") + .filter((m) => (m.context as Record<string, string>)["hot.fingerprint"] === fingerprint("demo-runtime", message)); + + await relay("console-warn", themeNotice); + await relay("console-error", consoleError); + + await expect.poll(() => points(consoleError).length, { timeout: 10_000 }).toBe(1); + expect(points(consoleError)[0]!.context).toMatchObject({ "hot.reason": "console" }); + // The page's own preview (if the bundler answers) relays the same notice on load. + await page.waitForTimeout(3000); + await flushFaro(page, eventSeen(captured)); + expect(points(themeNotice), "the notice, whoever relayed it, never counts").toHaveLength(0); + }); + + // A syntax error typed into a Tier-1 parcel example never reaches the + // bundler — the client-side pre-transpile rejects it — so + // `sandpack.compile_error` must fire for the most common compile error + // there is. Real keystrokes in the real editor, the real runtime, babel + // and collapse, and a real preview. + // + // E2E_LIVE, not just E2E_TELEMETRY: the edit path needs a mounted Sandpack + // client, and with every bundler host aborted `mount()` never resolves + // (the preview stays `booting`, measured) — so no keystroke reaches the + // runtime at all. The live preview is also what makes the keystroke- + // prefix rungs (`c`..`cons`) run and relay ReferenceErrors, which the + // compile failure must keep out of `preview.runtime_error` (the + // `replacesRun` rule). + test("a syntax error typed key by key reaches /telemetry/collect as one sandpack.compile_error, not a runtime error", async ({ page }) => { + test.skip(process.env.E2E_LIVE !== "1", "set E2E_LIVE=1 (needs the hosted Sandpack bundler) to run the typed compile-error check"); + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + await previewReady(page); + const measurementsSince = (mark: number) => captured.slice(mark).flatMap((b) => b.measurements ?? []); + const mark = captured.length; + + await activeEditor(page).click(); + await page.keyboard.press("ControlOrMeta+End"); + await page.keyboard.press("Enter"); + // No delay on purpose: the prefixes' runs relay their ReferenceErrors + // after later keystrokes (compile slower than the typist). With a 40 ms + // delay the relays land before the next keystroke and the `replacesRun` + // rule goes unexercised (measured: that mutation stayed green). + await page.keyboard.type("const R9C = ;", { delay: 0 }); + + await expect + .poll(() => measurementsSince(mark).filter((m) => m.type === "sandpack.compile_error").length, { timeout: 15_000 }) + .toBe(1); + const [point] = measurementsSince(mark).filter((m) => m.type === "sandpack.compile_error"); + const ctx = point!.context as Record<string, string>; + expect(ctx["hot.fingerprint"]).toMatch(/^sandpack\.compile_error:[0-9a-f]{16}$/); + expect(ctx["hot.ht_major"]).toMatch(/^\d+$/); + expect(ctx["hot.framework"]).toBeTruthy(); + // The compile point is only emitted when the burst closes, in the same + // flush as anything the burst still held. One short negative wait + // anyway: nothing trickles in afterwards. + await page.waitForTimeout(1500); + await flushFaro(page, eventSeen(captured)); + const after = measurementsSince(mark); + expect(after.filter((m) => m.type === "sandpack.compile_error")).toHaveLength(1); + expect( + after.filter((m) => m.type === "preview.runtime_error"), + "no runtime error (of any reason) from the rungs of a line that ends in a syntax error", + ).toHaveLength(0); + // No authored text on the wire (contract §3): the point carries a hash only. + expect(JSON.stringify(captured.slice(mark))).not.toContain("R9C"); + }); + + // Every prefix of a typed throwing line runs and relays in the same preview + // document, and the closing `;` transpiles to the sandbox already running, + // so nothing re-runs after it. Waiting for the finished line's relay before + // typing the `;` pins that order. E2E_LIVE for the same reason as the test + // above: the edit path needs a mounted Sandpack client. + test("a runtime error typed key by key reaches /telemetry/collect as one preview.runtime_error with its message", async ({ page }) => { + test.skip(process.env.E2E_LIVE !== "1", "set E2E_LIVE=1 (needs the hosted Sandpack bundler) to run the typed runtime-error check"); + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/?example=javascript"); + await previewReady(page); + const mark = captured.length; + // Letters only: digits would be normalised to `<n>` in the record's shape. + const marker = "typed" + Math.random().toString(36).replace(/[^a-z]/g, "").slice(0, 8); + const uncaught = () => + captured + .slice(mark) + .flatMap((b) => b.measurements ?? []) + .filter((m) => m.type === "preview.runtime_error") + .filter((m) => (m.context as Record<string, string>)["hot.reason"] === "uncaught"); + const records = () => + captured + .slice(mark) + .flatMap((b) => b.exceptions ?? []) + .filter((e) => String(e.value ?? "").includes(marker)); + + await page.evaluate(() => { + const w = window as unknown as { __e2eRelays: string[] }; + w.__e2eRelays = []; + window.addEventListener("message", (e) => { + if (e.data?.type === "hot-runner-monitor") w.__e2eRelays.push(String(e.data.message)); + }); + }); + + await activeEditor(page).click(); + await page.keyboard.press("ControlOrMeta+End"); + await page.keyboard.press("Enter"); + await page.keyboard.type(`setTimeout(() => { throw new Error('${marker}'); }, 100)`, { delay: 20 }); + await page.waitForFunction( + (m) => (window as unknown as { __e2eRelays: string[] }).__e2eRelays.includes(m), + marker, + { timeout: 15_000 }, + ); + await page.keyboard.type(";"); + expect(await activeEditor(page).innerText(), "guard: the editor holds the finished line").toContain( + `setTimeout(() => { throw new Error('${marker}'); }, 100);`, + ); + + await expect.poll(() => uncaught().length, { timeout: 15_000 }).toBe(1); + await expect.poll(() => records().length).toBe(1); + await page.waitForTimeout(1500); + await flushFaro(page, eventSeen(captured)); + expect(uncaught(), "one point for the finished line, none for its prefixes").toHaveLength(1); + const [record] = records(); + expect(record!.type).toBe("DemoError"); + expect(uncaught()[0]!.context).toMatchObject({ "hot.surface": "demo-runtime", "hot.framework": "javascript" }); + }); +}); + +// ---- Sentry scope switch = uncaught ---------------------------------------- +// A second dist, built by this describe block's own `beforeAll` with +// VITE_SENTRY_SCOPE=uncaught (a build-time define, needing its own build +// and port). Proves the other half of the scope truth table: uncaught +// still reaches Sentry (ADR §E.1, regardless of scope), but reportError and +// demo-runtime do not (ADR §E.3: moved diagnostic reports go to the facade +// only once the scope is uncaught). +test.describe("Sentry scope switch = uncaught", () => { + test.skip( + process.env.E2E_TELEMETRY !== "1", + "set E2E_TELEMETRY=1 and build with VITE_TELEMETRY_LOCAL=1 first", + ); + test.describe.configure({ mode: "serial" }); + + const UNCAUGHT_PORT = Number(process.env.E2E_TELEMETRY_UNCAUGHT_PORT ?? 4712); + const UNCAUGHT_BASE_URL = `http://127.0.0.1:${UNCAUGHT_PORT}`; + const OUT_DIR = "dist-uncaught-scope"; + test.use({ baseURL: UNCAUGHT_BASE_URL }); + + let server: ChildProcess; + + test.beforeAll(async () => { + const already = await fetch(UNCAUGHT_BASE_URL).then(() => true).catch(() => false); + if (already) { + throw new Error( + `something is already answering on :${UNCAUGHT_PORT} — kill it first (lsof -ti :${UNCAUGHT_PORT} | xargs kill)`, + ); + } + // A genuinely separate build: VITE_SENTRY_SCOPE is a build-time + // `import.meta.env` read (`sentry.ts`'s `resolveSentryScope`), so there is + // no way to flip it per-request against the `full`-scope dist above. + execSync("node_modules/.bin/vite build --outDir " + OUT_DIR, { + cwd: AUTHORING_DIR, + env: { ...process.env, VITE_TELEMETRY_LOCAL: "1", VITE_SENTRY_SCOPE: "uncaught" }, + stdio: "pipe", + }); + server = spawn( + "node_modules/.bin/vite", + ["preview", "--outDir", OUT_DIR, "--host", "127.0.0.1", "--port", String(UNCAUGHT_PORT), "--strictPort"], + { cwd: AUTHORING_DIR, stdio: "pipe" }, + ); + const diagnostics = captureServerDiagnostics(server); + try { + await waitForServer(UNCAUGHT_BASE_URL, 30_000); + } catch (err) { + throw new Error(`preview server on :${UNCAUGHT_PORT} never came up: fetch: ${formatFetchFailure(err)} | server output: ${diagnostics.get() || "(none)"}`); + } + }); + + test.afterAll(() => { + server?.kill(); + }); + + function captureTelemetry(page: Page): FaroBody[] { + const bodies: FaroBody[] = []; + void page.route("**/telemetry/collect", async (route: Route) => { + bodies.push(route.request().postDataJSON() as FaroBody); + await route.fulfill({ status: 200, body: "" }); + }); + return bodies; + } + + test("an uncaught error still reaches Sentry under uncaught scope (ADR §E.1)", async ({ page }) => { + await stubShell(page); + await page.goto("/"); + await page.evaluate(() => { + setTimeout(() => { throw new Error("T06 e2e I3 uncaught-scope uncaught probe " + Date.now()); }); + }); + await expect.poll(() => readSentryEvents(page).then((e) => e.length)).toBeGreaterThan(0); + const events = await readSentryEvents(page); + const hit = events.find((e) => + JSON.stringify((e as { exception?: unknown }).exception ?? "").includes("T06 e2e I3 uncaught-scope uncaught probe"), + ); + assert(hit, "an uncaught error must reach Sentry under EVERY scope, including uncaught"); + }); + + test("reportError does NOT reach Sentry under uncaught scope, but still reaches the facade", async ({ page }) => { + await page.route("**/api/versions", (route) => route.fulfill({ status: 500, body: "boom" })); + await page.route("https://sandpack.codesandbox.io/**", (route) => route.abort()); + await page.route("https://sandpack-bundler.codesandbox.io/**", (route) => route.abort()); + await page.route("**/broker/login**", (route) => route.abort()); + const captured = captureTelemetry(page); + await page.goto("/"); + // The facade side must still fire (contract-mandated, scope-independent) — + // wait on that first so a false pass below can't be "nothing ran yet". + await expect.poll(() => captured.flatMap((b) => b.exceptions ?? []).length).toBeGreaterThan(0); + const faroHit = captured + .flatMap((b) => b.exceptions ?? []) + .find((e) => String((e as { fingerprint?: string }).fingerprint ?? "").startsWith("versions-fetch:")); + assert(faroHit, "reportError must still reach the facade under uncaught scope"); + + const events = await readSentryEvents(page); + const sentryHit = events.find((e) => (e as { tags?: { context?: string } }).tags?.context === "versions-fetch"); + assert(!sentryHit, "reportError must NOT reach Sentry under uncaught scope (ADR §E.3)"); + }); + + test("a demo-runtime event does NOT reach Sentry under uncaught scope, but still reaches the facade", async ({ page }) => { + await stubShell(page); + const captured = captureTelemetry(page); + await page.goto("/"); + await callReportDemoEvent(page, "T06 e2e I3 uncaught-scope demo-runtime probe"); + await expect.poll(() => captured.flatMap((b) => b.measurements ?? []).length).toBeGreaterThan(0); + const events = await readSentryEvents(page); + const sentryHit = events.find((e) => (e as { tags?: { surface?: string } }).tags?.surface === "demo-runtime"); + assert(!sentryHit, "a demo-runtime event must NOT reach Sentry under uncaught scope (ADR §E.3 / task Scope)"); + }); +}); + +function assert(value: unknown, message: string): asserts value { + if (!value) throw new Error(message); +} diff --git a/runner/e2e/telemetry-metrics.spec.ts b/runner/e2e/telemetry-metrics.spec.ts new file mode 100644 index 0000000000..30ba2e39c1 --- /dev/null +++ b/runner/e2e/telemetry-metrics.spec.ts @@ -0,0 +1,299 @@ +import { test, expect, type Route, type Page } from "@playwright/test"; +import { spawn, execSync, type ChildProcess } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { previewReady, expectGridRendered, trackSessions, activeEditor, flushFaro } from "./helpers.js"; + +// Observability contract §5 browser metric catalogue, live. +// +// Gated: needs a dist built with VITE_TELEMETRY_LOCAL=1 (contract §10, same +// as e2e/telemetry-faro.spec.ts) and a real preview mount (E2E_LIVE=1). The +// Tier-2 case additionally needs a local API worker — this spec starts its +// own `wrangler dev` inside workers/api, and builds the dist with +// `VITE_API_BASE` pointed at it. Never the production API: a plain +// `VITE_TELEMETRY_LOCAL=1` build inherits +// `apps/authoring/.env.production`'s `VITE_API_BASE=https://demos.handsontable.com` +// (that file outranks `.env.local`, and the app's own `:8787` fallback only +// applies to a falsy value), so a Tier-2 session created here without +// overriding it would land in the real production container pool. +// +// cd workers/api && npx wrangler dev --port 4810 --inspector-port 4811 +// (needs workers/api/.dev.vars — copy from the main checkout, PREVIEW_HOST=localhost:4810) +// VITE_TELEMETRY_LOCAL=1 VITE_API_BASE=http://localhost:4810 \ +// pnpm --filter @handsontable/demo-authoring exec vite build --outDir dist-telemetry-metrics +// E2E_LIVE=1 E2E_TELEMETRY=1 pnpm e2e e2e/telemetry-metrics.spec.ts +// +// This spec manages its own preview server (like telemetry-faro.spec.ts), +// never the shared playwright.config.ts webServer (:4173, no +// VITE_TELEMETRY_LOCAL). +// +// CI home: `.github/workflows/e2e-o11y-local.yml` — on workflow_dispatch, +// nightly, and PRs touching the o11y ingest path. Not ci.yml's +// `e2e-telemetry` job: that job's two specs are self-contained (page.route +// mocks, no real API worker), this one needs Docker + a real +// `wrangler dev`, which the shared Playwright container image can't provide. + +// Env-overridable (per-worktree port block) — a local reproduction of the +// CI job running alongside other worktrees on the same machine sets these +// to its own block instead of colliding on the default 4800-4899. +const AUTHORING_PORT = Number(process.env.E2E_TELEMETRY_METRICS_AUTHORING_PORT ?? 4800); +const API_PORT = Number(process.env.E2E_TELEMETRY_METRICS_API_PORT ?? 4810); +const API_INSPECTOR_PORT = Number(process.env.E2E_TELEMETRY_METRICS_API_INSPECTOR_PORT ?? 4811); +const BASE_URL = `http://localhost:${AUTHORING_PORT}`; +const API_BASE_URL = `http://localhost:${API_PORT}`; +const AUTHORING_DIR = fileURLToPath(new URL("../apps/authoring", import.meta.url)); +const API_DIR = fileURLToPath(new URL("../workers/api", import.meta.url)); +const OUT_DIR = "dist-telemetry-metrics"; + +function waitForServer(url: string, timeoutMs: number): Promise<void> { + const deadline = Date.now() + timeoutMs; + return new Promise((resolve, reject) => { + const attempt = () => { + fetch(url) + .then(() => resolve()) + .catch((err) => { + if (Date.now() > deadline) reject(err); + else setTimeout(attempt, 200); + }); + }; + attempt(); + }); +} + +/** One decoded Faro transport body — same shape telemetry-faro.spec.ts + * captures. */ +interface FaroBody { + measurements?: { type?: string; values?: Record<string, number>; context?: Record<string, string> }[]; + events?: { name?: string; attributes?: Record<string, string> }[]; +} + +/** Every `/telemetry/collect` POST this test has seen, decoded. Registered + * before navigation so nothing is missed — same helper as telemetry-faro.spec.ts. */ +function captureTelemetry(page: Page): FaroBody[] { + const bodies: FaroBody[] = []; + void page.route("**/telemetry/collect", async (route: Route) => { + bodies.push(route.request().postDataJSON() as FaroBody); + await route.fulfill({ status: 200, body: "" }); + }); + return bodies; +} + +function measurementsOf(captured: FaroBody[], type: string) { + return captured.flatMap((b) => b.measurements ?? []).filter((m) => m.type === type); +} + +/** Insert a line at the top of the visible editor through CodeMirror's own + * dispatch (same as `editor-download.spec.ts#insertAtTop`) — `.cm-content` is + * contenteditable but virtualised, so a dispatch is the reliable edit path. */ +async function insertAtTop(page: Page, text: string) { + await activeEditor(page).waitFor(); + await page.evaluate(`(() => { + const view = document.querySelector('[data-pane-active="true"] .cm-content').cmTile.view; + view.dispatch({ changes: { from: 0, insert: ${JSON.stringify(text + "\n")} } }); + })()`); +} + +test.describe("Browser metrics catalogue, live", () => { + test.skip( + process.env.E2E_LIVE !== "1" || process.env.E2E_TELEMETRY !== "1", + "set E2E_LIVE=1 and E2E_TELEMETRY=1, and build with VITE_TELEMETRY_LOCAL=1 first", + ); + // One worker, one pair of servers: both the authoring preview and the local + // API worker are shared `--strictPort` processes for this whole file, so + // parallel Playwright workers would race on the ports (AGENTS.md's 4173-reuse + // trap, one port block over). + test.describe.configure({ mode: "serial" }); + test.use({ baseURL: BASE_URL }); + test.setTimeout(300_000); + + let authoringServer: ChildProcess; + let apiServer: ChildProcess; + + test.beforeAll(async () => { + // The default hook timeout (60s) is not enough for wrangler dev's own + // startup (the Sandbox container image check/build) plus the authoring + // build plus the preview server — all sequential, all inside one hook. + // On a cold CI runner (no cached image layers) the + // container check/build step alone can approach the old 180s budget, + // so the whole hook intermittently tripped the timeout on attempt 1 and + // only passed on Playwright's retry (masking the failure as green CI). + // 360s gives the cold-build path real headroom without masking a hang. + test.setTimeout(360_000); + const authoringAlready = await fetch(BASE_URL).then(() => true).catch(() => false); + if (authoringAlready) { + throw new Error( + `something is already answering on :${AUTHORING_PORT} — kill it first (lsof -ti :${AUTHORING_PORT} | xargs kill)`, + ); + } + const apiAlready = await fetch(API_BASE_URL).then(() => true).catch(() => false); + if (apiAlready) { + throw new Error( + `something is already answering on :${API_PORT} — kill it first (lsof -ti :${API_PORT} | xargs kill)`, + ); + } + + // The local API worker (Tier-2 needs Docker running — AGENTS.md). Started + // before the authoring build: the build only reads the URL, it does not + // need the server up yet, but starting it first means its own boot log is + // visible in the console if it fails before the (slower) build even runs. + apiServer = spawn( + "node_modules/.bin/wrangler", + ["dev", "--port", String(API_PORT), "--inspector-port", String(API_INSPECTOR_PORT)], + { cwd: API_DIR, stdio: "pipe" }, + ); + let apiStderr = ""; + apiServer.stderr?.on("data", (chunk) => { apiStderr += String(chunk); }); + apiServer.stdout?.on("data", (chunk) => { apiStderr += String(chunk); }); + try { + await waitForServer(API_BASE_URL, 60_000); + } catch (err) { + throw new Error(`local API worker on :${API_PORT} never came up: ${apiStderr || String(err)}`); + } + + // A genuinely separate build: VITE_API_BASE/VITE_TELEMETRY_LOCAL are + // build-time `import.meta.env` reads, so there is no way to point an + // existing dist at this run's local API after the fact — see the file + // header for why a bare VITE_TELEMETRY_LOCAL=1 build is not safe here. + execSync(`node_modules/.bin/vite build --outDir ${OUT_DIR}`, { + cwd: AUTHORING_DIR, + env: { ...process.env, VITE_TELEMETRY_LOCAL: "1", VITE_API_BASE: API_BASE_URL }, + stdio: "pipe", + }); + // AGENTS.md's own leak check (the dev-login bypass) — still applies to any + // build. NOT a bare "demos.handsontable.com" grep: that string is + // legitimately compiled in as the production-hostname constant + // (`reportingGate.ts`) and in guide/error-message text, so it fires on + // every build and would prove nothing. What this spec actually depends on + // is that `API_BASE` compiled to the LOCAL worker, checked positively + // below — a wrong host there is exactly the DEMOS-1x-shaped mistake this + // file's header warns about, and the positive check catches it whether the + // fallback silently won or `.env.production` did. + const devLeak = execSync(`grep -rl "VITE_DEV_USER\\|dev@handsontable.com" ${OUT_DIR} || true`, { + cwd: AUTHORING_DIR, + }).toString().trim(); + if (devLeak) throw new Error(`dev-login bypass leaked into ${OUT_DIR}:\n${devLeak}`); + const apiBaseCompiled = execSync(`grep -rl "localhost:${API_PORT}" ${OUT_DIR}/assets || true`, { + cwd: AUTHORING_DIR, + }).toString().trim(); + if (!apiBaseCompiled) { + throw new Error( + `VITE_API_BASE did not compile to http://localhost:${API_PORT} anywhere in ${OUT_DIR}/assets — ` + + `a Tier-2 session from this run would target the wrong (or production) API`, + ); + } + + authoringServer = spawn( + "node_modules/.bin/vite", + ["preview", "--outDir", OUT_DIR, "--port", String(AUTHORING_PORT), "--strictPort"], + { cwd: AUTHORING_DIR, stdio: "pipe" }, + ); + let authoringStderr = ""; + authoringServer.stderr?.on("data", (chunk) => { authoringStderr += String(chunk); }); + try { + await waitForServer(BASE_URL, 30_000); + } catch (err) { + throw new Error(`authoring preview on :${AUTHORING_PORT} never came up: ${authoringStderr || String(err)}`); + } + }); + + test.afterAll(() => { + authoringServer?.kill(); + apiServer?.kill(); + }); + + test("Tier-1 (Sandpack): opening an example emits exactly one preview.ready_ms, tier=1, right framework — and a recompile does not re-emit", async ({ + page, + }) => { + // Only the login redirect neutered (T06's `stubShell` also aborts both + // Sandpack hosts, which would prevent the live mount this test needs). + await page.route("**/broker/login**", (route) => route.abort()); + const captured = captureTelemetry(page); + + await page.goto("/?example=react"); + await previewReady(page, "sandpack"); + await expectGridRendered(page); + + await expect.poll(() => measurementsOf(captured, "preview.ready_ms").length, { timeout: 30_000 }).toBe(1); + const [point] = measurementsOf(captured, "preview.ready_ms"); + expect(point?.context?.["hot.tier"]).toBe("1"); + expect(point?.context?.["hot.framework"]).toBe("react"); + expect(point?.context?.["hot.outcome"]).toBe("ready"); + // `hot.bucket` must survive the real browser scrub — a + // non-empty string proves it reached the wire, not the `attrs.ts` unit + // test's own literal input (this is the one attribute this Tier-1 flow + // naturally sets; `reason`/`fingerprint` are proven against the real + // `scrubTelemetry`/`toAePoint` functions in + // `pipeline/telemetry-ae-only-attrs.test.mjs`, since neither a version + // switch nor a compile error is part of this spec's flow). + expect(typeof point?.context?.["hot.bucket"] === "string" && point.context["hot.bucket"].length > 0).toBe(true); + const durationMs = point?.values?.duration_ms; + expect(typeof durationMs === "number" && durationMs >= 0).toBe(true); + test.info().annotations.push({ + type: "measured preview.ready_ms (tier 1, react)", + description: String(durationMs), + }); + + // The recompile guard (`pipeline/browser-metrics.test.mjs` proves this + // synthetically; this proves it against the real bundler): an edit + // recompiles the sandbox (SandpackRuntime's `onReady` fires again on every + // clean compile), and `preview.ready_ms` must not re-emit. + const compileCountBefore = measurementsOf(captured, "sandpack.compile_ms").length; + await insertAtTop(page, "// t07-e2e-recompile-probe"); + await expect + .poll(() => measurementsOf(captured, "sandpack.compile_ms").length, { timeout: 30_000 }) + .toBeGreaterThan(compileCountBefore); + expect(measurementsOf(captured, "preview.ready_ms").length, "guard: a recompile must not re-emit preview.ready_ms").toBe(1); + }); + + test("Tier-2 (container): opening an example emits exactly one preview.ready_ms, tier=2, right framework", async ({ + page, + request, + }) => { + test.setTimeout(300_000); + const tracked = trackSessions(page); + try { + await page.route("**/broker/login**", (route) => route.abort()); + const captured = captureTelemetry(page); + + await page.goto("/?example=react-js"); + await previewReady(page, "container"); + await expectGridRendered(page); + + await expect.poll(() => measurementsOf(captured, "preview.ready_ms").length, { timeout: 30_000 }).toBe(1); + const [point] = measurementsOf(captured, "preview.ready_ms"); + expect(point?.context?.["hot.tier"]).toBe("2"); + expect(point?.context?.["hot.framework"]).toBe("react-js"); + expect(point?.context?.["hot.outcome"]).toBe("ready"); + const durationMs = point?.values?.duration_ms; + expect(typeof durationMs === "number" && durationMs >= 0).toBe(true); + test.info().annotations.push({ + type: "measured preview.ready_ms (tier 2, react-js)", + description: String(durationMs), + }); + + // session.start_ms rides the same create-POST clock; it must also have + // fired exactly once by the time the preview is ready. + await expect.poll(() => measurementsOf(captured, "session.start_ms").length).toBe(1); + expect(measurementsOf(captured, "session.start_ms")[0]?.context?.["hot.outcome"]).toBe("ready"); + + // An edit, for the HMR observation and the same + // no-re-emission guard `preview.ready_ms` gets on Tier-1. Not gated on + // an `hmr.roundtrip_ms` point actually landing — the same + // table predicts most starters use in-place HMR, which this hook cannot + // see (see `HmrRoundtripEvent`'s doc comment). + await insertAtTop(page, "// t07-e2e-hmr-probe"); + await page.waitForTimeout(5_000); + await flushFaro(page, (ref) => captured.some((b) => (b.events ?? []).some((e) => e.attributes?.["hot.ref"] === ref))); + expect( + measurementsOf(captured, "preview.ready_ms").length, + "guard: an edit must not re-emit preview.ready_ms on Tier-2 either", + ).toBe(1); + const hmr = measurementsOf(captured, "hmr.roundtrip_ms"); + test.info().annotations.push({ + type: "hmr.roundtrip_ms observed on react-js", + description: hmr.length > 0 ? String(hmr[0]?.values?.duration_ms) : "not observed (see T07-D4 / the support table)", + }); + } finally { + await tracked.cleanup(request); + } + }); +}); diff --git a/runner/package.json b/runner/package.json index 61bee34c8f..ccce037a9a 100644 --- a/runner/package.json +++ b/runner/package.json @@ -17,8 +17,13 @@ "test": "pnpm --filter @handsontable/demo-runtime build && node --experimental-strip-types --test pipeline/*.test.mjs", "e2e": "playwright test", "check:compiler-chunk": "node scripts/check-compiler-chunk.mjs", + "check:telemetry-leak": "node scripts/check-telemetry-leak.mjs", "e2e:matrix": "E2E_STARTER_MATRIX=1 PLAYWRIGHT_JSON_OUTPUT_NAME=test-results/starter-matrix.json playwright test e2e/starter-matrix.spec.ts --workers=2 --retries=2 --reporter=list,json", - "e2e:matrix:report": "node scripts/starter-matrix-report.mjs" + "e2e:matrix:report": "node scripts/starter-matrix-report.mjs", + "o11y:dev": "node scripts/o11y-dev.mjs", + "dev": "node scripts/dev.mjs --tier=1", + "dev:live": "node scripts/dev.mjs --tier=2", + "dev:full": "node scripts/dev.mjs --tier=full" }, "devDependencies": { "@playwright/test": "1.61.1", diff --git a/runner/packages/runtime/package.json b/runner/packages/runtime/package.json index 86f048b93c..a6527aec03 100644 --- a/runner/packages/runtime/package.json +++ b/runner/packages/runtime/package.json @@ -33,6 +33,10 @@ "./entry-script": { "types": "./dist/entry-script.d.ts", "default": "./dist/entry-script.js" + }, + "./telemetry": { + "types": "./dist/telemetry/index.d.ts", + "default": "./dist/telemetry/index.js" } }, "scripts": { diff --git a/runner/packages/runtime/src/container.ts b/runner/packages/runtime/src/container.ts index 813fc3c377..c8a2de2422 100644 --- a/runner/packages/runtime/src/container.ts +++ b/runner/packages/runtime/src/container.ts @@ -13,8 +13,15 @@ import type { DemoRuntime, FilesMap, HandsontableVersionRef, + HmrRoundtripEvent, + SessionStartTimingEvent, WriteFileOptions, } from "./types.js"; +// Re-exported so existing `@handsontable/demo-runtime/container` importers +// (`apps/authoring/src/telemetry/metrics.ts`) keep working — the interfaces +// themselves live in `types.ts`, so `DemoRuntime` can name the hook methods +// without a circular import. +export type { HmrRoundtripEvent, SessionStartTimingEvent } from "./types.js"; import { mintSessionId } from "./session.js"; import { applyHandsontableCss, applyHandsontableVersion } from "./version.js"; import { MONITOR_EVENT_CEILING, normalizeMonitorMessage, truncateMessage } from "./monitor.js"; @@ -326,6 +333,45 @@ const RELOAD_TIMEOUT_MS = 10_000; const FAILED_POLL_INTERVAL_MS = 10_000; const FAILED_POLLS_MAX = 12; +// Observability contract §5 timing hooks: `SessionStartTimingEvent`, +// `HmrRoundtripEvent` and the `onSessionStart`/`onHmr` methods below are +// declared on `DemoRuntime` itself (`types.ts`), as OPTIONAL members — this +// module implements them, never imports `@handsontable/demo-runtime/telemetry`, +// and `apps/authoring/src/telemetry/metrics.ts#wireRuntimeMetrics` is what +// turns the callbacks into `session.start_ms`/`hmr.roundtrip_ms` points +// against an injected `Telemetry`, through `runtime.onX?.(cb)` — no cast to +// the concrete class needed at the call site. + +/** Real in-place HMR (Vite) never triggers the consume-on-ready + * `onFrameLoad` path — only a dev server that full-page-reloads on an edit + * does — so this could otherwise sit set for minutes and get reported as + * the HMR round-trip duration for an unrelated later reload. A real round + * trip completes in at most a few seconds; 30s is generous headroom + * above that, chosen to bound the staleness window without being tight + * enough to false-negative a slow-but-real reload. */ +const HMR_ROUNDTRIP_STALE_MS = 30_000; + +/** + * Classify a failed `POST /api/session` the same way `sessionStartMessage` already + * tiers it for the user-facing message, but onto `session.start`'s outcome set + * (§5) instead of a sentence. Mirrors that function's precedence exactly — in + * particular the DEMOS-9 interception case (an envelope-less, ray-less 504) is + * checked before the generic timeout tier, because 504 is a member of both: "the + * response carries no sign of having come from our servers" is not a boot timeout. + */ +function classifySessionStartOutcome( + status: number, + failure: { code?: string; envelope: boolean }, + edge: { ray: string | null; headersReadable: boolean }, +): SessionStartTimingEvent["outcome"] { + if (failure.code?.startsWith("budget_")) return "budget_denied"; + if (failure.code === "at_capacity") return "at_capacity"; + if (failure.code === "container_starting") return "container_starting"; + if (!failure.envelope && status === UNREACHED_STATUS && edge.headersReadable && !edge.ray) return "error"; + if (!failure.envelope && TIMEOUT_STATUSES.has(status)) return "boot_timeout"; + return "error"; +} + /** The live-session API accepts only relative POSIX paths. */ function relativeFiles(files: FilesMap): FilesMap { return Object.fromEntries( @@ -385,6 +431,17 @@ export class ContainerRuntime implements DemoRuntime { private flushTimer: ReturnType<typeof setTimeout> | null = null; private readonly progressCbs = new Set<(log: string) => void>(); private readonly stderrCbs = new Set<(line: string) => void>(); + // ---- Timing hooks --------------------------------------------------- + private readonly sessionStartCbs = new Set<(e: SessionStartTimingEvent) => void>(); + private readonly hmrCbs = new Set<(e: HmrRoundtripEvent) => void>(); + /** Set true for the span of an explicit `reload()` navigation, so `onFrameLoad` + * does not mistake it for an HMR-driven full-page reload. */ + private reloadInFlight = false; + /** When the most recent post-ready edit was flushed to the dev server (§5 + * `hmr.roundtrip_ms`'s dispatch clock) — set only once the preview is already + * ready, by `flush()`, and consumed by the next `onFrameLoad` that is not our + * own `reload()`. */ + private lastEditFlushDispatchedAt: number | null = null; /** Dev-server stderr lines already relayed, keyed on `normalizeMonitorMessage` — * the same fingerprint `sentry.ts` groups the Sentry issue by — so a message the * server repeats every keystroke, or with only its clock changed (a build @@ -461,6 +518,11 @@ export class ContainerRuntime implements DemoRuntime { */ private readonly onFrameLoad = () => { if (this.disposed) return; + // Captured before anything below can set `didReady` — a load that arrives + // while the preview was already ready is a candidate HMR round trip, and this + // is the only navigation where that distinction is a fact about THIS load + // rather than about what happens after it (see the block near the end). + const wasReadyBeforeThisLoad = this.didReady; this.frameLoads += 1; const state = this.pendingFrameState; this.pendingFrameState = "unknown"; @@ -478,6 +540,22 @@ export class ContainerRuntime implements DemoRuntime { this.emitProgress("Dev server not answering yet — retrying…"); return; } + // §5 `hmr.roundtrip_ms`. Non-invasive: everything below is unchanged — + // this only reports a timing for a load the grace-timer dance below already + // treats as ready (a no-op `confirmAndEmitReady`, since `didReady` is already + // true). Excludes our own `reload()` navigation (`reloadInFlight`) and the very + // first, pre-ready navigation (`wasReadyBeforeThisLoad`). See `HmrRoundtripEvent`. + if (wasReadyBeforeThisLoad && !this.reloadInFlight && this.lastEditFlushDispatchedAt !== null) { + const durationMs = Math.round(performance.now() - this.lastEditFlushDispatchedAt); + this.lastEditFlushDispatchedAt = null; + // A load arriving long after the flush is not this flush's round trip + // — real in-place HMR never reaches this handler at all, so a stale + // timestamp here means an unrelated later reload. Drop it rather than + // report a bogus duration. + if (durationMs <= HMR_ROUNDTRIP_STALE_MS) { + for (const cb of this.hmrCbs) cb({ durationMs }); + } + } this.graceTimer = setTimeout(() => { this.graceTimer = null; void this.confirmAndEmitReady(); @@ -566,6 +644,19 @@ export class ContainerRuntime implements DemoRuntime { onStderr(cb: (line: string) => void): void { this.stderrCbs.add(cb); } + /** §5 `session.start_ms` — fires exactly once per `mount()` call, on both the + * success and the failure path (see `mount()`'s try/catch). */ + onSessionStart(cb: (e: SessionStartTimingEvent) => void): void { + this.sessionStartCbs.add(cb); + } + /** §5 `hmr.roundtrip_ms` — see `HmrRoundtripEvent` for what this does and does + * not observe. */ + onHmr(cb: (e: HmrRoundtripEvent) => void): void { + this.hmrCbs.add(cb); + } + private emitSessionStart(elapsedMs: number, outcome: SessionStartTimingEvent["outcome"]): void { + for (const cb of this.sessionStartCbs) cb({ elapsedMs, outcome }); + } private emitReady() { if (this.didReady) return; this.didReady = true; @@ -645,6 +736,10 @@ export class ContainerRuntime implements DemoRuntime { let previewUrl: string; let port: number; + // Declared here, not inside the try, so the catch below can still emit §5 + // `session.start_ms` when `fetch()` itself throws (a raw network error — no + // response, and so no `SessionStartError.diagnostics` to read a duration off). + let startedAt = performance.now(); try { // Serialised BEFORE the clock starts, not inline in the fetch call below. // Argument expressions are evaluated after `startedAt` would have been @@ -668,7 +763,7 @@ export class ContainerRuntime implements DemoRuntime { // fixed ceiling into an apparent spread, destroying the one distinction the // number exists to make. `performance.now()` for monotonicity: a clock step // must not read as a slow container. - const startedAt = performance.now(); + startedAt = performance.now(); const res = await fetch(`${this.opts.apiBase}/api/session`, { method: "POST", headers: { @@ -712,7 +807,30 @@ export class ContainerRuntime implements DemoRuntime { ); } ({ previewUrl, port } = (await res.json()) as { previewUrl: string; port: number }); + // §5 `session.start_ms`, "ready" — the create POST itself succeeded. Emitted + // here, inside the try, so a later `disposed` check (the mid-flight-dispose + // race just below) cannot suppress a measurement that already happened. + this.emitSessionStart(elapsedMs, "ready"); } catch (err) { + // §5 `session.start_ms` for the failure path, emitted BEFORE `dispose()` + // below clears `sessionStartCbs` — a listener added after dispose would never + // see it. `SessionStartError` (the res.ok-false branch above) always carries + // `diagnostics`; anything else reaching here is `fetch()` itself throwing (a + // raw network error), which has no response to read a duration or an edge id + // off, only the clock this function already keeps. + if (err instanceof SessionStartError && err.diagnostics) { + const envelope = typeof err.code === "string"; + this.emitSessionStart( + err.diagnostics.elapsedMs, + classifySessionStartOutcome( + err.status, + { code: err.code, envelope }, + { ray: err.diagnostics.ray, headersReadable: err.diagnostics.headersReadable }, + ), + ); + } else { + this.emitSessionStart(Math.round(performance.now() - startedAt), "error"); + } // A failed create can still leave a half-created session server-side // (the POST handler's file writes boot the container before the step // that failed). Tear the runtime down and DELETE by the local id — @@ -896,6 +1014,10 @@ export class ContainerRuntime implements DemoRuntime { } catch { /* a write that failed reports through onError, not through refresh */ } } if (this.disposed || !this.pointed || !this.previewUrl) return; + // Marks the navigation this method is about to make as "ours", so `onFrameLoad` + // does not mistake it for an HMR-driven full-page reload (§5 `hmr.roundtrip_ms`). + // Cleared in `settle()`, the only way out of the promise below. + this.reloadInFlight = true; return new Promise<void>((resolve) => { const iframe = this.opts.iframe; let settled = false; @@ -905,6 +1027,7 @@ export class ContainerRuntime implements DemoRuntime { clearTimeout(timer); iframe.removeEventListener("load", settle); this.reloadSettlers.delete(settle); + this.reloadInFlight = false; resolve(); }; // A container whose dev server died never navigates, and an unresolved promise @@ -1021,6 +1144,10 @@ export class ContainerRuntime implements DemoRuntime { const batch = [...this.quietPending.entries(), ...this.pending.entries()]; this.quietPending.clear(); this.pending.clear(); + // §5 `hmr.roundtrip_ms` dispatch clock — only once the preview is already + // ready: the buffered flush `mount()` triggers for edits made mid-create is not + // an HMR round trip, there is no preview yet for it to refresh. + if (this.didReady && batch.length > 0) this.lastEditFlushDispatchedAt = performance.now(); for (const [path, contents] of batch) { // Re-check per iteration: a dispose() during an earlier await must stop // the rest of the batch — writes to a torn-down session are pointless @@ -1072,6 +1199,10 @@ export class ContainerRuntime implements DemoRuntime { this.sessionId = null; this.readyCbs.clear(); this.errorCbs.clear(); + this.sessionStartCbs.clear(); + this.hmrCbs.clear(); + this.lastEditFlushDispatchedAt = null; + this.reloadInFlight = false; if (id) this.deleteSession(id); } } diff --git a/runner/packages/runtime/src/index.ts b/runner/packages/runtime/src/index.ts index 8d50faf108..e1bcc93071 100644 --- a/runner/packages/runtime/src/index.ts +++ b/runner/packages/runtime/src/index.ts @@ -9,6 +9,11 @@ export type { DemoRuntime, HandsontableVersionRef, WriteFileOptions, + SandpackCompileTimingEvent, + SandpackCompileErrorEvent, + SandpackBundlerUnreachableEvent, + SessionStartTimingEvent, + HmrRoundtripEvent, } from "./types.js"; export { diff --git a/runner/packages/runtime/src/monitor.ts b/runner/packages/runtime/src/monitor.ts index e7b26e496d..827a0f3a4f 100644 --- a/runner/packages/runtime/src/monitor.ts +++ b/runner/packages/runtime/src/monitor.ts @@ -13,12 +13,17 @@ // into workers/api — a second copy is a second set of caps to keep in sync. import { injectedScriptTag, insertInjectedTag } from "./inject-html.js"; +// ADR §C.5, contract §9: imported from the leaf modules directly, never from +// `./telemetry/index.js` — `scrub.ts` and `fingerprint.ts` already import +// `../monitor.js`, so a barrel import here would be a cycle. +import { LITE_PAYLOAD_MAX_BYTES, type LiteSurface } from "./telemetry/lite.js"; +import type { Framework, HtMajor } from "./telemetry/attrs.js"; /** The `postMessage` discriminator. Also the injection idempotency marker. */ export const MONITOR_MESSAGE_TYPE = "hot-runner-monitor"; /** - * Hard ceiling on relayed events per page load. + * Hard ceiling on relayed events per page load (per run on Tier 1, see `MONITOR_RESET`). * * The kill switch is build-time (see docs/run-and-deploy.md), so turning this * feature off costs a deploy. That makes the in-page ceiling the only brake that @@ -27,6 +32,11 @@ export const MONITOR_MESSAGE_TYPE = "hot-runner-monitor"; */ export const MONITOR_EVENT_CEILING = 20; +/** What the Tier-1 runtime posts into the preview before each dispatched run + * (`{ type: MONITOR_MESSAGE_TYPE, reset: MONITOR_RESET }`): the reporter's error + * budget and dedupe are per run, because the preview document outlives its runs. */ +export const MONITOR_RESET = "run"; + /** * Ceiling on relayed `console-warn` events per page load, counted separately from * `MONITOR_EVENT_CEILING` (DEV-2539). @@ -128,9 +138,13 @@ export const PREVIEW_HOST_PLACEHOLDER = "<preview>"; * This is the parent's backstop. The reporter redacts its own `location.host` before * sending, which is the precise version; this catches whatever crossed the boundary * anyway, including a payload from a demo that never ran the reporter. + * + * The label is bounded to `{1,63}` (RFC 1035 §2.3.4), so a suffix-less input + * cannot backtrack quadratically; `pipeline/o11y-redos.test.mjs` pins the + * timing. Same fix class as `text-scrub.ts`'s `EMAIL_PATTERN`/`USER_AGENT_PATTERN`. */ export function redactPreviewHosts(value: string): string { - return value.replace(/\b[a-z0-9-]+\.demos\.handsontable\.com\b/gi, PREVIEW_HOST_PLACEHOLDER); + return value.replace(/\b[a-z0-9-]{1,63}\.demos\.handsontable\.com\b/gi, PREVIEW_HOST_PLACEHOLDER); } /** @@ -259,7 +273,7 @@ export function monitorDedupeKey(kind: string, message: string, stack?: string): } /** - * A relay budget: the same ceiling and dedupe the in-page reporter applies, counted + * A relay budget: the in-page reporter's ceiling and dedupe, counted per page load * somewhere the demo cannot reach. * * The reporter's copy is not a cap. It runs *inside* the preview, alongside code @@ -346,9 +360,18 @@ export function createMonitorBudget(ceiling: number = MONITOR_EVENT_CEILING): { * which still needs to parse in Safari <16.4. * * Used for the Sentry fingerprint, not for the message the issue displays. + * + * Bounded to {@link NORMALIZE_MESSAGE_INPUT_MAX} chars before any pass below + * runs — the result is sliced to 200 chars anyway (last line), so nothing + * past a few thousand input characters can survive into the output; + * truncating first bounds the cost of every pass on a caller-controlled + * message. */ +const NORMALIZE_MESSAGE_INPUT_MAX = 4096; + export function normalizeMonitorMessage(message: string): string { - return message + const bounded = message.length > NORMALIZE_MESSAGE_INPUT_MAX ? message.slice(0, NORMALIZE_MESSAGE_INPUT_MAX) : message; + return bounded .replace(/https?:\/\/\S+/g, "<url>") .replace(/\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?/g, "<ts>") .replace(/["'`][^"'`]*["'`]/g, "<str>") @@ -356,7 +379,13 @@ export function normalizeMonitorMessage(message: string): string { // (`l`, `li`, `lic`, … `licenseKey` `is not defined`). Must precede the // number rule below (see the doc comment) and follow the quoted-string // rule above. - .replace(/[A-Za-z_$][\w$]*(?:\.[\w$]+)*(?= is not defined\b)/g, "<ident>") + // + // Both quantifiers are bounded ({0,256}/{1,32} segments): unbounded, a + // long identifier-shaped run with no trailing " is not defined" could + // backtrack quadratically (O(n²); measured via the real `fingerprint()`: + // 40k chars ~1.9s, 80k ~7.7s). No real identifier or dotted path is + // anywhere near that size. + .replace(/[A-Za-z_$][\w$]{0,256}(?:\.[\w$]{1,256}){0,32}(?= is not defined\b)/g, "<ident>") // DEV-2853 rule 2 — the partial locale in a RangeError from Intl, e.g. // `Invalid language tag: zh-c` ladders alongside `zh-`, `z`, and the // empty tail `Invalid language tag: `. `[ \t]*`, not `\s*`: `\s` matches @@ -651,6 +680,21 @@ export const REPORTER_SOURCE = `(function () { }); } catch (e) { /* ignore */ } + // A Tier-1 document is re-evaluated in place on every compile, so a typed line's + // prefix runs would otherwise spend the whole budget before the finished line throws. + // Only a reset from the parent is honoured (a demo can bypass the reporter anyway, so + // the parent's budget is the cap); the warning budget stays per page (breadcrumb trail). + try { + window.addEventListener("message", function (event) { + try { + var data = event.data; + if (event.source !== parent || !data || data.type !== TYPE || data.reset !== ${JSON.stringify(MONITOR_RESET)}) return; + used = 0; + for (var k in seen) if (k.indexOf("console-warn|") !== 0) delete seen[k]; + } catch (e) { /* ignore */ } + }); + } catch (e) { /* ignore */ } + try { var origError = console.error; var origWarn = console.warn; @@ -842,3 +886,169 @@ export function injectReporter(files: Record<string, string>, entryPath: string) : REPORTER_MODULE_LINE + "\n" + source; return { ...files, [entryPath]: injected }; } + +// ---- The lite beacon — standalone mode for `/d` and `/embed` ------------ +// ADR §C.5, contract §9. A separate reporter from `REPORTER_SOURCE` above, +// injected only at the `share.ts` serve seam — standalone by construction, +// no `postMessage`-to-parent transport needed. Sends only +// `error`/`unhandledrejection` and four sampled web vitals via +// `navigator.sendBeacon` to same-origin `/telemetry/lite`. + +/** Same-origin beacon target (contract §9). */ +export const LITE_ENDPOINT = "/telemetry/lite"; + +/** The injection idempotency marker — distinct from `MONITOR_MESSAGE_TYPE`. + * Deliberately the same string as the reporter's own double-injection guard + * property (`window.__hotLiteMonitor`) below, so it costs no extra bytes. */ +export const LITE_REPORTER_MARKER = "__hotLiteMonitor"; + +/** §9: "Vitals are sampled at 10% per page view, decided once per page." */ +export const LITE_VITALS_SAMPLE_RATE = 0.1; + +/** Client-side truncation caps, in **UTF-8 bytes** — tighter than the + * contract's own per-field ceilings, since a maxed-out stack alone already + * exceeds `LITE_PAYLOAD_MAX_BYTES` (2048). Bytes, not characters: `.length` + * counts UTF-16 code units, so a char-count cap could let a non-ASCII + * payload exceed the byte budget. */ +export const LITE_CLIENT_NAME_MAX_BYTES = 100; +export const LITE_CLIENT_MESSAGE_MAX = 300; +export const LITE_CLIENT_STACK_MAX = 300; + +/** + * Size budget for the *injected script itself* — distinct from + * `LITE_PAYLOAD_MAX_BYTES`, which bounds one beacon body. Target was under + * 2 KB; measured (`pipeline/lite-beacon.test.mjs`) at ~2.8 KB for a + * realistic config after cutting every inline comment and non-essential + * whitespace. Set from the measured size, with headroom for a longer + * `demo`/`fw` string. + */ +export const LITE_REPORTER_MAX_BYTES = 3072; + +/** Baked into the injected script at the `share.ts` serve seam — one build's + * worth of context the client cannot otherwise know (its own demo id, pinned + * Handsontable major, and framework). */ +export interface LiteReporterConfig { + surface: LiteSurface; + demo: string; + ht: HtMajor; + fw: Framework; +} + +/** Defence in depth for embedding `config`'s (allow-listed, but not worth + * trusting blindly) strings inside an inline `<script>` body: a literal + * `</script` in the JSON would otherwise close the tag early. None of §9's + * `demo`/`ht`/`fw` values can contain this today (a `shortId()`, a closed + * `HT_MAJORS` member, a `config/frameworks.json` key) — this is a backstop + * against that staying true, not a defence this reporter currently needs. */ +function escapeScriptClose(source: string): string { + return source.replace(/<\/(script)/gi, "<\\/$1"); +} + +/** + * The standalone reporter, as ES5 source — parsed and executed by + * `pipeline/lite-beacon.test.mjs` against a fake DOM. Written with no inline + * comments (the shipped script has its own byte budget, + * {@link LITE_REPORTER_MAX_BYTES}). The four vitals are intentional + * approximations, not the spec metrics (e.g. LCP is the LAST candidate + * before hide, not the first; INP is the single longest `event` duration, + * not a 98th-percentile grouping) — see the test file for the exact shape + * each measures. + * + * `bt(s,n)` cuts to `n` chars first, THEN runs the byte loop, avoiding a + * per-character re-encode that would be quadratic for a huge message + * (measured: 10k chars ~100ms, 50k ~2.4s). + */ +function reporterSource(config: LiteReporterConfig): string { + return `(function(){ +try{if(window.__hotLiteMonitor)return;window.__hotLiteMonitor=true;}catch(e){return;} +var EP=${JSON.stringify(LITE_ENDPOINT)},SURF=${JSON.stringify(config.surface)},DEMO=${JSON.stringify(config.demo)},HTM=${JSON.stringify(config.ht)},FWK=${JSON.stringify(config.fw)}; +var CEIL=${MONITOR_EVENT_CEILING},NMAX=${LITE_CLIENT_NAME_MAX_BYTES},MMAX=${LITE_CLIENT_MESSAGE_MAX},SMAX=${LITE_CLIENT_STACK_MAX},PMAX=${LITE_PAYLOAD_MAX_BYTES},RATE=${LITE_VITALS_SAMPLE_RATE}; +var used=0,sent={}; +function bl(s){try{return unescape(encodeURIComponent(s)).length;}catch(e){return 1e9;}} +function bt(s,n){if(s.length>n)s=s.slice(0,n);while(bl(s)>n)s=s.slice(0,-1);return s;} +function dv(){var u="";try{u=(navigator&&navigator.userAgent)||"";}catch(e){} +return /ipad|tablet|playbook|silk/i.test(u)?"tablet":/mobi|iphone|ipod|android.*mobile|windows phone/i.test(u)?"mobile":"desktop";} +var DEV=dv(); +function bc(t,f){try{ +var p={v:1,t:t,s:SURF,demo:DEMO,ht:HTM,fw:FWK,dev:DEV,ts:Date.now(),id:Math.random().toString(36).slice(2,10)}; +for(var k in f)p[k]=f[k]; +var j=JSON.stringify(p); +if(bl(j)>PMAX)return; +if(navigator&&typeof navigator.sendBeacon==="function")navigator.sendBeacon(EP,j); +}catch(e){}} +function se(n,m,st){try{ +if(used>=CEIL)return; +used+=1; +var f={n:bt(n||"Error",NMAX),m:bt(m||"unknown error",MMAX),val:null}; +if(st)f.st=bt(st,SMAX); +bc("err",f); +}catch(e){}} +function sv(n,val){try{ +if(sent[n])return; +if(typeof val!=="number"||!isFinite(val))return; +sent[n]=true; +bc("vital",{n:n,val:val}); +}catch(e){}} +try{ +window.addEventListener("error",function(ev){try{ +if(!ev||(!ev.error&&ev.target&&ev.target!==window))return; +var er=ev.error; +se((er&&er.name)||"Error",(er&&er.message)||(ev&&ev.message)||"unknown error",er&&er.stack); +}catch(e){}},true); +window.addEventListener("unhandledrejection",function(ev){try{ +var r=ev&&ev.reason; +se((r&&r.name)||"UnhandledRejection",r&&r.message?r.message:String(r),r&&r.stack); +}catch(e){}}); +}catch(e){} +var smp=false; +try{smp=Math.random()<RATE;}catch(e){} +if(smp){ +var lc=null,cls=0,inp=0,rep=false; +var ob=function(t,cb,dt){try{ +var o=new PerformanceObserver(cb),op={type:t,buffered:true}; +if(dt)op.durationThreshold=dt; +o.observe(op); +}catch(e){}}; +ob("largest-contentful-paint",function(l){var es=l.getEntries();if(es.length)lc=es[es.length-1];}); +ob("layout-shift",function(l){var es=l.getEntries();for(var i=0;i<es.length;i++){if(!es[i].hadRecentInput)cls+=es[i].value||0;}}); +ob("event",function(l){var es=l.getEntries();for(var i=0;i<es.length;i++){var en=es[i];if(en.interactionId&&en.interactionId>0&&en.duration>inp)inp=en.duration;}},40); +var rp=function(){ +if(rep)return; +rep=true; +try{if(lc)sv("LCP",lc.renderTime||lc.loadTime||0);}catch(e){} +sv("CLS",cls); +if(inp>0)sv("INP",inp); +try{ +var nv=performance&&performance.getEntriesByType&&performance.getEntriesByType("navigation")[0]; +if(nv&&typeof nv.responseStart==="number")sv("TTFB",nv.responseStart); +}catch(e){} +}; +try{ +document.addEventListener("visibilitychange",function(){try{if(document.visibilityState==="hidden")rp();}catch(e){}}); +window.addEventListener("pagehide",rp); +}catch(e){} +} +})(); +`; +} + +/** True when `html` already carries the lite reporter (`LITE_REPORTER_MARKER` + * survives the JSON-escaping of the source, same as `MONITOR_MESSAGE_TYPE` + * does for the framed reporter — see `alreadyInjected` above). */ +function liteAlreadyInjected(html: string): boolean { + return html.indexOf(LITE_REPORTER_MARKER) !== -1; +} + +/** + * Insert the standalone lite reporter into a `/d`/`/embed` document, exactly + * where `injectReporterIntoHtml` inserts the framed one (`insertInjectedTag`) + * and with the same DEV-2580 self-removing tag (`injectedScriptTag`) — the + * same Remix hydration constraint applies here: a `/d`/`/embed` build can be + * any of the same SSR frameworks. + * + * Idempotent: returns `html` unchanged when already injected. + */ +export function injectLiteReporterIntoHtml(html: string, config: LiteReporterConfig): string { + if (liteAlreadyInjected(html)) return html; + return insertInjectedTag(html, injectedScriptTag(escapeScriptClose(reporterSource(config)))); +} diff --git a/runner/packages/runtime/src/sandpack.ts b/runner/packages/runtime/src/sandpack.ts index c86507881e..776886528d 100644 --- a/runner/packages/runtime/src/sandpack.ts +++ b/runner/packages/runtime/src/sandpack.ts @@ -15,13 +15,27 @@ import type { DemoRuntime, FilesMap, HandsontableVersionRef, + SandpackBundlerUnreachableEvent, + SandpackCompileErrorEvent, + SandpackCompileTimingEvent, WriteFileOptions, } from "./types.js"; -import { isCompilerUnavailable, transpileFilesForParcel } from "./transpile.js"; +// Re-exported so existing `@handsontable/demo-runtime/sandpack` importers +// (`apps/authoring/src/telemetry/metrics.ts`) keep working — the interfaces +// themselves live in `types.ts`, so `DemoRuntime` can name the hook methods +// without a circular import. +export type { + SandpackBundlerUnreachableEvent, + SandpackCompileErrorEvent, + SandpackCompileTimingEvent, +} from "./types.js"; +import { isCompilerUnavailable, isTranspileFailure, transpileFilesForParcel } from "./transpile.js"; import { applyDepShims } from "./dep-shims.js"; import { HTML_ENTRY_ENVS, resolveSandboxEntry, toParcelEntry } from "./sandbox-entry.js"; import { MONITOR_COMPILE_MESSAGE_MAX, + MONITOR_MESSAGE_TYPE, + MONITOR_RESET, REPORTER_MODULE_LINE, injectReporter, redactPreviewHosts, @@ -189,6 +203,17 @@ export class SandpackEvaluationError extends Error { } } +// Observability contract §5 timing hooks: `SandpackCompileTimingEvent`, +// `SandpackCompileErrorEvent`, `SandpackBundlerUnreachableEvent` and the +// `onCompileTiming`/`onCompileError`/`onBundlerUnreachable` methods below are +// declared on `DemoRuntime` itself (`types.ts`), as OPTIONAL members — this +// module implements them, never imports `@handsontable/demo-runtime/telemetry`, +// and `apps/authoring/src/telemetry/metrics.ts#wireRuntimeMetrics` is what +// turns the callbacks into +// `sandpack.compile_ms`/`sandpack.compile_error`/`sandpack.bundler_unreachable` +// points against an injected `Telemetry`, through `runtime.onX?.(cb)` — no +// cast to the concrete class needed at the call site. + const COMPILE_ERROR_FALLBACK = "Sandpack compile error"; /** Inline source maps the bundler echoes back inside a compile message. A @@ -299,8 +324,50 @@ export class SandpackRuntime implements DemoRuntime { * so a mount still in flight when we are disposed would resurrect a torn-down * preview after the caller had already blanked it. */ private disposed = false; - /** Our claim on the iframe, registered in `mount()` before the first await. */ - private claim: object | null = null; + + // ---- Timing hooks --------------------------------------------------- + private readonly compileTimingCbs = new Set<(e: SandpackCompileTimingEvent) => void>(); + private readonly compileErrorCbs = new Set<(e: SandpackCompileErrorEvent) => void>(); + private readonly bundlerUnreachableCbs = new Set<(e: SandpackBundlerUnreachableEvent) => void>(); + private readonly pushOutcomeCbs = new Set<(outcome: "rerun" | "unchanged") => void>(); + /** Pushes dispatched to the bundler whose `start` has not arrived yet. */ + private pushesAwaitingStart = 0; + /** When the compile currently in flight was dispatched to the bundler — either + * `loadSandpackClient`'s initial compile (mount) or `updateSandbox` (an edit or + * `reload()`). Cleared once the terminal message for it arrives. Only ever one + * compile is in flight at a time: `pushUpdate`'s own sequence guard means a + * superseded push never reaches `updateSandbox`, and `mount()` is called once. */ + private compileDispatchedAt: number | null = null; + + /** Timing for every dispatched compile — the initial mount and every later push — + * resolved once (`ok` on a clean `done`, `error` on a `SandpackCompileError`). Fires + * once per real compile, never for a `sameFiles` no-op skip (nothing is dispatched, + * so nothing to time) and never twice for one dispatch. */ + onCompileTiming(cb: (e: SandpackCompileTimingEvent) => void): void { + this.compileTimingCbs.add(cb); + } + /** A compile diagnostic (`sandpack.compile_error`, §5) — never the evaluation-error + * sibling, which is a runtime throw already reported elsewhere. */ + onCompileError(cb: (e: SandpackCompileErrorEvent) => void): void { + this.compileErrorCbs.add(cb); + } + /** The hosted bundler's connection itself failed (§5 `sandpack.bundler_unreachable`) — + * see the interface doc comment for what this covers. */ + onBundlerUnreachable(cb: (e: SandpackBundlerUnreachableEvent) => void): void { + this.bundlerUnreachableCbs.add(cb); + } + /** See the interface doc. `unchanged` fires for the newest push only, never for a + * failed transpile; `rerun` fires when the bundler starts a run. */ + onPushOutcome(cb: (outcome: "rerun" | "unchanged") => void): void { + this.pushOutcomeCbs.add(cb); + } + + private resolveCompileTiming(outcome: "ok" | "error"): void { + if (this.compileDispatchedAt === null) return; + const durationMs = Math.round(performance.now() - this.compileDispatchedAt); + this.compileDispatchedAt = null; + for (const cb of this.compileTimingCbs) cb({ durationMs, outcome }); + } constructor(entry: CatalogEntry, opts: SandpackRuntimeOptions) { if (entry.engine !== "sandpack") { @@ -528,11 +595,40 @@ export class SandpackRuntime implements DemoRuntime { // Claim the iframe before the first await, so a successor mounting on the same frame // takes ownership synchronously and this instance can tell it has been superseded. const claim = {}; - this.claim = claim; IFRAME_OWNER.set(this.opts.iframe, claim); - const setup = await this.buildSetup(files); - const client = await loadSandpackClient(this.opts.iframe, setup, this.clientOptions()); + let setup: SandboxSetup; + try { + setup = await this.buildSetup(files); + } catch (err) { + // A demo whose source does not parse at mount — a saved, shared or + // `?payload=` demo, or a remount of a broken workspace — is a compile + // error from the very first run, and counts at once (no edit burst to + // collapse). Reported, then rethrown unchanged: the mount still rejects + // exactly as before, so the error card, `preview.ready_ms + // outcome=error` and the Sentry capture downstream (`tier1Report`) see + // the same error they always did. + if (isTranspileFailure(err)) this.reportTranspileFailure(err); + throw err; + } + // The dispatch clock for the initial compile (§5 `sandpack.compile_ms`). Started + // right before `loadSandpackClient`, which both connects to the bundler AND runs + // the first compile — `buildSetup` above is our own transpile/injection work, not + // the bundler's, and must stay outside the measured window. + const dispatchedAt = performance.now(); + this.compileDispatchedAt = dispatchedAt; + let client: SandpackClientInstance; + try { + client = await loadSandpackClient(this.opts.iframe, setup, this.clientOptions()); + } catch (err) { + // The client never connected — distinct from a `SandpackCompileError`/ + // `SandpackEvaluationError`, both of which only arrive over `onMessage` once a + // client exists. See `SandpackBundlerUnreachableEvent`. + if (this.compileDispatchedAt === dispatchedAt) this.compileDispatchedAt = null; + const durationMs = Math.round(performance.now() - dispatchedAt); + for (const cb of this.bundlerUnreachableCbs) cb({ durationMs }); + throw err; + } // Both awaits above can outlive a `dispose()`. `loadSandpackClient` has by now pointed // the iframe at the bundler origin, so returning quietly is not enough — undo it, or a @@ -560,9 +656,18 @@ export class SandpackRuntime implements DemoRuntime { payload?: { frames?: unknown }; }; switch (m.type) { + // `rerun` at the bundler's `start`, not at dispatch: the bundler runs one compile at + // a time, so what the previous run relays still arrives between the two. Only a + // pushed compile's start counts; the mount's own compile is not an edit's run. + case "start": + if (this.pushesAwaitingStart === 0) break; + this.pushesAwaitingStart -= 1; + for (const cb of this.pushOutcomeCbs) cb("rerun"); + break; case "done": // (`compilatonError` is misspelled in the upstream payload. Leave it.) - if (m.compilatonError) return; // error surfaced via its own message + if (m.compilatonError) return; // error surfaced via its own message; see "show-error" + this.resolveCompileTiming("ok"); this.emitReady(); break; case "action": @@ -588,6 +693,14 @@ export class SandpackRuntime implements DemoRuntime { const frames = m.payload?.frames; const evaluated = Array.isArray(frames) && frames.length > 0; const message = boundCompileMessage(m.message); + // Only a real compile diagnostic (no frames — the module never evaluated) + // resolves the compile clock and reports §5 `sandpack.compile_error`. An + // evaluation error's compile already reached "done" (`ok`) — the module ran + // and threw afterwards, a runtime fault, not a compile error. + if (!evaluated) { + this.resolveCompileTiming("error"); + for (const cb of this.compileErrorCbs) cb({ message, origin: "bundler" }); + } this.emitError( evaluated ? new SandpackEvaluationError(message) : new SandpackCompileError(message), ); @@ -744,7 +857,10 @@ export class SandpackRuntime implements DemoRuntime { // // `reload()` passes `force`, and its stamp guarantees a diff, so the refresh // button still re-runs the sandbox rather than being skipped here. - if (!opts.force && sameFiles(candidate, this.published)) return; + if (!opts.force && sameFiles(candidate, this.published)) { + for (const cb of this.pushOutcomeCbs) cb("unchanged"); + return; + } // Recorded *after* the push, never before. `setupFrom` throws when the resolved // entry is transiently missing (mid-rename, the DEV-2130 guard), and a `published` // set ahead of that throw would claim the bundler holds a sandbox it never @@ -752,11 +868,25 @@ export class SandpackRuntime implements DemoRuntime { // byte-identical compile this skip exists to prevent — the blank preview, back // again, on the rename path. const setup = this.setupFrom(candidate); + // Dispatch clock for this compile (§5 `sandpack.compile_ms`) — right before the + // bundler call, so `setupFrom`'s own DEV-2130 throw (caught below, not a compile + // dispatch at all) never starts a clock nothing will stop. + this.compileDispatchedAt = performance.now(); + this.resetMonitorBudget(); this.client.updateSandbox(setup, false); this.published = candidate; + this.pushesAwaitingStart += 1; }) .catch((cause: unknown) => { - /* mid-edit parse error — the user is still typing. + /* mid-edit parse error — the user is still typing. Nothing reaches the bundler and + * the last good render stays on screen (no error card per keystroke). + * + * It is still the preview's compile error, though, and the only place it exists: + * reported to `onCompileError`, and only for the newest + * push — a superseded keystroke's failure is already typed past, and reporting it + * would put a stale diagnostic into the edit burst the authoring app collapses + * (`demoEventCollapse.ts`), which counts once per burst. `emitError` is NOT + * called: the card and the Sentry capture stay exactly as they were. * * One exception (DEV-2569): the compiler chunk itself failing to load is not the * visitor's half-typed code, and swallowing it here left a stranded tab silently @@ -764,12 +894,36 @@ export class SandpackRuntime implements DemoRuntime { * most likely to be on. Emitted once: `loadBabel` has latched by now, so every * later keystroke arrives here with the same terminal error, and the card is * already showing it. */ + if (isTranspileFailure(cause)) { + if (this.client && seq === this.updateSeq) this.reportTranspileFailure(cause); + return; + } if (this.compilerFailureEmitted || !isCompilerUnavailable(cause)) return; this.compilerFailureEmitted = true; this.emitError(cause as Error); }); } + /** Re-arm the in-preview reporter for the run about to be dispatched. Posted to the + * same window as the compile, so it is delivered first. */ + private resetMonitorBudget(): void { + if (!this.opts.monitor) return; + try { + this.opts.iframe.contentWindow?.postMessage({ type: MONITOR_MESSAGE_TYPE, reset: MONITOR_RESET }, "*"); + } catch { + /* a detached frame: its next document starts with a fresh budget anyway */ + } + } + + /** §5 `sandpack.compile_error` for a parcel pre-transpile failure — the babel + * parse error the bundler never sees. Same event, and the same bounded message, as a + * bundler `show-error` diagnostic; no compile clock is involved (nothing was + * dispatched, so `sandpack.compile_ms` has nothing to time). */ + private reportTranspileFailure(cause: unknown): void { + const message = boundCompileMessage((cause as Error).message); + for (const cb of this.compileErrorCbs) cb({ message, origin: "transpile" }); + } + dispose(): void { this.disposed = true; try { @@ -780,6 +934,10 @@ export class SandpackRuntime implements DemoRuntime { this.client = null; this.readyCbs.clear(); this.errorCbs.clear(); + this.compileTimingCbs.clear(); + this.compileErrorCbs.clear(); + this.bundlerUnreachableCbs.clear(); + this.compileDispatchedAt = null; // No reload bookkeeping to drain: `reload()` settles on its own transpile, and // `pushUpdate` always settles (it catches), so a dispose mid-refresh cannot leave a // promise hanging. diff --git a/runner/packages/runtime/src/telemetry/attrs.ts b/runner/packages/runtime/src/telemetry/attrs.ts new file mode 100644 index 0000000000..2d6e6e44d1 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/attrs.ts @@ -0,0 +1,231 @@ +// Observability contract §3 — attribute keys, allowed values, Loki labels. +// Pure data/types, no DOM, no Cloudflare imports: imported by the authoring +// app, both Workers and `pipeline/` tests. `pipeline/telemetry-contract.test.mjs` +// parses `docs/observability-contract.md` §3 and fails if this file +// disagrees with it — edit both together. + +/** The four deployables that stamp `service.name` on every record they emit. */ +export const SERVICE_NAMES = [ + "demos-authoring", + "demos-api", + "demos-o11y", + "demos-embed", +] as const; +export type ServiceName = (typeof SERVICE_NAMES)[number]; + +export const ENVIRONMENTS = ["production", "local"] as const; +export type Environment = (typeof ENVIRONMENTS)[number]; + +export const SURFACES = [ + "authoring", + "share", + "embed", + "d", + "api", + "demo-runtime", + "o11y", +] as const; +export type Surface = (typeof SURFACES)[number]; + +export const TIERS = ["1", "2", "static", "none"] as const; +export type Tier = (typeof TIERS)[number]; + +/** `hot.ht_major` — every supported major, plus the `next` channel and `none` + * (a signal with no HT version attached). Closed set: the contract's `…` range + * is a shorthand for these five numbers. */ +export const HT_MAJORS = ["15", "16", "17", "18", "19", "next", "none"] as const; +export type HtMajor = (typeof HT_MAJORS)[number]; + +/** `hot.framework` on a stored record: the `config/frameworks.json` keys plus + * `none`; ingest stores anything else as {@link OTHER_ATTR_VALUE}. Kept in step + * with `frameworks.json` by `pipeline/telemetry-contract.test.mjs`. */ +export const KNOWN_FRAMEWORKS = [ + "blank", + "blank-ts", + "blank-react", + "example1", + "javascript", + "typescript", + "react", + "react-js", + "ant-design", + "mui", + "base-web", + "fluent-ui", + "vue", + "angular", + "next.js", + "next-shadcn.js", + "astro", + "nuxt", + "remix", + "none", +] as const; + +export type Framework = string; + +/** `hot.outcome` allowed values are per metric (§5, `metrics.ts`); a record no + * metric describes (a log, an exception, a plain event) carries only these. */ +export const RECORD_OUTCOMES = ["none"] as const; + +export type Outcome = string; + +/** What ingest stores for a `hot.framework`/`hot.outcome` outside its known set: + * both are Loki labels, so every distinct value multiplies the stream count + * (contract §3). */ +export const OTHER_ATTR_VALUE = "other"; + +/** Bounds a client-sent `fw` before it reaches the known-set mapping, so a + * multi-kilobyte value is refused outright (ADR §B.4). */ +export const OPEN_ATTR_VALUE_PATTERN = /^[a-z0-9][a-z0-9._-]{0,47}$/; + +export function isValidOpenAttrValue(value: string): boolean { + return OPEN_ATTR_VALUE_PATTERN.test(value); +} + +// ---- Resource attribute keys (OTLP), §3 --------------------------------------- + +export const ATTR_SERVICE_NAME = "service.name"; +export const ATTR_SERVICE_VERSION = "service.version"; +export const ATTR_DEPLOYMENT_ENVIRONMENT_NAME = "deployment.environment.name"; +export const ATTR_HOT_SURFACE = "hot.surface"; +export const ATTR_HOT_TIER = "hot.tier"; +export const ATTR_HOT_FRAMEWORK = "hot.framework"; +export const ATTR_HOT_HT_MAJOR = "hot.ht_major"; +export const ATTR_HOT_OUTCOME = "hot.outcome"; + +/** Structured metadata only (§3): never a Loki label, never an Analytics Engine + * index. `hot.kind` is the Faro item kind (`exception`, `log`, `event`, + * `measurement`). */ +export const ATTR_HOT_DEMO_ID = "hot.demo_id"; +export const ATTR_SESSION_ID = "session.id"; +export const ATTR_CF_RAY = "cf.ray"; +export const ATTR_HOT_KIND = "hot.kind"; + +/** §3's closed value set for `hot.kind` — the Faro item kind. `trace` is + * excluded: no trace is ever exported (ADR §C.4). */ +export const HOT_KINDS = ["exception", "log", "event", "measurement"] as const; + +export const STRUCTURED_METADATA_KEYS = [ + ATTR_HOT_DEMO_ID, + ATTR_SESSION_ID, + ATTR_CF_RAY, + ATTR_HOT_KIND, +] as const; + +/** §3 "Diagnostic tags": flat, non-dotted metadata on a handled-error or + * diagnostic-event report (§6), hoisted into `attributes` by + * `convert.ts#hoistAttributes`'s `STRUCTURED_KEY_SET` — never a Loki label. + * Each entry is a boolean flag, an opaque platform id, an enum/bucketed + * value, or a call site's own name — never user or request content. */ +export const DIAGNOSTIC_TAG_KEYS = [ + "handled", + "context", + "sentry_event_id", + "versions_fetch_attempts", + "versions_fetch_outcome", + "versions_fetch_elapsed_bucket", + "versions_fetch_online", + "api_base_origin", + "net_effective_type", +] as const; + +/** The AE-only browser attribute channel (ADR-0042): `HotAttrs` fields + * `toAePoint` accepts but §3 gives no resource/structured-metadata slot to. + * Read only by `browser-attrs.ts#readAeOnlyAttrs` from the raw wire body, + * BEFORE `scrubTelemetry` runs; `convert.ts#hoistAttributes` does not + * recognise these keys, so they never reach `resourceAttributes`/`attributes`. */ +export const ATTR_HOT_BUCKET = "hot.bucket"; +export const ATTR_HOT_REASON = "hot.reason"; +export const ATTR_HOT_FINGERPRINT = "hot.fingerprint"; +export const ATTR_HOT_METRIC_KIND = "hot.metric_kind"; +export const ATTR_HOT_REF = "hot.ref"; +export const ATTR_HOT_AREA = "hot.area"; + +export const AE_ONLY_ATTRIBUTE_KEYS = [ + ATTR_HOT_BUCKET, + ATTR_HOT_REASON, + ATTR_HOT_FINGERPRINT, + ATTR_HOT_METRIC_KIND, + ATTR_HOT_REF, + ATTR_HOT_AREA, +] as const; + +/** One row per §3 resource attribute: OTLP key, Loki label (`undefined` = + * none), and Analytics Engine blob slot (`AE_COLUMNS` in `metrics.ts` is + * the single source `sink.ts`/`inbox.ts` read). */ +export interface ResourceAttrDef { + key: string; + lokiLabel?: string; + aeSlot: string; +} + +export const RESOURCE_ATTRS: readonly ResourceAttrDef[] = [ + { key: ATTR_SERVICE_NAME, lokiLabel: "service_name", aeSlot: "blob1" }, + { key: ATTR_SERVICE_VERSION, aeSlot: "blob2" }, + { + key: ATTR_DEPLOYMENT_ENVIRONMENT_NAME, + lokiLabel: "deployment_environment_name", + aeSlot: "blob3", + }, + { key: ATTR_HOT_SURFACE, lokiLabel: "hot_surface", aeSlot: "blob4" }, + { key: ATTR_HOT_TIER, lokiLabel: "hot_tier", aeSlot: "blob5" }, + { key: ATTR_HOT_FRAMEWORK, lokiLabel: "hot_framework", aeSlot: "blob6" }, + { key: ATTR_HOT_HT_MAJOR, lokiLabel: "hot_ht_major", aeSlot: "blob7" }, + { key: ATTR_HOT_OUTCOME, lokiLabel: "hot_outcome", aeSlot: "blob8" }, +]; + +/** The Loki label list `otlp_config` promotes resource attributes to (ADR §B.4). */ +export const LOKI_LABELS: readonly string[] = RESOURCE_ATTRS.filter((a) => a.lokiLabel).map( + (a) => a.lokiLabel as string, +); + +/** §3's forbidden list enforced as an allowlist, not a denylist: every + * forbidden attribute (`url.full`, geo, an IP, a user-agent string) is + * simply absent, so a future one is dropped without a code change. */ +export const ALLOWED_ATTRIBUTE_KEYS: ReadonlySet<string> = new Set([ + ...RESOURCE_ATTRS.map((a) => a.key), + ...STRUCTURED_METADATA_KEYS, + ...DIAGNOSTIC_TAG_KEYS, + ...AE_ONLY_ATTRIBUTE_KEYS, +]); + +/** §4 blob15 `device`. Only `classify.ts#deviceOf` and the `web_vital` metric use + * it, but the set is shared so a bogus device class fails the same way an + * unlisted outcome does. */ +export const DEVICE_CLASSES = ["desktop", "mobile", "tablet"] as const; +export type DeviceClass = (typeof DEVICE_CLASSES)[number]; + +/** + * §6's `HotAttrs` — the attribute bag every `Telemetry.metric`/`.event`/`.error` + * call carries. Every field is optional: a given metric only reads the columns + * its §5 registry row lists (`metrics.ts#toAePoint`); passing more is ignored, + * passing `outcome`/`reason` for a metric that lists neither is a thrown error. + */ +export interface HotAttrs { + surface?: Surface; + tier?: Tier; + framework?: Framework; + ht_major?: HtMajor; + outcome?: Outcome; + reason?: string; + route_class?: string; + fingerprint?: string; + demo_id?: string; + model?: string; + provider?: string; + device?: DeviceClass; + bucket?: string; + kind?: string; + ref?: string; + area?: string; +} + +/** The three resource attrs stamped on every record (blob1–3), never part of a + * metric's own §5 "Blobs used" column because they are universal, not + * metric-specific. */ +export interface CommonResourceAttrs { + service_name: ServiceName; + service_version: string; + environment: Environment; +} diff --git a/runner/packages/runtime/src/telemetry/classify.ts b/runner/packages/runtime/src/telemetry/classify.ts new file mode 100644 index 0000000000..0144e472e2 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/classify.ts @@ -0,0 +1,35 @@ +// Bot filter and UA classifiers, moved here from +// `workers/api/src/analytics.ts` (DEV-2030) so both the anonymous-audience +// counters there and the o11y ingest gates (ADR §B.5 `BOT_RE` filter, §4 blob15 +// `device`) share one definition. `analytics.ts` imports these. + +export const BOT_RE = + /bot|crawler|spider|crawling|slurp|bingpreview|headlesschrome|lighthouse|curl\/|wget\/|python-requests|node-fetch|axios\/|monitoring|uptime|pingdom|semrush|ahrefs|facebookexternalhit|whatsapp|telegrambot|preview/i; + +export const isBot = (userAgent: string): boolean => BOT_RE.test(userAgent); + +/** Coarse device class (§4 blob15). Deliberately three buckets — anything finer + * starts to look like a fingerprint. */ +export function deviceOf(ua: string): string { + if (/ipad|tablet|playbook|silk/i.test(ua)) return "tablet"; + if (/mobi|iphone|ipod|windows phone/i.test(ua)) return "mobile"; + return "desktop"; +} + +export function browserOf(ua: string): string { + if (/edg\//i.test(ua)) return "edge"; + if (/opr\/|opera/i.test(ua)) return "opera"; + if (/chrome|crios|chromium/i.test(ua)) return "chrome"; + if (/firefox|fxios/i.test(ua)) return "firefox"; + if (/safari/i.test(ua)) return "safari"; + return "other"; +} + +export function osOf(ua: string): string { + if (/windows/i.test(ua)) return "windows"; + if (/iphone|ipad|ipod|ios/i.test(ua)) return "ios"; + if (/mac os x|macintosh/i.test(ua)) return "macos"; + if (/android/i.test(ua)) return "android"; + if (/linux|x11|cros/i.test(ua)) return "linux"; + return "other"; +} diff --git a/runner/packages/runtime/src/telemetry/convert.ts b/runner/packages/runtime/src/telemetry/convert.ts new file mode 100644 index 0000000000..16e376d63d --- /dev/null +++ b/runner/packages/runtime/src/telemetry/convert.ts @@ -0,0 +1,281 @@ +// Observability contract §6, §9 — Faro item → OTLP log record, and lite-beacon +// → OTLP log record. One converter for both. +// +// Scrub/convert order is NOT symmetric: `scrubTelemetry` accepts the Faro +// item shape or the OTLP record shape, never `LiteBeaconPayload` (§9). +// - Faro: `scrubTelemetry(item)` → `faroItemToRecord(scrubbedItem, …)` +// - Beacon: `beaconToRecord(payload, …)` → `scrubTelemetry(record)` +// `beaconToRecord`'s caller MUST run `scrubTelemetry` on its return value +// before storage, or an unscrubbed code frame / preview host in `m`/`st` +// reaches the inbox. + +import { + ATTR_DEPLOYMENT_ENVIRONMENT_NAME, + ATTR_HOT_DEMO_ID, + ATTR_HOT_FRAMEWORK, + ATTR_HOT_HT_MAJOR, + ATTR_HOT_KIND, + ATTR_HOT_OUTCOME, + ATTR_HOT_SURFACE, + ATTR_HOT_TIER, + ATTR_SERVICE_NAME, + ATTR_SERVICE_VERSION, + DIAGNOSTIC_TAG_KEYS, + ENVIRONMENTS, + HOT_KINDS, + HT_MAJORS, + KNOWN_FRAMEWORKS, + OTHER_ATTR_VALUE, + RECORD_OUTCOMES, + RESOURCE_ATTRS, + STRUCTURED_METADATA_KEYS, + SURFACES, + TIERS, + type Environment, + type ServiceName, +} from "./attrs.js"; +import { METRICS } from "./metrics.js"; +import type { NormalisedRecord } from "./inbox.js"; +import type { ScrubbableFaroItem } from "./scrub.js"; +import type { LiteBeaconPayload } from "./lite.js"; + +/** ADR §C.2: browser/beacon item timestamps are clamped to the envelope's + * `received_at` ± 5 minutes. */ +const CLAMP_WINDOW_MS = 5 * 60 * 1000; + +/** Clamps a candidate event-time (ms since epoch) to within + * `CLAMP_WINDOW_MS` of `receivedAtMs`; an absent or out-of-window candidate + * falls back to `receivedAtMs`, so no record ever reaches Loki without a + * timestamp (ADR §C.2). Also used by `normalise/otlp.ts`'s export path, at + * nanosecond precision. */ +export function clampTimestampMs(candidateMs: number | undefined, receivedAtMs: number): number { + if (candidateMs === undefined || !Number.isFinite(candidateMs)) return receivedAtMs; + return Math.abs(candidateMs - receivedAtMs) <= CLAMP_WINDOW_MS ? candidateMs : receivedAtMs; +} + +/** ms since epoch → OTLP `timeUnixNano` (decimal-string nanoseconds). Built with + * `BigInt`, never `ms * 1e6` — that exceeds `Number.MAX_SAFE_INTEGER` and + * silently loses precision. */ +export function msToUnixNano(ms: number): string { + return (BigInt(Math.round(ms)) * 1_000_000n).toString(); +} + +const RESOURCE_ATTR_KEYS = new Set(RESOURCE_ATTRS.map((a) => a.key)); +// `STRUCTURED_KEY_SET`: §3's structured-metadata keys plus +// `DIAGNOSTIC_TAG_KEYS` — everything `scrub.ts#allowlistAttributes` keeps +// that isn't a resource attribute; lands in `attributes`, never +// `resourceAttributes`. +const STRUCTURED_KEY_SET = new Set<string>([...STRUCTURED_METADATA_KEYS, ...DIAGNOSTIC_TAG_KEYS]); + +/** Split a merged, already-allowlisted attribute bag into OTLP resource + * attributes (`hot.surface`, `hot.tier`, … — ADR §B.2) and structured + * metadata (`hot.demo_id`, `session.id`, `cf.ray`, `hot.kind`, and the + * diagnostic tag keys — never a resource attribute, §3). */ +export function hoistAttributes( + merged: Record<string, string> | undefined, +): { resourceAttributes: Record<string, string>; attributes: Record<string, string> } { + const resourceAttributes: Record<string, string> = {}; + const attributes: Record<string, string> = {}; + for (const [key, value] of Object.entries(merged ?? {})) { + if (RESOURCE_ATTR_KEYS.has(key)) resourceAttributes[key] = value; + else if (STRUCTURED_KEY_SET.has(key)) attributes[key] = value; + } + return { resourceAttributes, attributes }; +} + +export interface ServiceIdentity { + name: ServiceName; + version: string; + environment: Environment; +} + +export interface ConvertOptions { + service: ServiceIdentity; + /** The envelope's arrival time, ms since epoch — the clamp anchor (§C.2). */ + receivedAtMs: number; +} + +function serviceResourceAttributes(service: ServiceIdentity): Record<string, string> { + return { + [ATTR_SERVICE_NAME]: service.name, + [ATTR_SERVICE_VERSION]: service.version, + [ATTR_DEPLOYMENT_ENVIRONMENT_NAME]: service.environment, + }; +} + +/** Closed-set resource attributes, checked at record level. A failing value is + * dropped, and `normalise/points.ts#withResourceAttrDefaults` fills its default. */ +const CLOSED_SET_BY_KEY: Readonly<Record<string, readonly string[]>> = { + [ATTR_DEPLOYMENT_ENVIRONMENT_NAME]: ENVIRONMENTS, + [ATTR_HOT_SURFACE]: SURFACES, + [ATTR_HOT_TIER]: TIERS, + [ATTR_HOT_HT_MAJOR]: HT_MAJORS, +}; + +/** The `hot.outcome` values `metric`'s §5 row allows, or {@link RECORD_OUTCOMES} + * for a record no outcome-carrying metric describes. */ +function outcomeSetFor(metric: string | undefined): readonly string[] { + if (metric === undefined || !Object.prototype.hasOwnProperty.call(METRICS, metric)) return RECORD_OUTCOMES; + return METRICS[metric as keyof typeof METRICS].values?.["outcome"] ?? RECORD_OUTCOMES; +} + +/** + * Record-level enforcement of §3's value rules on every Loki-label attribute. + * A closed-set value outside its enum is dropped. `hot.framework` and + * `hot.outcome` outside their known sets (`KNOWN_FRAMEWORKS`; `metric`'s + * outcomes) become `"other"`, so the label set stays bounded however many + * distinct values a client sends. `metric` is the item's metric name, if any. + */ +export function sanitizeResourceAttributes( + resourceAttributes: Record<string, string>, + metric?: string, +): Record<string, string> { + const out: Record<string, string> = {}; + for (const [key, value] of Object.entries(resourceAttributes)) { + const closedSet = CLOSED_SET_BY_KEY[key]; + if (closedSet) { + if (closedSet.includes(value)) out[key] = value; + continue; + } + if (key === ATTR_HOT_FRAMEWORK) { + out[key] = (KNOWN_FRAMEWORKS as readonly string[]).includes(value) ? value : OTHER_ATTR_VALUE; + continue; + } + if (key === ATTR_HOT_OUTCOME) { + out[key] = outcomeSetFor(metric).includes(value) ? value : OTHER_ATTR_VALUE; + continue; + } + out[key] = value; + } + return out; +} + +/** ADR §C.3, symbolication at drain: renders one stack frame + * in the standard V8 ` at <fn> (<file>:<line>:<col>)` shape — + * `workers/o11y/src/drain/symbolicate.ts` parses this exact text back out. + * A frame with no `filename` carries nothing a symbolicator could resolve + * or a human could read, so it is skipped rather than rendered as a bare + * `at <fn>` line (that would be ambiguous with a genuinely-anonymous, + * file-less frame, which does not occur in a browser stack). */ +export function formatStackFrame(frame: { filename?: string; function?: string; lineno?: number; colno?: number }): string | null { + if (!frame.filename) return null; + const fn = frame.function || "<anonymous>"; + // A `lineno < 1` (or non-finite) is not a real source position: + // `@jridgewell/trace-mapping#originalPositionFor` throws on it. + // `symbolicate.ts#resolveBody` guards this too; dropping only the position + // suffix keeps the existing "unresolvable, rendered as-is" shape. + const position = + typeof frame.lineno === "number" && + typeof frame.colno === "number" && + Number.isFinite(frame.lineno) && + Number.isFinite(frame.colno) && + frame.lineno >= 1 + ? `:${frame.lineno}:${frame.colno}` + : ""; + return ` at ${fn} (${frame.filename}${position})`; +} + +/** Stack frames are appended to the body as plain text (`body` is the one + * free-text field `NormalisedRecord` has). `symbolicate.ts` parses this + * exact format back out at drain and rewrites resolved lines in place, so + * this function's output must already be a faithful, minified stack. */ +function faroBody(item: ScrubbableFaroItem): string { + switch (item.type) { + case "exception": { + const value = item.payload.value ?? ""; + const head = item.payload.type ? `${item.payload.type}: ${value}` : value; + const frameLines = (item.payload.stacktrace?.frames ?? []) + .map(formatStackFrame) + .filter((line): line is string => line !== null); + return frameLines.length > 0 ? `${head}\n${frameLines.join("\n")}` : head; + } + case "log": + return item.payload.message ?? ""; + case "event": + return item.payload.name ?? ""; + case "measurement": + return JSON.stringify(item.payload.values ?? {}); + default: + return item.payload.message ?? item.payload.value ?? ""; + } +} + +function faroTimestampMs(item: ScrubbableFaroItem): number | undefined { + const ts = item.payload.timestamp; + if (typeof ts !== "string") return undefined; + const ms = Date.parse(ts); + return Number.isFinite(ms) ? ms : undefined; +} + +/** Faro item → normalised OTLP log record (§6). `item` must already be + * scrubbed. `hot.kind` is always `item.type`, overwriting anything the + * client's own context carried under that key. Throws if `item.type` is + * not one of `HOT_KINDS` (e.g. `"trace"`, never exported, ADR §C.4) — + * reachable from untrusted input via `POST /telemetry/collect`. */ +export function faroItemToRecord(item: ScrubbableFaroItem, options: ConvertOptions): NormalisedRecord { + if (!(HOT_KINDS as readonly string[]).includes(item.type)) { + throw new Error(`faroItemToRecord: not a valid hot.kind: ${JSON.stringify(item.type)}`); + } + + const merged = { ...(item.payload.context ?? {}), ...(item.payload.attributes ?? {}) }; + const { resourceAttributes, attributes } = hoistAttributes(merged); + attributes[ATTR_HOT_KIND] = item.type; + // Only a measurement carries a metric outcome, and it becomes an AE point, never + // a stored record. A stored record's `hot.outcome` is always the `none` + // default, which keeps the browser tenant's label tuples under the box's Loki + // stream limit (contract §3). + if (item.type !== "measurement") delete resourceAttributes[ATTR_HOT_OUTCOME]; + const metric = item.type === "measurement" ? item.payload.type : undefined; + + return { + body: faroBody(item), + timeUnixNano: msToUnixNano(clampTimestampMs(faroTimestampMs(item), options.receivedAtMs)), + // `serviceResourceAttributes` spreads LAST: a client-hoisted value under + // `service.name`/`service.version`/`deployment.environment.name` must + // never win over the route's own identity. `sanitizeResourceAttributes` + // closes the matching hole for the remaining `hot.*` keys. + resourceAttributes: { + ...sanitizeResourceAttributes(resourceAttributes, metric), + ...serviceResourceAttributes(options.service), + }, + attributes, + }; +} + +function beaconBody(payload: LiteBeaconPayload): string { + if (payload.t === "err") { + return payload.st ? `${payload.n}: ${payload.m}\n${payload.st}` : `${payload.n}: ${payload.m}`; + } + return `${payload.n}=${payload.val}`; +} + +/** §9's lite beacon → normalised OTLP log record, via "the same converter as + * Faro items" (§9) — same clamp, same resource-attribute shape. `hot.tier` is + * always `"static"`: the lite beacon only ever fires from `/d` and `/embed`, + * never a live editing session. */ +export function beaconToRecord(payload: LiteBeaconPayload, options: ConvertOptions): NormalisedRecord { + const attributes: Record<string, string> = { + [ATTR_HOT_DEMO_ID]: payload.demo, + [ATTR_HOT_KIND]: payload.t === "err" ? "exception" : "measurement", + }; + + // `s`/`ht` are already closed-set-checked by `isValidLitePayload`; `fw` is + // client-supplied free text, so it (and the whole bag, defensively) go + // through the same record-level check `faroItemToRecord` uses. + const rawResourceAttributes: Record<string, string> = { + [ATTR_HOT_SURFACE]: payload.s, + [ATTR_HOT_TIER]: "static", + [ATTR_HOT_FRAMEWORK]: payload.fw, + [ATTR_HOT_HT_MAJOR]: payload.ht, + }; + + return { + body: beaconBody(payload), + timeUnixNano: msToUnixNano(clampTimestampMs(payload.ts, options.receivedAtMs)), + resourceAttributes: { + ...sanitizeResourceAttributes(rawResourceAttributes), + ...serviceResourceAttributes(options.service), + }, + attributes, + }; +} diff --git a/runner/packages/runtime/src/telemetry/facade.ts b/runner/packages/runtime/src/telemetry/facade.ts new file mode 100644 index 0000000000..0aa5ec51a4 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/facade.ts @@ -0,0 +1,98 @@ +// Observability contract §6 — the browser facade. The app never calls Faro +// directly: `apps/authoring/src/telemetry/index.ts` exports `telemetry: +// Telemetry`, `noopTelemetry` until `initTelemetry()` runs, and a separate +// module implements the Faro-backed one. No Faro import here, so this file +// can be imported from `pipeline/` under plain Node. + +import type { HotAttrs } from "./attrs.js"; +import type { MetricName, MetricValues } from "./metrics.js"; + +/** Event names are open (`example.open` and friends, or any Faro `event`/`log` + * line) — never a closed set the way `MetricName` is. */ +export type EventName = string; + +export interface Telemetry { + metric(name: MetricName, values: MetricValues, attrs: HotAttrs): void; + event(name: EventName, attrs: HotAttrs & Record<string, string>): void; + /** A handled error (§5 `error.handled`) — never an uncaught one; those stay on + * `window.onerror`/`unhandledrejection`/`Sentry.ErrorBoundary` (ADR §E.1). */ + error(err: unknown, context: string, attrs?: HotAttrs): void; + /** The in-memory page-load id minted once at page load (§6, §3 `session.id`), + * identical across every call for the life of the page. */ + pageLoadId(): string; +} + +function mintPageLoadId(): string { + if (typeof crypto !== "undefined" && typeof crypto.randomUUID === "function") { + return crypto.randomUUID(); + } + // Defensive fallback only — every runtime this module ships to (evergreen + // browsers, workerd, Node 20+) has Web Crypto. + return `plid-${Date.now().toString(36)}-${Math.random().toString(36).slice(2)}`; +} + +/** + * The facade before `initTelemetry()` runs (or when reporting is gated off). + * Every call is a no-op except `pageLoadId()`, which still mints and returns a + * real, stable id — call sites that read it to tag `x-hot-session` before init + * has run must not see an empty string. + * + * Lazy on purpose: minting eagerly in a module-top-level IIFE calls + * `crypto.randomUUID()` at import time, which is disallowed "global scope" + * async/random I/O under workerd (`Uncaught Error: Disallowed operation + * called within global scope ... generating random values are not allowed + * within global scope`, thrown at Worker boot). `pageLoadId()` still returns + * the exact same id on every call after the first — only *when* the mint + * happens changes. + */ +let noopPageLoadId: string | undefined; +export const noopTelemetry: Telemetry = { + metric() {}, + event() {}, + error() {}, + pageLoadId: () => (noopPageLoadId ??= mintPageLoadId()), +}; + +export interface RecordedMetricCall { + name: MetricName; + values: MetricValues; + attrs: HotAttrs; +} +export interface RecordedEventCall { + name: EventName; + attrs: HotAttrs & Record<string, string>; +} +export interface RecordedErrorCall { + err: unknown; + context: string; + attrs?: HotAttrs; +} + +export interface RecordingTelemetry extends Telemetry { + readonly metrics: RecordedMetricCall[]; + readonly events: RecordedEventCall[]; + readonly errors: RecordedErrorCall[]; +} + +/** Test double: records every call instead of sending anything, so a test can + * assert on `.metrics`/`.events`/`.errors` directly. */ +export function recordingTelemetry(pageLoadId: string = mintPageLoadId()): RecordingTelemetry { + const metrics: RecordedMetricCall[] = []; + const events: RecordedEventCall[] = []; + const errors: RecordedErrorCall[] = []; + return { + metrics, + events, + errors, + metric(name, values, attrs) { + metrics.push({ name, values, attrs }); + }, + event(name, attrs) { + events.push({ name, attrs }); + }, + error(err, context, attrs) { + errors.push({ err, context, attrs }); + }, + pageLoadId: () => pageLoadId, + }; +} diff --git a/runner/packages/runtime/src/telemetry/fingerprint.ts b/runner/packages/runtime/src/telemetry/fingerprint.ts new file mode 100644 index 0000000000..6aa1bc2adc --- /dev/null +++ b/runner/packages/runtime/src/telemetry/fingerprint.ts @@ -0,0 +1,103 @@ +// Observability contract §7 — fingerprint. +// +// Synchronous and identical in browser and Worker (both run this same module), +// so a browser-computed and a Worker-computed fingerprint for the same message +// always agree — `pipeline/telemetry-fingerprint.test.mjs` checks that against a +// build of this module run through Node directly, not just self-consistency. + +import { normalizeMonitorMessage } from "../monitor.js"; +import type { Surface } from "./attrs.js"; + +const FNV_OFFSET_BASIS_64 = 0xcbf29ce484222325n; +const FNV_PRIME_64 = 0x100000001b3n; +const MASK_64 = 0xffffffffffffffffn; + +/** FNV-1a, 64-bit, over the UTF-8 bytes of `value`. Pinned to the published test + * vectors (byte encoding is UTF-8, the natural choice for a JS string): + * `""` → `cbf29ce484222325` (the bare offset basis), `"a"` → `af63dc4c8601ec8c`, + * `"foobar"` → `85944171f73967e8`. */ +function fnv1a64Hex(value: string): string { + let hash = FNV_OFFSET_BASIS_64; + const bytes = new TextEncoder().encode(value); + for (const byte of bytes) { + hash ^= BigInt(byte); + hash = (hash * FNV_PRIME_64) & MASK_64; + } + return hash.toString(16).padStart(16, "0"); +} + +// Every quantifier is capped: adjacent unbounded `[ \t]*` runs cost O(n²) on a +// line of spaces (250k spaces measured at ~30 s), and a real Babel gutter +// never pads past a few characters. +const GUTTER_LINE = /^[ \t]{0,32}>?[ \t]{0,32}\d{1,9}[ \t]{0,32}\|.*$/; +const CARET_LINE = /^[ \t]{0,32}\|[ \t]{0,4096}\^{1,4096}[ \t]{0,32}$/; + +/** + * Strips a Babel code-frame (`@babel/standalone`'s `codeFrameColumns`, see + * `transpile.ts`) out of a message, so authored source text does not survive + * into the fingerprint or (via `scrub.ts`) the inbox. Must run BEFORE + * `normalizeMonitorMessage`, which collapses newlines this shape depends on. + */ +export function stripCodeFrame(message: string): string { + return message + .split("\n") + .filter((line) => !GUTTER_LINE.test(line) && !CARET_LINE.test(line)) + .join("\n") + .replace(/\n{2,}/g, "\n") + .trim(); +} + +/** + * `<context>:<16 hex chars of FNV-1a 64 over the normalised message>` (§7). + * + * `context` is caller-chosen and not normalised — it is typically `hot.surface` + * or a metric name, kept short and already a controlled value, unlike `message`. + */ +export function fingerprint(context: string, message: string): string { + return `${context}:${fnv1a64Hex(fingerprintShape(message))}`; +} + +/** `normalizeMonitorMessage` keeps only its first 4096 characters anyway; cutting + * here too keeps `stripCodeFrame` off the rest of a client-sized message. */ +const FINGERPRINT_INPUT_MAX_CHARS = 4096; + +/** + * The normalised text `fingerprint()` hashes: code frame stripped, then + * `normalizeMonitorMessage`. Also the only message text a demo-runtime Faro + * record carries (contract §3, §6). Idempotent + * (`pipeline/demo-event-collapse.test.mjs`), so the record's own fingerprint + * agrees with the metric point's. + */ +export function fingerprintShape(message: string): string { + return normalizeMonitorMessage(stripCodeFrame(message.slice(0, FINGERPRINT_INPUT_MAX_CHARS))); +} + +/** + * §7: "Demo-runtime fingerprints never feed the new-fingerprint alert" — the + * exact first-seen registry (`fp:<fingerprint>` in `InboxWriter`, contract §8, + * ADR §F.3) is keyed by fingerprint, but skips writing/checking the key when + * this returns `false`. Authored-code keystroke ladders are expected, not a + * signal of a new defect. + */ +export function feedsNewFingerprintAlert(surface: Surface): boolean { + return surface !== "demo-runtime"; +} + +/** + * §7's exact wire shape (`<context>:<16 hex chars>`) — validates a + * client-supplied fingerprint before it is trusted verbatim, so an + * attacker's string cannot reach the `fp:` first-seen registry + * (`InboxWriter`) or an unescaped Slack alert line. Anchors on the LAST `:` + * to allow `context`'s own `:`-joined segments (e.g. + * `"docs-example-load:fetch"`). One shared pattern with + * `normalise/otlp.ts#apiFingerprintFeed` and `normalise/faro.ts#resolveFingerprint`. + * `context` is capped at {@link MAX_FINGERPRINT_CONTEXT_LENGTH}. + */ +const MAX_FINGERPRINT_CONTEXT_LENGTH = 128; +const FINGERPRINT_PATTERN = + /^[a-z][a-z0-9._-]*(?::[a-z0-9._-]+)*:[0-9a-f]{16}$/; + +export function isValidFingerprint(value: string): boolean { + if (value.length > MAX_FINGERPRINT_CONTEXT_LENGTH + 17) return false; // +1 `:` + 16 hex + return FINGERPRINT_PATTERN.test(value); +} diff --git a/runner/packages/runtime/src/telemetry/inbox.ts b/runner/packages/runtime/src/telemetry/inbox.ts new file mode 100644 index 0000000000..4cf5969154 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/inbox.ts @@ -0,0 +1,231 @@ +// Observability contract §8 — the inbox: OTLP `ResourceLogs` builders, NDJSON +// encode/decode, the `inbox/...` object key and the `InboxWriter` storage-key +// shapes. `InboxWriter` itself is the only writer; this module is just the +// shapes and pure string/JSON functions it and the drain share. + +export type Tenant = "browser" | "worker"; + +// ---- OTLP `ResourceLogs`, one log record each ----------------------------------- +// +// The OTLP JSON representation (protobuf's `google.protobuf.Struct`-free JSON +// mapping), narrowed to exactly what a normalised record needs: one resource, one +// scope, one log record. Structural, no `@opentelemetry/api` import. + +export interface OtlpAnyValue { + stringValue?: string; + intValue?: string; + doubleValue?: number; + boolValue?: boolean; +} + +export interface OtlpKeyValue { + key: string; + value: OtlpAnyValue; +} + +export interface OtlpLogRecord { + /** Nanoseconds since epoch, as a decimal string (OTLP JSON's `fixed64` mapping + * — a plain JS number loses precision past 2^53, so this is always built and + * read as a string, never `Number`). */ + timeUnixNano: string; + observedTimeUnixNano?: string; + severityText?: string; + body?: OtlpAnyValue; + attributes?: OtlpKeyValue[]; +} + +export interface OtlpScopeLogs { + logRecords: OtlpLogRecord[]; +} + +export interface OtlpResourceLogs { + resource: { attributes: OtlpKeyValue[] }; + scopeLogs: OtlpScopeLogs[]; +} + +function kv(key: string, value: string): OtlpKeyValue { + return { key, value: { stringValue: value } }; +} + +/** What `convert.ts` hands to `buildResourceLogs`: already scrubbed, already + * attribute-allowlisted, already timestamp-clamped. */ +export interface NormalisedRecord { + body: string; + /** Nanoseconds since epoch, decimal string (see `OtlpLogRecord.timeUnixNano`). */ + timeUnixNano: string; + /** Promoted to OTLP resource attributes — `service.*`, `deployment.*`, + * `hot.surface`/`tier`/`framework`/`ht_major`/`outcome` (§3). */ + resourceAttributes: Record<string, string>; + /** Structured metadata only (§3): `hot.demo_id`, `session.id`, `cf.ray`, + * `hot.kind`. Never promoted to a resource attribute. */ + attributes?: Record<string, string>; + severityText?: string; +} + +/** Build one OTLP `ResourceLogs` object wrapping exactly one log record — the + * shape one NDJSON line holds (§8: "Each NDJSON line is one OTLP `ResourceLogs` + * object"). */ +export function buildResourceLogs(record: NormalisedRecord): OtlpResourceLogs { + const logRecord: OtlpLogRecord = { + timeUnixNano: record.timeUnixNano, + body: { stringValue: record.body }, + }; + if (record.severityText !== undefined) logRecord.severityText = record.severityText; + const attrEntries = Object.entries(record.attributes ?? {}); + if (attrEntries.length > 0) logRecord.attributes = attrEntries.map(([k, v]) => kv(k, v)); + + return { + resource: { attributes: Object.entries(record.resourceAttributes).map(([k, v]) => kv(k, v)) }, + scopeLogs: [{ logRecords: [logRecord] }], + }; +} + +// ---- NDJSON ---------------------------------------------------------------- + +/** One `ResourceLogs` JSON object per line, `\n`-terminated (empty input → `""`, + * never a bare newline). */ +export function encodeNdjson(records: readonly OtlpResourceLogs[]): string { + if (records.length === 0) return ""; + return records.map((r) => JSON.stringify(r)).join("\n") + "\n"; +} + +/** Inverse of `encodeNdjson`. Blank lines (a trailing newline, or one a + * redelivery introduced) are skipped rather than failing the whole object. */ +export function decodeNdjson(text: string): OtlpResourceLogs[] { + return text + .split("\n") + .filter((line) => line.trim().length > 0) + .map((line) => JSON.parse(line) as OtlpResourceLogs); +} + +// ---- Inbox object key (§8) ------------------------------------------------------ + +const SEQ_DIGITS = 12; + +/** `inbox/<tenant>/<yyyy-mm-dd>/<hh>/<seq:012d>.ndjson.gz`. `date` is read with + * UTC getters — the pack alarm runs in a Worker (UTC), and a local-time build + * would put the last hour of a UTC day in tomorrow's prefix. */ +export function inboxKey(tenant: Tenant, date: Date, seq: number): string { + const yyyy = date.getUTCFullYear().toString().padStart(4, "0"); + const mm = (date.getUTCMonth() + 1).toString().padStart(2, "0"); + const dd = date.getUTCDate().toString().padStart(2, "0"); + const hh = date.getUTCHours().toString().padStart(2, "0"); + const seqStr = seq.toString().padStart(SEQ_DIGITS, "0"); + return `inbox/${tenant}/${yyyy}-${mm}-${dd}/${hh}/${seqStr}.ndjson.gz`; +} + +export interface ParsedInboxKey { + tenant: Tenant; + /** `yyyy-mm-dd`, UTC. */ + date: string; + /** `hh`, UTC, zero-padded. */ + hour: string; + seq: number; +} + +const INBOX_KEY_RE = + /^inbox\/(browser|worker)\/(\d{4}-\d{2}-\d{2})\/(\d{2})\/(\d{12})\.ndjson\.gz$/; + +export function parseInboxKey(key: string): ParsedInboxKey | null { + const m = INBOX_KEY_RE.exec(key); + if (!m || m[1] === undefined || m[2] === undefined || m[3] === undefined || m[4] === undefined) { + return null; + } + return { tenant: m[1] as Tenant, date: m[2], hour: m[3], seq: Number(m[4]) }; +} + +/** `state/wakes/<wakeId>/clean` (Loki bucket) — written by the box on a clean + * stop, read (existence only) by `InboxWriter` to resolve a wake's provisional + * keys (ADR §B.3). */ +export function cleanMarkerKey(wakeId: string): string { + return `state/wakes/${wakeId}/clean`; +} + +// ---- `InboxWriter` storage keys and value shapes (§8 table) --------------------- + +export const SEQ_STORAGE_KEY = "seq"; +export const DRAINS_PAUSED_STORAGE_KEY = "drainsPaused"; +export const HEARTBEAT_STORAGE_KEY = "heartbeat"; + +/** Digits `pendingRowStorageKey` zero-pads `n` to, so native ascending key + * order equals arrival order for a bounded `list({prefix, limit})` read. + * Matches the packed-object `<seq>` width (§8's `inboxKey`). */ +export const ROW_SEQ_DIGITS = 12; + +export function pendingRowStorageKey(n: number): string { + return `row:${n.toString().padStart(ROW_SEQ_DIGITS, "0")}`; +} +export function inboxKeyStorageKey(key: string): string { + return `key:${key}`; +} +/** A `key:<inbox key>` entry that reaches `committed` moves OUT of `key:` + * into `done:<inbox key>` (contract §8), so `key:` holds only the live set + * every drain/backlog/resolve read cares about. */ +export function doneKeyStorageKey(key: string): string { + return `done:${key}`; +} +export function fingerprintStorageKey(fp: string): string { + return `fp:${fp}`; +} +/** How many digits {@link fingerprintTimeIndexKey} zero-pads its epoch-ms + * component to — 15 covers every ms timestamp until the year 5138, far + * past any realistic operational lifetime for this key shape. */ +export const FPTS_TIMESTAMP_DIGITS = 15; +/** `fpts:<firstSeenMs, zero-padded>:<fingerprint>` — a time-ordered + * secondary index alongside `fp:<fp>`, so `newFingerprintsSince` can do a + * bounded range read instead of listing the entire `fp:` prefix. Deleted + * together with its `fp:<fp>` twin by `pruneFingerprintRegistry`. */ +export function fingerprintTimeIndexKey(firstSeenMs: number, fp: string): string { + return `fpts:${Math.max(0, Math.trunc(firstSeenMs)).toString().padStart(FPTS_TIMESTAMP_DIGITS, "0")}:${fp}`; +} +export function alertStorageKey(rule: string): string { + return `alert:${rule}`; +} +export function wakeStorageKey(wakeId: string): string { + return `wake:${wakeId}`; +} + +/** `key:<inbox key>` value (§8 ledger, ADR §B.3). */ +export type InboxKeyState = "written" | "committed" | `provisional:${string}` | `rejected:${string}`; + +/** `wake:<wakeId>` value. `over` is set once a newer wake starts or the + * container is observed not running (ADR §B.3); `InboxWriter.recordWake` + * (COMMON.md interface 1) is what marks every earlier wake `over: true`. */ +export interface WakeState { + startedAt: number; + reason: "backlog" | "visit"; + over: boolean; + /** Wake-to-ready time in ms (contract §5 `o11y.wake` `duration_ms`, + * "to ready"; exit criterion 6): from `GrafanaBox.wake()` minting this + * wake to its first successful `isReady()`. Written once by + * `InboxWriter.recordWakeReady`; absent while the box has not yet become + * ready, and forever for a wake that never did. */ + readyMs?: number; +} + +/** `alert:<rule>` value (ADR §F.3 — notify once on fire, once on resolve). */ +export interface AlertState { + state: "firing" | "resolved"; + since: number; + lastNotified: number; +} + +/** `heartbeat` value — read by the API worker's five-minute cron over the `API` + * service binding (ADR §F.3 stale-stack alert). */ +export interface Heartbeat { + lastCron: number; + lastIngest: number; +} + +/** Records over this size are dropped at ingest step 1 (ADR §B.2). */ +export const INBOX_RECORD_MAX_BYTES = 256 * 1024; +/** A `row:<n>` storage row holds at most this many bytes of pending records. */ +export const INBOX_ROW_MAX_BYTES = 1024 * 1024; +/** A drain request to Loki carries at most this many decompressed bytes (§8). */ +export const LOKI_REQUEST_MAX_BYTES = 1024 * 1024; +/** The pack alarm interval (ADR §B.2 step 6). */ +export const PACK_ALARM_INTERVAL_MS = 60_000; +/** Pack early once this many bytes are stored, without waiting for the alarm. */ +export const PACK_AT_BYTES = 4 * 1024 * 1024; +/** The dedupe hash window (ADR §B.2 step 4). */ +export const DEDUPE_WINDOW_MS = 24 * 60 * 60 * 1000; diff --git a/runner/packages/runtime/src/telemetry/index.ts b/runner/packages/runtime/src/telemetry/index.ts new file mode 100644 index 0000000000..e1939057fd --- /dev/null +++ b/runner/packages/runtime/src/telemetry/index.ts @@ -0,0 +1,36 @@ +// @handsontable/demo-runtime/telemetry — the observability contract +// (docs/observability-contract.md) as one importable module. +// +// Pure: no DOM, no Cloudflare imports (same rule as `monitor.ts`), so the +// authoring app, the API worker, the o11y worker and `pipeline/` tests all +// import the same definitions. `pipeline/telemetry-contract.test.mjs` parses +// the contract doc and fails when this barrel's exports disagree with it — +// edit the doc and the relevant file here together (README "Contract" rule). +// +// One file per concern, re-exported flat here: +// attrs.ts — §3 attribute keys, allowed values, Loki labels, `HotAttrs`. +// metrics.ts — §4 Analytics Engine layout, §5 metric registry, `toAePoint`. +// fingerprint.ts — §7 fingerprint, the Babel-code-frame stripper. +// scrub.ts — §3 / ADR §E.4 scrubber, browser and ingest alike. +// classify.ts — bot filter, device/browser/OS classifiers (moved from +// `workers/api/src/analytics.ts`). +// inbox.ts — §8 OTLP `ResourceLogs` builders, NDJSON, inbox keys, +// `InboxWriter` storage-key shapes. +// lite.ts — §9 lite beacon payload type and validator. +// sink.ts — `AeSink`: the real Analytics Engine binding, the local +// ClickHouse shim, and an in-memory sink for tests. +// facade.ts — §6 browser `Telemetry` interface, `noopTelemetry`, +// `recordingTelemetry` (a Faro-backed one is implemented +// separately). +// convert.ts — §6/§9 Faro item / beacon → OTLP log record. + +export * from "./attrs.js"; +export * from "./metrics.js"; +export * from "./fingerprint.js"; +export * from "./scrub.js"; +export * from "./classify.js"; +export * from "./inbox.js"; +export * from "./lite.js"; +export * from "./sink.js"; +export * from "./facade.js"; +export * from "./convert.js"; diff --git a/runner/packages/runtime/src/telemetry/lite.ts b/runner/packages/runtime/src/telemetry/lite.ts new file mode 100644 index 0000000000..683dba52dd --- /dev/null +++ b/runner/packages/runtime/src/telemetry/lite.ts @@ -0,0 +1,137 @@ +// Observability contract §9 — the lite beacon payload sent by the ES5 reporter's +// standalone mode (ADR §C.5) from `/d` and `/embed`, `navigator.sendBeacon`, +// `application/json`, at most 2 KB. + +import { + HT_MAJORS, + DEVICE_CLASSES, + isValidOpenAttrValue, + type DeviceClass, + type Framework, + type HtMajor, +} from "./attrs.js"; + +export const LITE_SURFACES = ["embed", "d"] as const; +export type LiteSurface = (typeof LITE_SURFACES)[number]; + +export const LITE_VITAL_NAMES = ["LCP", "INP", "CLS", "TTFB"] as const; +export type LiteVitalName = (typeof LITE_VITAL_NAMES)[number]; + +/** Matches `MONITOR_MESSAGE_MAX` (`monitor.ts`) — restated here to avoid a + * `monitor.ts` import cycle. */ +export const LITE_MESSAGE_MAX = 500; +/** Matches `MONITOR_STACK_MAX`. */ +export const LITE_STACK_MAX = 2000; + +/** + * Total payload cap (§9: "at most 2 KB"), in bytes of the serialized JSON — + * "2 KB" means 2048 bytes, not decimal 2000. `LITE_MESSAGE_MAX` (500) + + * `LITE_STACK_MAX` (2000) already exceed this on their own (per-field + * ceilings, not a promise both fit together); `isValidLitePayload` enforces + * this total. + * + * Measured (`pipeline/telemetry-lite.test.mjs`): `st` alone at + * `LITE_STACK_MAX` already runs ~2150 bytes — over budget by itself. + * Producing a payload that fits is the sender's job: truncate `st` before + * `m`, and re-validate with `isValidLitePayload` after truncating. + */ +export const LITE_PAYLOAD_MAX_BYTES = 2048; + +interface LiteBase { + v: 1; + s: LiteSurface; + demo: string; + ht: HtMajor; + fw: Framework; + dev: DeviceClass; + /** Epoch ms. Clamped to receive time ± 5 minutes at ingest (§9), same rule as + * every other record's timestamp (ADR §C.2). */ + ts: number; + /** Per-beacon random id: `Math.random().toString(36).slice(2,10)` from + * the reporter, 0-8 lowercase base-36 characters — `Math.random()` landing + * on exactly `0` yields `""`, which is a valid id, not a missing one. Used + * only to keep two byte-identical beacons (same demo/error/millisecond, + * e.g. two parallel page loads, or several throws in one synchronous pass) + * from being deduped as a single record — see `workers/o11y/src/lite.ts`'s + * `hashRecord` call. Absent entirely for a beacon sent by an old, cached + * reporter still on a page; that must hash exactly as before this field + * existed. */ + id?: string; +} + +export interface LiteErrorPayload extends LiteBase { + t: "err"; + /** Error name/type (e.g. `TypeError`). */ + n: string; + /** Normalised message, ≤ `LITE_MESSAGE_MAX` chars. */ + m: string; + /** Stack, ≤ `LITE_STACK_MAX` chars. */ + st?: string; + val: null; +} + +export interface LiteVitalPayload extends LiteBase { + t: "vital"; + n: LiteVitalName; + m?: string; + st?: string; + val: number; +} + +export type LiteBeaconPayload = LiteErrorPayload | LiteVitalPayload; + +function isDeviceClass(v: unknown): v is DeviceClass { + return typeof v === "string" && (DEVICE_CLASSES as readonly string[]).includes(v); +} +function isHtMajor(v: unknown): v is HtMajor { + return typeof v === "string" && (HT_MAJORS as readonly string[]).includes(v); +} +function isLiteSurface(v: unknown): v is LiteSurface { + return v === "embed" || v === "d"; +} + +/** + * Validates shape, field caps, and — decisively — the total serialized size. + * Used both by the sender, to decide whether a built payload may be + * sent at all, and at ingest, on untrusted input: a client-crafted beacon is + * otherwise indistinguishable from a real one, so every bound here is re-checked + * server-side, exactly like `sanitizeMonitorPayload` for the demo-runtime relay. + */ +export function isValidLitePayload(data: unknown): data is LiteBeaconPayload { + if (typeof data !== "object" || data === null) return false; + const d = data as Record<string, unknown>; + + if (d["v"] !== 1) return false; + if (!isLiteSurface(d["s"])) return false; + if (typeof d["demo"] !== "string" || d["demo"].length === 0) return false; + if (!isHtMajor(d["ht"])) return false; + // Bounded, not just "non-empty string" — an unbounded `fw` lets a client + // hoist a multi-kilobyte value into the `hot.framework` Loki label. Same + // bounded charset every other §3 `hot.*` open-set value uses + // (`convert.ts#sanitizeResourceAttributes`). + if (typeof d["fw"] !== "string" || !isValidOpenAttrValue(d["fw"])) return false; + if (!isDeviceClass(d["dev"])) return false; + if (typeof d["ts"] !== "number" || !Number.isFinite(d["ts"])) return false; + if (d["m"] !== undefined && (typeof d["m"] !== "string" || d["m"].length > LITE_MESSAGE_MAX)) return false; + if (d["st"] !== undefined && (typeof d["st"] !== "string" || d["st"].length > LITE_STACK_MAX)) return false; + // `id` is absent for an old/cached reporter — accepted, not required. + // `{0,16}` deliberately allows the empty string: the reporter's + // `Math.random().toString(36).slice(2,10)` can legitimately produce `""` + // when `Math.random()` returns exactly 0, and that must not reject the + // whole beacon. + if (d["id"] !== undefined && (typeof d["id"] !== "string" || !/^[0-9a-z]{0,16}$/.test(d["id"]))) return false; + + if (d["t"] === "err") { + if (typeof d["n"] !== "string" || d["n"].length === 0) return false; + if (typeof d["m"] !== "string") return false; + if (d["val"] !== null) return false; + } else if (d["t"] === "vital") { + if (!(LITE_VITAL_NAMES as readonly string[]).includes(d["n"] as string)) return false; + if (typeof d["val"] !== "number" || !Number.isFinite(d["val"])) return false; + } else { + return false; + } + + const bytes = new TextEncoder().encode(JSON.stringify(data)).length; + return bytes <= LITE_PAYLOAD_MAX_BYTES; +} diff --git a/runner/packages/runtime/src/telemetry/metrics.ts b/runner/packages/runtime/src/telemetry/metrics.ts new file mode 100644 index 0000000000..878ede61b5 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/metrics.ts @@ -0,0 +1,445 @@ +// Observability contract §4 (Analytics Engine layout) and §5 (metric registry). +// +// `pipeline/telemetry-contract.test.mjs` parses both tables out of +// `docs/observability-contract.md` and compares them against `AE_COLUMNS` and +// `METRICS` below, slot for slot and outcome for outcome — edit the doc and this +// file together (README "Contract" rule); editing either alone fails the test. + +import { + DEVICE_CLASSES, + ENVIRONMENTS, + HT_MAJORS, + SERVICE_NAMES, + SURFACES, + TIERS, + type CommonResourceAttrs, + type HotAttrs, +} from "./attrs.js"; + +// ---- §4: Analytics Engine layout (`runner_events`) ----------------------------- + +/** Column name → Analytics Engine slot (`blobN`/`doubleN`), plus `metric` → + * `index1`: the local shim's table uses these same slot names as columns, + * so `SELECT blob8 AS outcome` is the only translation layer. */ +export const AE_COLUMNS: Readonly<Record<string, string>> = { + metric: "index1", + service_name: "blob1", + service_version: "blob2", + environment: "blob3", + surface: "blob4", + tier: "blob5", + framework: "blob6", + ht_major: "blob7", + outcome: "blob8", + reason: "blob9", + route_class: "blob10", + fingerprint: "blob11", + demo_id: "blob12", + model: "blob13", + provider: "blob14", + device: "blob15", + bucket: "blob16", + kind: "blob17", + ref: "blob18", + area: "blob19", + count: "double1", + duration_ms: "double2", + value: "double3", + usd: "double4", + tokens_in: "double5", + tokens_out: "double6", + bytes: "double7", + cap: "double8", +}; + +/** Highest blob/double slot any column occupies — a point is always written + * at this fixed width so a stored row's shape never depends on which + * columns a particular metric filled. */ +const BLOB_SLOT_COUNT = 20; +const DOUBLE_SLOT_COUNT = 20; + +// ---- §5: metric registry ------------------------------------------------------- + +interface MetricDefData { + /** Free text from the "Emitted by" column — informational, not slot-checked. */ + readonly emittedBy: string; + /** Attribute/column names this metric's "Blobs used" cell lists, beyond the + * three universal resource attrs (`service_name`, `service_version`, + * `environment`), which every metric carries. */ + readonly blobs: readonly string[]; + /** Column names this metric's "Doubles" cell lists. */ + readonly doubles: readonly string[]; + /** Attribute name → its closed set of allowed values, only for attributes the + * doc actually constrains (`outcome`, `reason`, or — for + * `preview.runtime_error` — a fixed `surface`). An attribute in `blobs` with + * no entry here is open: any string is accepted. */ + readonly values?: Readonly<Record<string, readonly string[]>>; +} + +const EXAMPLE_ACTION_BLOBS = ["kind", "ref", "area", "framework", "ht_major", "bucket"] as const; +const EXAMPLE_ACTION: MetricDefData = { + emittedBy: "browser (ADR-0042)", + blobs: EXAMPLE_ACTION_BLOBS, + doubles: ["count"], +}; + +const SERVE_DEF: MetricDefData = { + emittedBy: "API worker", + blobs: ["outcome", "demo_id"], + doubles: ["count", "bytes"], + values: { outcome: ["2xx", "304", "4xx", "5xx"] }, +}; + +/** `session.start`'s outcome set — `session.start_ms` reuses it verbatim (the + * doc says so as "outcomes as `session.start`"; the contract test resolves that + * alias by parsing the referenced row, and the two arrays are compared for + * equality, not for a shared reference). */ +const SESSION_START_OUTCOMES = [ + "ready", + "at_capacity", + "container_starting", + "boot_timeout", + "budget_denied", + "error", +] as const; + +const REGISTRY_DATA = { + "preview.ready_ms": { + emittedBy: "browser", + blobs: ["surface", "tier", "framework", "ht_major", "outcome", "bucket"], + doubles: ["duration_ms"], + values: { outcome: ["ready", "error", "timeout", "abandoned"] }, + }, + "sandpack.compile_ms": { + emittedBy: "browser", + blobs: ["tier", "framework", "ht_major", "outcome"], + doubles: ["duration_ms"], + values: { outcome: ["ok", "error"] }, + }, + "sandpack.compile_error": { + emittedBy: "browser", + blobs: ["framework", "ht_major", "fingerprint"], + doubles: ["count"], + }, + "sandpack.bundler_unreachable": { + emittedBy: "browser", + blobs: ["ht_major"], + doubles: ["count", "duration_ms"], + }, + "preview.runtime_error": { + emittedBy: "browser", + blobs: ["surface", "tier", "framework", "ht_major", "fingerprint", "reason"], + doubles: ["count"], + values: { + surface: ["demo-runtime"], + reason: ["uncaught", "console", "network", "stderr"], + }, + }, + "version.switch": { + emittedBy: "browser", + // "ht_major (to)" / "reason (from)" — descriptive, not enumerable: `ht_major` + // is the version switched *to*, `reason` carries the framework/version + // switched *from* as a free-form label. + blobs: ["framework", "ht_major", "reason", "bucket"], + doubles: ["count"], + }, + "bucket.resolve_ms": { + emittedBy: "browser", + blobs: ["bucket", "outcome"], + doubles: ["duration_ms"], + values: { outcome: ["ok", "error"] }, + }, + "session.start_ms": { + emittedBy: "browser", + blobs: ["framework", "ht_major", "outcome", "reason"], + doubles: ["duration_ms"], + values: { outcome: SESSION_START_OUTCOMES, reason: ["cold", "warm"] }, + }, + "hmr.roundtrip_ms": { + emittedBy: "browser", + blobs: ["framework", "ht_major"], + doubles: ["duration_ms"], + }, + web_vital: { + emittedBy: "browser, beacon", + blobs: ["surface", "framework", "ht_major", "reason", "device", "demo_id"], + doubles: ["value"], + values: { reason: ["LCP", "INP", "CLS", "TTFB"] }, + }, + "error.uncaught": { + emittedBy: "browser, beacon", + blobs: ["surface", "fingerprint", "demo_id"], + doubles: ["count"], + }, + "error.handled": { + emittedBy: "browser, API worker", + blobs: ["surface", "route_class", "fingerprint"], + doubles: ["count"], + }, + "example.open": { + emittedBy: "browser (ADR-0042)", + blobs: [...EXAMPLE_ACTION_BLOBS, "reason"], + doubles: ["count"], + // "reason (`entry`)" in the Blobs cell adds `entry` (the first open) to the + // reason values the Outcomes/reason cell lists for later navigations. + values: { reason: ["entry", "deep-link", "picker", "switch", "version-switch", "fork"] }, + }, + "example.engaged": EXAMPLE_ACTION, + "example.forked": EXAMPLE_ACTION, + "example.saved": { ...EXAMPLE_ACTION, emittedBy: "API worker (ADR-0042)" }, + "example.shared": EXAMPLE_ACTION, + "example.downloaded": EXAMPLE_ACTION, + "api.request": { + emittedBy: "API worker", + blobs: ["route_class", "outcome"], + doubles: ["count", "duration_ms"], + values: { outcome: ["2xx", "3xx", "4xx", "5xx"] }, + }, + "session.start": { + emittedBy: "API worker", + blobs: ["framework", "ht_major", "outcome"], + doubles: ["count", "duration_ms"], + values: { outcome: SESSION_START_OUTCOMES }, + }, + "session.end": { + emittedBy: "API worker", + blobs: ["framework", "reason"], + doubles: ["count", "value"], + values: { reason: ["pagehide", "sleep_after", "teardown_failed", "budget_closed"] }, + }, + "container.boot_ms": { + emittedBy: "API worker", + blobs: ["framework", "outcome", "reason"], + doubles: ["duration_ms"], + values: { outcome: ["ready", "window_exceeded", "error"], reason: ["cold", "warm"] }, + }, + "pool.gauge": { + emittedBy: "API worker */5", + blobs: ["reason"], + doubles: ["value", "cap"], + values: { reason: ["live", "builder"] }, + }, + "budget.gauge": { + emittedBy: "API worker */5", + // "reason (tier)" — descriptive, not enumerable: reason carries the budget + // tier name. + blobs: ["reason"], + doubles: ["value", "usd"], + }, + "snapshot.build": { + emittedBy: "API worker", + blobs: ["framework", "outcome", "reason"], + doubles: ["count", "duration_ms", "bytes"], + values: { outcome: ["ok", "failed"], reason: ["inline", "detached"] }, + }, + "serve.share": SERVE_DEF, + "serve.d": SERVE_DEF, + "serve.embed": SERVE_DEF, + "chat.answer": { + emittedBy: "API worker", + blobs: ["model", "outcome"], + doubles: ["count", "duration_ms", "usd", "tokens_in", "tokens_out"], + values: { outcome: ["answered", "denied", "error"] }, + }, + "chat.edit": { + emittedBy: "API worker", + blobs: ["outcome"], + doubles: ["count"], + values: { outcome: ["proposed", "applied", "undone"] }, + }, + "theme.ai": { + emittedBy: "API worker", + blobs: ["model", "outcome"], + doubles: ["count", "duration_ms", "usd"], + values: { outcome: ["answered", "denied", "error"] }, + }, + "import.url": { + emittedBy: "API worker", + blobs: ["provider", "outcome", "reason"], + doubles: ["count", "duration_ms"], + values: { outcome: ["ok", "refused", "error"] }, + }, + "payload.boot": { + emittedBy: "API worker", + blobs: ["framework", "outcome"], + doubles: ["count"], + values: { outcome: ["ok", "error"] }, + }, + "reconcile.run": { + emittedBy: "API worker cron", + blobs: ["outcome"], + doubles: ["count", "duration_ms", "usd"], + values: { outcome: ["ok", "skipped", "error"] }, + }, + "o11y.ingest": { + emittedBy: "o11y worker", + blobs: ["reason", "outcome"], + doubles: ["count", "bytes"], + // "reason = gate" — the reason is the gate name (open, see §B.5), not a + // fixed set. + values: { outcome: ["accepted", "dropped", "duplicate"] }, + }, + "o11y.drain": { + emittedBy: "o11y worker", + blobs: ["reason", "outcome"], + doubles: ["count", "duration_ms", "bytes", "value"], + values: { outcome: ["ok", "partial", "error"], reason: ["backlog", "visit", "reopen"] }, + }, + "o11y.wake": { + emittedBy: "o11y worker", + blobs: ["reason", "outcome"], + doubles: ["count", "duration_ms"], + values: { reason: ["backlog", "visit"], outcome: ["clean", "unclean"] }, + }, + "o11y.backlog": { + emittedBy: "o11y worker cron", + blobs: [], + doubles: ["value", "bytes"], + }, + "o11y.alert": { + emittedBy: "o11y worker cron", + // "reason (rule id)" — descriptive, not enumerable: reason carries the + // alert rule's id. + blobs: ["reason", "outcome"], + doubles: ["count"], + values: { outcome: ["fired", "resolved"] }, + }, +} as const satisfies Record<string, MetricDefData>; + +export type MetricName = keyof typeof REGISTRY_DATA; + +export interface MetricDef extends MetricDefData { + readonly name: MetricName; +} + +export const METRICS: Readonly<Record<MetricName, MetricDef>> = Object.fromEntries( + Object.entries(REGISTRY_DATA).map(([name, def]) => [name, { name: name as MetricName, ...def }]), +) as Record<MetricName, MetricDef>; + +export const METRIC_NAMES: readonly MetricName[] = Object.keys(REGISTRY_DATA) as MetricName[]; + +// ---- Points ---------------------------------------------------------------- + +/** The doubles a metric's "Doubles" column can name — a subset is present on + * any one call, matching the metric's own `doubles` list. */ +export interface MetricValues { + count?: number; + duration_ms?: number; + value?: number; + usd?: number; + tokens_in?: number; + tokens_out?: number; + bytes?: number; + cap?: number; +} + +export interface AePoint { + /** `index1` — the sampling key every query filters on directly. */ + indexes: [string]; + blobs: string[]; + doubles: number[]; +} + +/** Closed sets `toAePoint` enforces regardless of which metric is being + * written — on top of the per-metric `values` constraints in `METRICS`. */ +const CLOSED_ATTRIBUTE_SETS: Readonly<Record<string, readonly string[]>> = { + service_name: SERVICE_NAMES, + environment: ENVIRONMENTS, + surface: SURFACES, + tier: TIERS, + ht_major: HT_MAJORS, + device: DEVICE_CLASSES, +}; + +function slotOffset(slot: string): number { + const m = /^(?:blob|double)(\d+)$/.exec(slot); + if (!m || m[1] === undefined) throw new Error(`toAePoint: not a slot name: "${slot}"`); + return Number(m[1]) - 1; +} + +/** + * Build one Analytics Engine data point (§4) for `metric`, from its §5-declared + * doubles and attributes. Positional and dense: every call returns a + * `BLOB_SLOT_COUNT`/`DOUBLE_SLOT_COUNT`-wide array, unused slots `""`/`0`. + * + * Throws — never silently drops — when `metric` is unknown, when + * `attrs.outcome`/`attrs.reason` is set for a metric whose §5 row does not + * list that column, or when a value is outside the column's closed set. + * This is a producer contract on our own call sites, not a scrub of + * untrusted input — see `scrub.ts` for that. + * + * Load-bearing for the ingest route: `HotAttrs.outcome` is typed `string` + * (no type-level encoding), so only calling this function validates a + * client-supplied outcome — the route handler must catch the throw or + * pre-validate against `METRICS[metric].values`. + */ +export function toAePoint( + metric: MetricName, + values: MetricValues, + attrs: HotAttrs & CommonResourceAttrs, +): AePoint { + const def = METRICS[metric]; + if (!def) throw new Error(`toAePoint: unknown metric ${JSON.stringify(metric)}`); + + const blobs = new Array<string>(BLOB_SLOT_COUNT).fill(""); + const doubles = new Array<number>(DOUBLE_SLOT_COUNT).fill(0); + const attrBag = attrs as unknown as Record<string, string | undefined>; + const valueBag = values as Record<string, number | undefined>; + + const writeBlob = (column: string, value: string) => { + const slot = AE_COLUMNS[column]; + if (!slot) throw new Error(`toAePoint: no AE column for "${column}"`); + blobs[slotOffset(slot)] = value; + }; + const checkAllowed = (column: string, value: string) => { + const allowed = def.values?.[column] ?? CLOSED_ATTRIBUTE_SETS[column]; + if (allowed && !allowed.includes(value)) { + throw new Error( + `toAePoint: "${value}" is not an allowed "${column}" for metric "${metric}" ` + + `(allowed: ${allowed.join(", ")})`, + ); + } + }; + + // blob1–3: universal resource attrs, on every record. + for (const column of ["service_name", "service_version", "environment"] as const) { + const value = attrBag[column]; + if (value === undefined) continue; + checkAllowed(column, value); + writeBlob(column, value); + } + + // `outcome`/`reason` set for a metric that does not list them: a caller bug, + // not silently dropped input. + for (const column of ["outcome", "reason"] as const) { + if (attrBag[column] !== undefined && !def.blobs.includes(column)) { + throw new Error(`toAePoint: metric "${metric}" has no "${column}" slot`); + } + } + + for (const column of def.blobs) { + const value = attrBag[column]; + if (value === undefined) continue; + checkAllowed(column, value); + writeBlob(column, value); + } + + // double1 = count: universal, "1 per point unless pre-aggregated" (§4's + // reading rule — `SUM(_sample_interval * double1)` is how every count is + // read, so a point that never sets it reads back as zero regardless of how + // many really happened). This holds for every metric, not only the ones + // whose own §5 row happens to list "count" among its Doubles — the + // double-side analogue of blob1–3 being universal. + doubles[slotOffset(AE_COLUMNS["count"] as string)] = valueBag["count"] ?? 1; + + for (const column of def.doubles) { + if (column === "count") continue; // handled above, universally + const value = valueBag[column]; + if (value === undefined) continue; + const slot = AE_COLUMNS[column]; + if (!slot) throw new Error(`toAePoint: no AE column for "${column}"`); + doubles[slotOffset(slot)] = value; + } + + return { indexes: [metric], blobs, doubles }; +} diff --git a/runner/packages/runtime/src/telemetry/scrub.ts b/runner/packages/runtime/src/telemetry/scrub.ts new file mode 100644 index 0000000000..9557e03693 --- /dev/null +++ b/runner/packages/runtime/src/telemetry/scrub.ts @@ -0,0 +1,314 @@ +// Observability contract §3 / ADR §E.4 — the one scrubber, run in the browser +// (Faro's `beforeSend`) and authoritatively again at ingest, on the normalised +// OTLP record (ADR §B.2 step 1). Structural types only — no `@grafana/faro-core` +// import, so this module stays DOM/Cloudflare-free and importable from +// `pipeline/` under plain Node. Typechecked against the real +// `@grafana/faro-web-sdk` types with a temporary probe. + +import { redactPreviewHosts, type MonitorKind } from "../monitor.js"; +import { stripCodeFrame } from "./fingerprint.js"; +import { browserOf, deviceOf } from "./classify.js"; +import { ALLOWED_ATTRIBUTE_KEYS } from "./attrs.js"; +import { INBOX_RECORD_MAX_BYTES } from "./inbox.js"; + +// ---- Structural mirrors of the Faro shapes we scrub --------------------- +// Match `@grafana/faro-core`'s `TransportItem<P>`/`Meta` closely enough that +// a real Faro item satisfies them structurally, without importing the +// package (verified against the real `@grafana/faro-web-sdk` types with a +// temporary probe). No `[key: string]: unknown` index signature: a real +// `TransportItem` carries none, and TS requires the source type to match. +// `type` is `string`, not `TransportItemType`'s literal union: TS does not +// consider an enum member assignable to an unrelated literal union. +export interface ScrubbableFaroStackFrame { + filename?: string; + /** ADR §C.3 drain-time symbolication: the real `ExceptionStackFrame` + * already carries these at runtime; `structuredClone` copies them + * through unchanged, only the type never declared them. `function` is a + * JS identifier, never a URL — no new redaction rule needed. */ + function?: string; + lineno?: number; + colno?: number; +} + +export interface ScrubbableFaroPayload { + /** `LogEvent.message` */ + message?: string; + /** `ExceptionEvent.value` */ + value?: string; + /** `ExceptionEvent.type` (the error class name, e.g. `TypeError`) */ + type?: string; + /** `EventEvent.name` */ + name?: string; + /** `MeasurementEvent.values` */ + values?: Record<string, number>; + /** ISO 8601 — every Faro event shape carries its own `timestamp`. */ + timestamp?: string; + /** `ExceptionEvent.stacktrace` */ + stacktrace?: { frames?: ScrubbableFaroStackFrame[] }; + /** `LogEvent.context` / `ExceptionEvent.context` / `MeasurementEvent.context` */ + context?: Record<string, string>; + /** `EventEvent.attributes` */ + attributes?: Record<string, string>; +} + +export interface ScrubbableFaroMeta { + user?: unknown; + page?: { url?: string }; + /** Faro's raw `userAgent` in; `reduceBrowserMeta` replaces the whole object + * with just `browser`/`device` on the way out (§3, "reduce any browser + * meta to the device and browser classes"). */ + browser?: { userAgent?: string; browser?: string; device?: string }; + /** Faro's `app` config — `name`/`version`/`environment` are exactly `service.name` + * / `service.version` / `deployment.environment.name` (§3) under Faro's own + * naming, set once at `initTelemetry()`. */ + app?: { name?: string; version?: string; environment?: string }; + os?: unknown; + device?: unknown; +} + +export interface ScrubbableFaroItem { + /** `TransportItemType`'s runtime values (`"exception"`, `"log"`, + * `"measurement"`, `"trace"`, `"event"`), typed `string` rather than that + * literal union — see the file header. */ + type: string; + payload: ScrubbableFaroPayload; + meta: ScrubbableFaroMeta; +} + +/** A normalised OTLP log record (§8), or near enough — the ingest-time shape + * produced by `convert.ts` before it is packed into the inbox. */ +export interface ScrubbableOtlpRecord { + body?: string; + attributes?: Record<string, string>; + resourceAttributes?: Record<string, string>; +} + +export type Scrubbable = ScrubbableFaroItem | ScrubbableOtlpRecord; + +function isFaroItem(record: Scrubbable): record is ScrubbableFaroItem { + return ( + typeof (record as ScrubbableFaroItem).type === "string" && + typeof (record as ScrubbableFaroItem).payload === "object" && + (record as ScrubbableFaroItem).payload !== null && + typeof (record as ScrubbableFaroItem).meta === "object" && + (record as ScrubbableFaroItem).meta !== null + ); +} + +/** §3: strip the query string and fragment off a URL-valued field. Absolute + * URLs are parsed properly; anything else (a bare path, or not a URL at all) + * falls back to cutting at the first `?`/`#`, so a malformed value from + * untrusted input degrades safely instead of throwing. */ +export function stripQueryAndFragment(value: string): string { + try { + const url = new URL(value); + url.search = ""; + url.hash = ""; + return url.toString(); + } catch { + const cut = value.search(/[?#]/); + return cut === -1 ? value : value.slice(0, cut); + } +} + +/** Faro's console instrumentation is disabled (ADR §E.4), but a demo-runtime + * `console-error`/`console-warn` relay (`monitor.ts`) can surface as a Faro + * log item, tagged via `context["hot.relay"]`. §3 forbids console output + * outright, so a matching item is dropped, not scrubbed. */ +const CONSOLE_KINDS: ReadonlySet<MonitorKind> = new Set(["console-error", "console-warn"]); + +function isConsoleItem(item: ScrubbableFaroItem): boolean { + if (item.type !== "log") return false; + const kind = item.payload.context?.["hot.relay"]; + return typeof kind === "string" && CONSOLE_KINDS.has(kind as MonitorKind); +} + +/** §3: "reduce any browser meta to the device and browser classes" — + * replaces the rich `MetaBrowser`/`MetaOS`/`MetaDevice` objects with the + * same two coarse classes `classify.ts` computes for anonymous analytics. */ +function reduceBrowserMeta(meta: ScrubbableFaroMeta): void { + const ua = meta.browser?.userAgent; + if (ua !== undefined || meta.browser !== undefined || meta.os !== undefined || meta.device !== undefined) { + meta.browser = { browser: browserOf(ua ?? ""), device: deviceOf(ua ?? "") }; + } + delete meta.os; + delete meta.device; +} + +function allowlistAttributes(attrs: Record<string, string> | undefined): Record<string, string> | undefined { + if (!attrs) return attrs; + const out: Record<string, string> = {}; + for (const [key, value] of Object.entries(attrs)) { + if (ALLOWED_ATTRIBUTE_KEYS.has(key)) out[key] = value; + } + return out; +} + +/** §3: strips a query/fragment off a URL *embedded* inside a message/value + * string (unlike `stripQueryAndFragment`, which only handles a field that + * IS a URL). Exported so `text-scrub.ts` runs the same rule server-side. + * Matches after `redactPreviewHosts` (`scrubText` runs this last), so the + * pattern optionally consumes the `<preview>` placeholder first. */ +const EMBEDDED_URL_PATTERN = /\bhttps?:\/\/(?:<preview>)?[^\s"'<>)]*/gi; + +export function stripUrlQueriesInText(text: string): string { + return text.replace(EMBEDDED_URL_PATTERN, (url) => { + const cut = url.search(/[?#]/); + return cut === -1 ? url : url.slice(0, cut); + }); +} + +/** + * Contract §3's "never sent" list includes "an IP" — applied browser-side + * too, exported for `text-scrub.ts`'s server-side pass. Bounded quantifiers + * throughout: no ReDoS backtrack regardless of input shape + * (`pipeline/o11y-redos.test.mjs` pins the timing). + * + * The START boundary is a capturing alternation, never a lookbehind: + * `new RegExp` with `(?<!...)` throws on Safari <16.4, and this module is + * imported EAGERLY at browser boot — a throw here would fail the whole + * telemetry import on any older Safari. + */ +const IPV4_OCTET = "(?:25[0-5]|2[0-4]\\d|1\\d{2}|[1-9]?\\d)"; +const IPV4_PATTERN = new RegExp(`(^|[^\\w.-])(?:${IPV4_OCTET}\\.){3}${IPV4_OCTET}(?![\\w-]|\\.\\d)`, "g"); + +/** IPv6, the standard bounded form (7 alternatives, all `{1,4}`/`{1,7}`-capped) — + * same lookbehind-avoidance boundary as IPv4 above. */ +const IPV6_GROUP = "[0-9A-Fa-f]{1,4}"; +const IPV6_PATTERN = new RegExp( + "(^|[^\\w:])(?:" + + `(?:${IPV6_GROUP}:){7}${IPV6_GROUP}` + + `|(?:${IPV6_GROUP}:){1,7}:` + + `|(?:${IPV6_GROUP}:){1,6}:${IPV6_GROUP}` + + `|(?:${IPV6_GROUP}:){1,5}(?::${IPV6_GROUP}){1,2}` + + `|(?:${IPV6_GROUP}:){1,4}(?::${IPV6_GROUP}){1,3}` + + `|(?:${IPV6_GROUP}:){1,3}(?::${IPV6_GROUP}){1,4}` + + `|(?:${IPV6_GROUP}:){1,2}(?::${IPV6_GROUP}){1,5}` + + `|${IPV6_GROUP}:(?::${IPV6_GROUP}){1,6}` + + `|:(?:(?::${IPV6_GROUP}){1,7}|:)` + + ")(?![\\w:])", + "g", +); + +// IPv4 first, then IPv6: an IPv4-mapped IPv6 address's octets are +// hex-digit-shaped, so IPv6 alone can eat a leading fragment and leave a +// real piece of the address behind. +export function redactIpInText(text: string): string { + return text + .replace(IPV4_PATTERN, (_match, prefix: string) => `${prefix}<ip>`) + .replace(IPV6_PATTERN, (_match, prefix: string) => `${prefix}<ip>`); +} + +/** + * ReDoS defense-in-depth: bound a free-text string to this length BEFORE any + * scrub/redact regex in this module ever sees it, so an unidentified pattern + * still has a bounded worst case. + * + * Set to {@link INBOX_RECORD_MAX_BYTES} (contract §8's 256 KB drop limit), + * not a smaller number: `pipeline/o11y-normalise.test.mjs`'s 300 KB-message + * oversize tests rely on the untruncated length surviving scrub far enough + * that the record-level size check (which runs AFTER scrubbing) still + * measures over the limit. + */ +export const SCRUB_TEXT_MAX_CHARS = INBOX_RECORD_MAX_BYTES; + +export function truncateForScrub(value: string): string { + return value.length > SCRUB_TEXT_MAX_CHARS ? value.slice(0, SCRUB_TEXT_MAX_CHARS) : value; +} + +function scrubText(value: string | undefined): string | undefined { + if (value === undefined) return value; + return redactIpInText(stripUrlQueriesInText(stripCodeFrame(redactPreviewHosts(truncateForScrub(value))))); +} + +/** + * §3: "`redactPreviewHosts` on every string" — not only the fields the + * targeted rules above cover. A preview URL is a session credential, so it + * must never survive in an allowlisted attribute or resource attribute + * (`hot.framework` becomes a Loki label). Walks every string leaf in place; + * idempotent, so it can safely run last, after every targeted rule. + */ +function redactStringsDeep<V>(value: V): V { + if (typeof value === "string") return redactPreviewHosts(truncateForScrub(value)) as V; + if (Array.isArray(value)) { + for (let i = 0; i < value.length; i++) value[i] = redactStringsDeep(value[i]); + return value; + } + if (value !== null && typeof value === "object") { + const obj = value as Record<string, unknown>; + // Keys too, not only values: a `MeasurementEvent.values` object's keys + // are metric names, JSON.stringify'd straight into the OTLP body + // (`convert.ts#faroBody`) without ever passing back through a value + // position this walk would otherwise reach. + for (const key of Object.keys(obj)) { + const redactedKey = redactPreviewHosts(truncateForScrub(key)); + const redactedValue = redactStringsDeep(obj[key]); + if (redactedKey !== key) delete obj[key]; + obj[redactedKey] = redactedValue; + } + return value; + } + return value; +} + +/** + * The one scrubber (ADR §E.4). Never mutates its argument; returns a scrubbed + * clone, or `null` when the whole record must be dropped (a console item). + * + * Applies, in this order: drop console items; drop `meta.user`; reduce browser + * meta to device/browser classes; strip query/fragment and redact preview hosts + * on the page URL and every stack-frame filename; redact preview hosts and strip + * Babel code frames from message-bearing text; allowlist `attributes` / + * `resourceAttributes` / `context` (§3's forbidden attributes — `url.full`, geo, + * ASN, the user pseudonym, an email, an IP, a user-agent string — are simply + * never on the allowlist); finally, `redactPreviewHosts` on every + * remaining string in the record, not only the fields named above — an + * allowlisted attribute value (`session.id`, `hot.framework`) is still + * client-supplied and can carry a preview host too. + */ +export function scrubTelemetry<T extends Scrubbable>(record: T): T | null { + const clone = structuredClone(record) as T; + + if (isFaroItem(clone)) { + if (isConsoleItem(clone)) return null; + + delete clone.meta.user; + reduceBrowserMeta(clone.meta); + + if (clone.meta.page?.url !== undefined) { + clone.meta.page.url = redactPreviewHosts(stripQueryAndFragment(clone.meta.page.url)); + } + + for (const frame of clone.payload.stacktrace?.frames ?? []) { + // An untrusted client can send a `null`/non-object entry inside + // `stacktrace.frames` (`{"stacktrace": {"frames":[null]}}` is valid + // JSON) — `frame.filename` on a `null` would throw a `TypeError` that + // escapes as an uncaught `500`, contradicting this module's own + // "never a 500" contract. Skipped, not scrubbed: there is nothing in + // a non-object frame to redact. + if (!frame || typeof frame !== "object") continue; + // A stack frame's `filename` is a URL-valued field too (a bundler's + // cache-busting `?t=`/`?v=` query string shows up here as often as on + // `meta.page.url`), so it gets the same two rules. + if (frame.filename !== undefined) { + frame.filename = redactPreviewHosts(stripQueryAndFragment(frame.filename)); + } + } + + clone.payload.message = scrubText(clone.payload.message); + clone.payload.value = scrubText(clone.payload.value); + clone.payload.context = allowlistAttributes(clone.payload.context); + clone.payload.attributes = allowlistAttributes(clone.payload.attributes); + + redactStringsDeep(clone.payload); + redactStringsDeep(clone.meta); + return clone; + } + + const otlp = clone as ScrubbableOtlpRecord; + otlp.body = scrubText(otlp.body); + otlp.attributes = allowlistAttributes(otlp.attributes); + otlp.resourceAttributes = allowlistAttributes(otlp.resourceAttributes); + redactStringsDeep(otlp); + return clone; +} diff --git a/runner/packages/runtime/src/telemetry/sink.ts b/runner/packages/runtime/src/telemetry/sink.ts new file mode 100644 index 0000000000..5855a8333f --- /dev/null +++ b/runner/packages/runtime/src/telemetry/sink.ts @@ -0,0 +1,124 @@ +// One write surface for an `AePoint` (`metrics.ts#toAePoint`), so the o11y worker, +// the API worker and a local dev/test run all call the same shape. Structural +// only: `AnalyticsEngineDatasetLike` mirrors the real Workers binding without +// importing `@cloudflare/workers-types` (this package stays Cloudflare-free). + +import type { AePoint } from "./metrics.js"; + +export interface AeSink { + /** Mirrors the real `AnalyticsEngineDataset#writeDataPoint` signature: normally + * fire-and-forget (`void`), but `clickhouseSink`'s HTTP write returns a promise + * a caller that cares about local-dev delivery may await. */ + writeDataPoint(point: AePoint): void | Promise<void>; +} + +/** Structural mirror of Cloudflare's `AnalyticsEngineDataset` binding. */ +export interface AnalyticsEngineDatasetLike { + writeDataPoint(point: { indexes: string[]; blobs?: string[]; doubles?: number[] }): void; +} + +/** Production sink: the real Analytics Engine binding (§4, `RUNNER_EVENTS`). */ +export function bindingSink(dataset: AnalyticsEngineDatasetLike): AeSink { + return { + writeDataPoint(point: AePoint): void { + dataset.writeDataPoint(point); + }, + }; +} + +/** + * `date` as raw epoch milliseconds — the value to send a `DateTime64(3)` + * column over JSONEachRow (see `clickhouseSink`'s doc comment for the + * measured reasoning). Exported for + * `pipeline/telemetry-sink.test.mjs`. + */ +export function clickhouseTimestamp(date: Date): number { + return date.getTime(); +} + +export interface ClickhouseSinkOptions { + /** Table name — `runner_events` (§10) by default. */ + table?: string; + /** Injectable for tests; defaults to the global `fetch`. */ + fetchImpl?: typeof fetch; + /** ClickHouse HTTP user (`X-ClickHouse-User`). `"default"` if omitted, the + * same default `containers/o11y/compose.yml` uses. */ + user?: string; + /** ClickHouse HTTP password (`X-ClickHouse-Key`) — the local container + * always requires one (`CLICKHOUSE_PASSWORD`, defaulting to + * `local-dev-token` from `AE_SQL_TOKEN`); the API worker passes + * `env.AE_SQL_TOKEN`. + * Measured: with no credentials sent at all, the local + * container answers the insert with a non-2xx auth error, which a version + * of this sink that only checked "did `fetch` throw" swallowed — + * `writeDataPoint` resolved, `SELECT count()` on the table read `0`. Omit + * only against a ClickHouse that genuinely has no auth configured. */ + password?: string; +} + +/** + * Local-mode sink (§10): ClickHouse at `http://localhost:8123`, table + * `runner_events` with the §4 columns plus `timestamp`/`_sample_interval` + * (DDL in `containers/o11y/local/clickhouse-init.sql`). + * + * `timestamp` is sent as a raw epoch-millisecond integer + * (`clickhouseTimestamp`), not a formatted string: a `DateTime64(3)` column + * reads a plain integer as milliseconds (not seconds), and a formatted + * string is parsed in the SERVER's configured timezone — measured 9 hours + * off under a `session_timezone` override. A unit-less epoch-ms integer is + * immune to both. + * + * Authenticates and rejects on a non-2xx response (measured: no + * credentials gets `403` from the local container); credentials are sent + * whenever `options.user`/`.password` are given. + */ +export function clickhouseSink(url: string, options: ClickhouseSinkOptions = {}): AeSink { + const table = options.table ?? "runner_events"; + const doFetch = options.fetchImpl ?? fetch; + const headers: Record<string, string> = { "Content-Type": "application/json" }; + if (options.user !== undefined) headers["X-ClickHouse-User"] = options.user; + if (options.password !== undefined) headers["X-ClickHouse-Key"] = options.password; + + return { + writeDataPoint(point: AePoint): Promise<void> { + const row: Record<string, string | number> = { + timestamp: clickhouseTimestamp(new Date()), + _sample_interval: 1, + index1: point.indexes[0] ?? "", + }; + point.blobs.forEach((value, i) => { + row[`blob${i + 1}`] = value; + }); + point.doubles.forEach((value, i) => { + row[`double${i + 1}`] = value; + }); + const endpoint = `${url.replace(/\/$/, "")}/?query=${encodeURIComponent( + `INSERT INTO ${table} FORMAT JSONEachRow`, + )}`; + return doFetch(endpoint, { method: "POST", headers, body: `${JSON.stringify(row)}\n` }).then( + async (res: { ok: boolean; status: number; text?: () => Promise<string> }) => { + if (!res.ok) { + const body = (await res.text?.()) ?? ""; + throw new Error(`clickhouseSink: insert failed, ${res.status}: ${body.slice(0, 200)}`); + } + }, + ); + }, + }; +} + +export interface MemorySink extends AeSink { + /** Every point written so far, in write order. Tests read this directly. */ + readonly points: AePoint[]; +} + +/** Test sink: collects every point, writes nothing anywhere. */ +export function memorySink(): MemorySink { + const points: AePoint[] = []; + return { + points, + writeDataPoint(point: AePoint): void { + points.push(point); + }, + }; +} diff --git a/runner/packages/runtime/src/transpile.ts b/runner/packages/runtime/src/transpile.ts index 68216c0a2e..28cb36f1f1 100644 --- a/runner/packages/runtime/src/transpile.ts +++ b/runner/packages/runtime/src/transpile.ts @@ -114,6 +114,29 @@ export function isCompilerUnavailable(e: unknown): boolean { return e instanceof Error && (e as { compilerUnavailable?: boolean }).compilerUnavailable === true; } +/** + * The visitor's own source failed to parse: babel threw inside + * `transpileFilesForParcel`. This is the parcel Tier-1 compile error — the bundler never + * sees these sources, so it is the only place the failure exists as an error object. + * + * A plain `Error` marked after construction, not a subclass or a factory, and deliberately + * so. The same error reaches Sentry as the `cause` of the constant-titled + * `Tier1CompileError` on the mount path (`tier1Report`), and the linked-errors integration + * serialises its `name`, message and stack: a subclass would rename it or add a + * constructor frame, a factory would add its own frame, and this classification must not + * move a byte of what Sentry receives. The marker (not `instanceof`) also matches + * `CompilerUnavailableError`'s cross-bundle reasoning above. + */ +function markTranspileFailure(error: Error): void { + (error as { transpileFailed?: boolean }).transpileFailed = true; +} + +/** Whether an error is `transpileFilesForParcel`'s own parse failure (see `markTranspileFailure`) + * — never the compiler chunk failing to load, which is `isCompilerUnavailable`. */ +export function isTranspileFailure(e: unknown): boolean { + return e instanceof Error && (e as { transpileFailed?: boolean }).transpileFailed === true; +} + /** * Retry a failed chunk load once — against a *different URL* — then stop asking * (DEV-2569). @@ -376,7 +399,10 @@ export async function transpileFilesForParcel(files: FilesMap): Promise<FilesMap configFile: false, }).code ?? ""; } catch (e) { - throw new Error(`Failed to transpile ${path} for the parcel sandbox: ${(e as Error).message}`); + // Constructed right here, as before, so its stack is unchanged (see `markTranspileFailure`). + const failure = new Error(`Failed to transpile ${path} for the parcel sandbox: ${(e as Error).message}`); + markTranspileFailure(failure); + throw failure; } if (compiled.includes(JSX_PRAGMA)) compiled = JSX_IMPORT + compiled; const jsPath = path.replace(SOURCE_RE, ".js"); diff --git a/runner/packages/runtime/src/types.ts b/runner/packages/runtime/src/types.ts index 5de1e7212a..a2be8b734b 100644 --- a/runner/packages/runtime/src/types.ts +++ b/runner/packages/runtime/src/types.ts @@ -117,6 +117,57 @@ export interface WriteFileOptions { quiet?: boolean; } +// ---- Timing hooks (observability contract §5) -------------------------- +// +// Declared here, not in `sandpack.ts`/`container.ts`, so `DemoRuntime` can name +// them as optional members without a cycle (both engine files already import +// `DemoRuntime` from this file; the reverse import would be circular). Each +// engine implements the subset it can produce a real signal for — an optional +// interface member needs no stub on the class that has nothing to report — +// and `apps/authoring/src/telemetry/metrics.ts#wireRuntimeMetrics` calls every +// one of them through an optional-chained `runtime.onX?.(cb)`, so a caller +// holding a bare `DemoRuntime` never has to cast to the concrete engine type. + +export interface SandpackCompileTimingEvent { + readonly durationMs: number; + readonly outcome: "ok" | "error"; +} + +/** A compile diagnostic: a bundler `show-error` with no frames, or the + * parcel pre-transpile's own babel parse failure, which never reaches the + * bundler — on mount, and on the edit path for the newest push only. Never a + * Sandpack evaluation error — a runtime throw inside an already-evaluated + * module, error-reporting territory, not this signal. `message` is already bounded through + * `boundCompileMessage` in `sandpack.ts` — redacted and truncated, never raw + * authored code. */ +export interface SandpackCompileErrorEvent { + readonly message: string; + /** `transpile`: the client-side pre-transpile failed, nothing was dispatched. + * `bundler`: the bundler rejected a dispatched sandbox (frameless `show-error`). */ + readonly origin: "transpile" | "bundler"; +} + +/** `loadSandpackClient` itself rejected — the hosted bundler's connection + * never came up, as distinct from a compile/evaluation error, both of which + * only arrive once the client has connected. */ +export interface SandpackBundlerUnreachableEvent { + readonly durationMs: number; +} + +export interface SessionStartTimingEvent { + readonly elapsedMs: number; + /** Matches `session.start`'s outcome set (contract §5: "outcomes as `session.start`"). */ + readonly outcome: "ready" | "at_capacity" | "container_starting" | "boot_timeout" | "budget_denied" | "error"; +} + +/** A post-ready preview navigation that followed a real edit flush, and not + * the runtime's own `reload()`. Only a dev server that does a full page + * reload on an edit is observable this way — genuine in-place HMR (a module + * patched without navigating the frame) stays invisible by construction. */ +export interface HmrRoundtripEvent { + readonly durationMs: number; +} + export interface DemoRuntime { mount(files: FilesMap): Promise<{ previewUrl: string }>; writeFile(path: string, contents: string, opts?: WriteFileOptions): void; @@ -145,6 +196,19 @@ export interface DemoRuntime { reload?(): Promise<void> | void; onReady(cb: () => void): void; onError(cb: (e: Error) => void): void; + /** §5 `sandpack.compile_ms` (`SandpackRuntime` only). */ + onCompileTiming?(cb: (e: SandpackCompileTimingEvent) => void): void; + /** §5 `sandpack.compile_error` (`SandpackRuntime` only). */ + onCompileError?(cb: (e: SandpackCompileErrorEvent) => void): void; + /** §5 `sandpack.bundler_unreachable` (`SandpackRuntime` only). */ + onBundlerUnreachable?(cb: (e: SandpackBundlerUnreachableEvent) => void): void; + /** The newest push's outcome (`SandpackRuntime` only): `rerun` when the bundler starts + * running a new sandbox, `unchanged` when it matched the running one and nothing re-runs. */ + onPushOutcome?(cb: (outcome: "rerun" | "unchanged") => void): void; + /** §5 `session.start_ms` (`ContainerRuntime` only). */ + onSessionStart?(cb: (e: SessionStartTimingEvent) => void): void; + /** §5 `hmr.roundtrip_ms` (`ContainerRuntime` only). */ + onHmr?(cb: (e: HmrRoundtripEvent) => void): void; dispose(): void; } diff --git a/runner/pipeline/api-cors.test.mjs b/runner/pipeline/api-cors.test.mjs new file mode 100644 index 0000000000..b47609a68e --- /dev/null +++ b/runner/pipeline/api-cors.test.mjs @@ -0,0 +1,45 @@ +// `index.ts#cors()`'s `Access-Control-Allow-Headers` list +// (`Content-Type, Authorization`) must include `x-hot-session`, the header +// `apps/authoring/src/telemetry/index.ts#apiHeaders()` sets on nearly every +// fetch call site (ADR-0041 contract §6's session join). Production and +// the vite dev proxy are both same-origin, so no browser ever preflights +// this — a cross-origin local dev setup (`VITE_API_BASE` pointed straight +// at the API worker) is the only place a missing header here is visible at +// all, and it fails closed (the browser refuses the request before it is +// ever sent). Driven through the real router (`workers/api/src/index.ts`'s +// default export), not a re-declared copy of the header list. +// Run: node --experimental-strip-types --test pipeline/api-cors.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +function req(method, path, init = {}) { + return new Request(`https://demos.handsontable.com${path}`, { method, ...init }); +} + +test("an OPTIONS preflight allows x-hot-session, the telemetry session-join header", async () => { + const env = makeEnv(); + const res = await worker.fetch(req("OPTIONS", "/api/versions"), env, ctx); + assert.equal(res.status, 204); + const allowed = res.headers.get("Access-Control-Allow-Headers") ?? ""; + assert.ok( + allowed.split(",").map((s) => s.trim().toLowerCase()).includes("x-hot-session"), + `expected "x-hot-session" in Access-Control-Allow-Headers, got: "${allowed}"`, + ); +}); + +test("a real GET response also carries x-hot-session in Access-Control-Allow-Headers", async () => { + const env = makeEnv(); + const res = await worker.fetch(req("GET", "/api/versions"), env, ctx); + const allowed = res.headers.get("Access-Control-Allow-Headers") ?? ""; + assert.ok( + allowed.split(",").map((s) => s.trim().toLowerCase()).includes("x-hot-session"), + `expected "x-hot-session" in Access-Control-Allow-Headers, got: "${allowed}"`, + ); +}); diff --git a/runner/pipeline/api-cron-wiring.test.mjs b/runner/pipeline/api-cron-wiring.test.mjs new file mode 100644 index 0000000000..ee78727d59 --- /dev/null +++ b/runner/pipeline/api-cron-wiring.test.mjs @@ -0,0 +1,87 @@ +// Structural pins for the API worker's cron wiring +// (`workers/api/src/index.ts`). `runNightlyCron`/`runFiveMinuteCron` are +// private to that file and pull in the whole Worker's dependency graph +// (D1/KV/Sandbox/Sentry bindings), so — same rationale as +// `pipeline/master-workflow.test.mjs`'s structural pins on `master.yml` — +// these read the source as text and assert its exact shape rather than +// importing and executing it. +// Run: node --experimental-strip-types --test pipeline/api-cron-wiring.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const indexPath = join(__dirname, "..", "workers/api/src/index.ts"); +const source = readFileSync(indexPath, "utf8"); + +function bodyOf(fnName) { + const start = source.indexOf(`async function ${fnName}(`); + assert.ok(start > -1, `${fnName} must exist in index.ts`); + const braceStart = source.indexOf("{", start); + // Find the matching closing brace by depth-counting — the bodies below + // never contain a template literal with an unbalanced `{`, so this is safe. + let depth = 0; + let i = braceStart; + for (; i < source.length; i++) { + if (source[i] === "{") depth++; + else if (source[i] === "}") { + depth--; + if (depth === 0) break; + } + } + return source.slice(braceStart, i + 1); +} + +// `rollupExampleDaily` must run under its own `cronStep` call, outside the +// billing chain's callback body — inside the same +// `cronStep(env, "cron:nightly", ...)` chain, an upstream throw (e.g. +// `reconcileBilling`) would skip it for the whole night. +test("index.ts: the nightly example_daily rollup has its own independent cronStep, not nested in the billing chain", () => { + const nightlyBody = bodyOf("runNightlyCron"); + + const billingStart = nightlyBody.indexOf('cronStep(env, "cron:nightly",'); + assert.ok(billingStart > -1, "the billing cronStep call must exist"); + const billingCallbackStart = nightlyBody.indexOf("{", nightlyBody.indexOf("=>", billingStart)); + let depth = 0; + let i = billingCallbackStart; + for (; i < nightlyBody.length; i++) { + if (nightlyBody[i] === "{") depth++; + else if (nightlyBody[i] === "}") { + depth--; + if (depth === 0) break; + } + } + const billingCallbackBody = nightlyBody.slice(billingCallbackStart, i + 1); + + assert.doesNotMatch( + billingCallbackBody, + /rollupExampleDaily/, + "rollupExampleDaily must NOT be called inside the billing chain's cronStep callback", + ); + assert.match( + nightlyBody.slice(i + 1), + /await cronStep\(env, "cron:nightly:rollup", async \(\) => \{\s*await rollupExampleDaily\(env\);\s*\}\);/, + "rollupExampleDaily must run under its own cronStep call, after the billing chain's cronStep has returned", + ); +}); + +// The API worker's own `*/5` cron must write one structured +// `log.kind: "cron.tick"` line (via `telemetry/lines.ts`'s shared helper) +// so o11y's `heartbeat.lastIngest` watchdog check is a true end-to-end +// signal, not just "did the ingest pipeline exist". +test("index.ts: the five-minute cron writes a cron.tick line through logCronTickLine", () => { + const fiveMinuteBody = bodyOf("runFiveMinuteCron"); + assert.match( + fiveMinuteBody, + /await cronStep\(env, "cron:five-minute:tick", \(\) => \{\s*logCronTickLine\(env\);/, + "runFiveMinuteCron must call logCronTickLine(env) under its own cronStep", + ); + assert.match( + source, + /\blogCronTickLine\b/, + "logCronTickLine must be imported from telemetry/index.js", + ); +}); diff --git a/runner/pipeline/api-error.test.mjs b/runner/pipeline/api-error.test.mjs index bfb78f8a5a..e5a66672a5 100644 --- a/runner/pipeline/api-error.test.mjs +++ b/runner/pipeline/api-error.test.mjs @@ -255,3 +255,24 @@ test("other 409s stay unclassified", () => { assert.equal(failure.kind, "other"); assert.equal(failure.reportable, true); }); + +test("a build_failed 422 says nothing was saved and shows the build error, never the wire text", () => { + const detail = 'src/index.tsx:1:10: ERROR: Unexpected ";"'; + const failure = describeApiFailure( + 422, + { error: `build failed: ${detail}`, code: "build_failed", detail }, + "save failed (422)", + ); + assert.equal(failure.message, `The build failed, so nothing was saved. ${detail}`); + assert.doesNotMatch(failure.message, /build_failed|build failed:/); + assert.equal(failure.reportable, false, "the author's own build error is not a Sentry issue"); + assert.equal( + describeApiFailure(422, { error: "build failed: x", code: "build_failed" }, "x").message, + "The build failed, so nothing was saved.", + ); + assert.equal( + describeApiFailure(422, { error: "build_failed", detail }, "x").message, + `The build failed, so nothing was saved. ${detail}`, + "the older API's bare code still reads as the sentence", + ); +}); diff --git a/runner/pipeline/api-telemetry-config.test.mjs b/runner/pipeline/api-telemetry-config.test.mjs new file mode 100644 index 0000000000..d643c7738c --- /dev/null +++ b/runner/pipeline/api-telemetry-config.test.mjs @@ -0,0 +1,134 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import { fileURLToPath } from "node:url"; + +// API worker signals, error lines, the Sentry scope switch (ADR-0041 §D, +// §E.1, §E.3; contract §2). Pins the exact config the ADR names, so a +// revert of any one value goes red: full-fidelity head sampling with +// invocation logs off, no export destination yet (`o11y-logs` can only be +// created after this Worker's first deploy — run-and-deploy.md "First +// deploy, in order" — and is added back by a follow-up), the sampled 1% +// trace rate with no destination (no trace is ever exported), the `*/5` +// observability cron beside the unchanged nightly one, and +// `SERVICE_VERSION` wired into the deploy script. +// +// No JSON5 dependency: a small string-aware `//`-comment stripper is +// enough for this repo's actual `.jsonc` style (line comments only, no +// trailing commas). + +function stripLineComments(text) { + let out = ""; + let inString = false; + let escape = false; + for (let i = 0; i < text.length; i += 1) { + const c = text[i]; + if (inString) { + out += c; + if (escape) escape = false; + else if (c === "\\") escape = true; + else if (c === "\"") inString = false; + continue; + } + if (c === "\"") { + inString = true; + out += c; + continue; + } + if (c === "/" && text[i + 1] === "/") { + while (i < text.length && text[i] !== "\n") i += 1; + out += "\n"; + continue; + } + out += c; + } + return out; +} + +const workersApiDir = fileURLToPath(new URL("../workers/api/", import.meta.url)); +const wranglerJsonc = fs.readFileSync(`${workersApiDir}wrangler.jsonc`, "utf8"); +const wrangler = JSON.parse(stripLineComments(wranglerJsonc)); +const pkg = JSON.parse(fs.readFileSync(`${workersApiDir}package.json`, "utf8")); + +test("stripLineComments does not corrupt a string containing //", () => { + // wrangler.jsonc's own vars carry URLs — a naive per-line stripper would + // truncate them. Guards the parser this file's own assertions depend on. + assert.equal( + JSON.parse(stripLineComments('{"a": "https://example.com/x"}')).a, + "https://example.com/x", + ); +}); + +test("observability.logs: full fidelity, invocation logs off, persisted, no export destination yet", () => { + assert.deepEqual(wrangler.observability.logs, { + enabled: true, + head_sampling_rate: 1.0, + invocation_logs: false, + persist: true, + destinations: [], + }); +}); + +test("observability.traces: 1% sampled, persisted, no destination (ADR §C.4 — no trace export)", () => { + assert.deepEqual(wrangler.observability.traces, { + enabled: true, + head_sampling_rate: 0.01, + persist: true, + }); + assert.equal("destinations" in wrangler.observability.traces, false); +}); + +test("the */5 cron runs alongside the unchanged nightly one", () => { + assert.deepEqual(wrangler.triggers.crons, ["17 4 * * *", "*/5 * * * *"]); +}); + +test("RUNNER_EVENTS and O11Y bindings are wired (not just declared)", () => { + assert.equal(wrangler.analytics_engine_datasets[0].binding, "RUNNER_EVENTS"); + assert.equal(wrangler.analytics_engine_datasets[0].dataset, "runner_events"); + assert.equal(wrangler.services[0].binding, "O11Y"); + assert.equal(wrangler.services[0].service, "handsontable-demos-o11y"); +}); + +// `env.O11Y` must bind to the named `O11yHeartbeat` RPC entrypoint, not the +// o11y worker's default export, which has no HTTP route for this report. +// Also checks every o11y-service binding's declared entrypoint is a real +// named export of the target worker's `index.ts`. +test("every o11y service binding declares a real entrypoint, and O11Y's is O11yHeartbeat", () => { + const o11yIndexPath = fileURLToPath(new URL("../workers/o11y/src/index.ts", import.meta.url)); + const o11yIndexSrc = fs.readFileSync(o11yIndexPath, "utf8"); + const o11y = wrangler.services.find((svc) => svc.binding === "O11Y"); + assert.equal( + o11y?.entrypoint, + "O11yHeartbeat", + "env.O11Y must bind to entrypoint: \"O11yHeartbeat\" — without it, env.O11Y resolves to the o11y " + + "worker's default export, which has no HTTP route for the heartbeat report", + ); + for (const svc of wrangler.services) { + if (svc.service !== "handsontable-demos-o11y") continue; + assert.ok(svc.entrypoint, `services[] entry for "${svc.service}" (binding ${svc.binding}) must declare an entrypoint`); + const exportRe = new RegExp(`export\\s*\\{[^}]*\\b${svc.entrypoint}\\b[^}]*\\}|export\\s+class\\s+${svc.entrypoint}\\b`); + assert.match( + o11yIndexSrc, + exportRe, + `workers/o11y/src/index.ts does not export "${svc.entrypoint}", which ${svc.binding}'s entrypoint names`, + ); + } +}); + +test("SENTRY_SCOPE defaults to full (contract §11)", () => { + assert.equal(wrangler.vars.SENTRY_SCOPE, "full"); +}); + +test("the deploy script sets SERVICE_VERSION from GITHUB_SHA, alongside every existing --routes flag", () => { + const deploy = pkg.scripts.deploy; + for (const route of [ + "*.demos.handsontable.com/*", + "demos.handsontable.com/api/*", + "demos.handsontable.com/d/*", + "demos.handsontable.com/embed/*", + ]) { + assert.ok(deploy.includes(`--routes '${route}'`), `missing --routes '${route}'`); + } + assert.ok(deploy.includes("--var SENTRY_ENVIRONMENT:api-production")); + assert.ok(deploy.includes("--var SERVICE_VERSION:$GITHUB_SHA")); +}); diff --git a/runner/pipeline/api-telemetry-cron-step.test.mjs b/runner/pipeline/api-telemetry-cron-step.test.mjs new file mode 100644 index 0000000000..e5006b7562 --- /dev/null +++ b/runner/pipeline/api-telemetry-cron-step.test.mjs @@ -0,0 +1,62 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +// `cronStep` (`workers/api/src/telemetry/cron-step.ts`) needs a direct +// test: its explicit, ungated `Sentry.captureException` for a failed cron +// step is necessary because a throw inside `ctx.waitUntil(...)` is +// invisible to `Sentry.withSentry`'s own `scheduled` auto-capture. +// +// `cron-step.ts` is a leaf on purpose (only `lines.ts` -> `resource.ts`, +// plus the real `@sentry/cloudflare` package) so it can be copied the same +// way `api-telemetry-diagnostic.test.mjs` copies `diagnostic.ts`'s chain — +// see that file's header comment for why the copy lands inside +// `workers/api/` rather than the OS temp dir. +// +// `cronStep` takes an injectable `capture` function — the test below +// passes a recorder instead of the real `@sentry/cloudflare` call. + +const workersApiDir = join(import.meta.dirname, "..", "workers/api"); +const telemetrySrc = join(workersApiDir, "src/telemetry"); +const envSrc = join(workersApiDir, "src/env.ts"); +const dir = mkdtempSync(join(workersApiDir, ".hot-cron-step-")); +for (const file of ["cron-step.ts", "lines.ts", "resource.ts"]) { + writeFileSync(join(dir, file), readFileSync(join(telemetrySrc, file), "utf8").replaceAll('.js"', '.ts"')); +} +writeFileSync(join(dir, "env.ts"), readFileSync(envSrc, "utf8")); +const { cronStep } = await import(join(dir, "cron-step.ts")); +rmSync(dir, { recursive: true, force: true }); + +const ENV = { PREVIEW_HOST: "demos.handsontable.com" }; + +function recorder() { + const calls = []; + const capture = (err, context) => calls.push({ err, context }); + return { calls, capture }; +} + +test("cronStep: a throwing step is captured, and the step's own error is swallowed (does not rethrow)", async () => { + const { calls, capture } = recorder(); + const err = new Error("cron step failed"); + await assert.doesNotReject(() => cronStep(ENV, "cron:test-step", () => { throw err; }, capture)); + assert.equal(calls.length, 1); + assert.equal(calls[0].err, err); + assert.deepEqual(calls[0].context, { tags: { context: "cron:test-step" } }); +}); + +test("cronStep: a succeeding step never calls capture", async () => { + const { calls, capture } = recorder(); + await cronStep(ENV, "cron:test-step", async () => { /* ok */ }, capture); + assert.equal(calls.length, 0); +}); + +test("cronStep: an async rejection is captured the same way a sync throw is", async () => { + const { calls, capture } = recorder(); + await cronStep(ENV, "cron:test-step", () => Promise.reject(new Error("async boom")), capture); + assert.equal(calls.length, 1); +}); + +test("cronStep: with no capture argument, does not throw (falls back to the real Sentry call, inert with no client configured)", async () => { + await assert.doesNotReject(() => cronStep(ENV, "cron:test-step", () => { throw new Error("boom"); })); +}); diff --git a/runner/pipeline/api-telemetry-cron-tick.test.mjs b/runner/pipeline/api-telemetry-cron-tick.test.mjs new file mode 100644 index 0000000000..82a93e7a77 --- /dev/null +++ b/runner/pipeline/api-telemetry-cron-tick.test.mjs @@ -0,0 +1,45 @@ +// `telemetry/lines.ts#logCronTickLine` — the one structured line the API +// worker's `*/5` cron writes so o11y's `heartbeat.lastIngest` watchdog +// check is a true end-to-end signal, even during a real quiet period with +// no user traffic. See `lines.ts`'s own doc comment for why +// `normalise/otlp.ts` needed no change for this line to still count as an +// ingest event. +// +// `lines.ts` imports `./resource.js` relatively — the loader hook below +// remaps that to `resource.ts` on disk, the same way every other spec that +// imports straight from `workers/api/src/` does. +// Run: node --experimental-strip-types --test pipeline/api-telemetry-cron-tick.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { logCronTickLine } = await import("../workers/api/src/telemetry/lines.ts"); + +function captureConsoleLog(fn) { + const original = console.log; + const lines = []; + console.log = (line) => lines.push(line); + try { + fn(); + } finally { + console.log = original; + } + return lines; +} + +test("logCronTickLine: emits exactly one console.log line shaped log.kind=cron.tick", () => { + const lines = captureConsoleLog(() => logCronTickLine({ SERVICE_VERSION: "abc123" })); + assert.equal(lines.length, 1, "must write exactly one line per call"); + const parsed = JSON.parse(lines[0]); + assert.equal(parsed["log.kind"], "cron.tick"); + assert.equal(parsed["service.version"], "abc123"); +}); + +test("logCronTickLine: falls back to the same service.version resolution every other line uses", () => { + const lines = captureConsoleLog(() => logCronTickLine({})); + const parsed = JSON.parse(lines[0]); + assert.equal(parsed["service.version"], "dev", "matches resource.ts#serviceVersion's own documented fallback"); +}); diff --git a/runner/pipeline/api-telemetry-diagnostic.test.mjs b/runner/pipeline/api-telemetry-diagnostic.test.mjs new file mode 100644 index 0000000000..db84bcf8c0 --- /dev/null +++ b/runner/pipeline/api-telemetry-diagnostic.test.mjs @@ -0,0 +1,185 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; + +// `reportDiagnostic`'s `sentryScopeIsFull` gate +// (`workers/api/src/telemetry/diagnostic.ts`) needs a direct test, not just +// the decision function (`sentryScopeIsFull` itself, in +// `api-telemetry-signals.test.mjs`) — the wiring that actually calls +// `Sentry.captureException` behind it needs coverage too. `diagnostic.ts` +// imports `./lines.js` and `./points.js` (sibling `.ts` files), which +// `--experimental-strip-types` cannot resolve through a `.js` specifier +// from a single-file import (see `resource.ts`'s own doc comment) — so +// this file uses the same copy-and-rewrite harness +// `pipeline/chat-sanitise.test.mjs` uses for `chat.ts`, applied to the +// `telemetry/` subtree instead. +// +// Unlike `chat.ts`, `diagnostic.ts` also has two bare package specifiers +// (`@sentry/cloudflare`, `@handsontable/demo-runtime/telemetry`), so the copy +// has to land somewhere Node's module resolution still walks up into +// `workers/api/node_modules` (both are real symlinks there) — an OS temp +// dir does not have that ancestry (confirmed: +// `Cannot find package '@sentry/cloudflare'`). The copy below lands inside +// `workers/api/` itself instead, and is removed after. +// +// `reportDiagnostic` takes an injectable `capture` function — the test +// below passes a recorder instead of the real `@sentry/cloudflare` call. + +const workersApiDir = join(import.meta.dirname, "..", "workers/api"); +const telemetrySrc = join(workersApiDir, "src/telemetry"); +const envSrc = join(workersApiDir, "src/env.ts"); +const dir = mkdtempSync(join(workersApiDir, ".hot-diagnostic-")); +// The copy-and-import below must not run as a plain top-level sequence +// with `rmSync` last — a failing import (a typo in one of the copied +// files, a resolution error) would skip the cleanup and leave the scratch +// directory behind for `git add -A` to pick up (see `.gitignore`'s own +// `workers/api/.hot-*` entry). `try/finally` guarantees the directory is +// always removed, whether the import below succeeds or throws. +let reportDiagnostic; +try { + // diagnostic.ts's own chain: diagnostic.ts -> lines.ts, points.ts, scope.ts + // -> resource.ts. `env.ts` is imported everywhere as `import type` only + // (erased by strip-types, never resolved), but is copied too so a stray + // value use would fail loudly instead of silently resolving to nothing. + for (const file of ["diagnostic.ts", "lines.ts", "points.ts", "scope.ts", "resource.ts"]) { + writeFileSync(join(dir, file), readFileSync(join(telemetrySrc, file), "utf8").replaceAll('.js"', '.ts"')); + } + writeFileSync(join(dir, "env.ts"), readFileSync(envSrc, "utf8")); + ({ reportDiagnostic } = await import(join(dir, "diagnostic.ts"))); +} finally { + rmSync(dir, { recursive: true, force: true }); +} + +// `getSink`/`emitPoint` inside `reportDiagnostic` would otherwise reach for +// the local ClickHouse sink (a real `fetch` to localhost:8123) — production +// env shape with no `RUNNER_EVENTS` binding hits `resource.ts`'s documented +// no-op sink instead, so this test touches the network not at all. +const ENV_FULL = { PREVIEW_HOST: "demos.handsontable.com", SENTRY_SCOPE: "full" }; +const ENV_UNCAUGHT = { PREVIEW_HOST: "demos.handsontable.com", SENTRY_SCOPE: "uncaught" }; + +function recorder() { + const calls = []; + const capture = (err, context) => calls.push({ err, context }); + return { calls, capture }; +} + +test("reportDiagnostic: SENTRY_SCOPE=full calls the injected capture once", () => { + const { calls, capture } = recorder(); + const err = new Error("boom"); + reportDiagnostic(ENV_FULL, err, { context: "test-site", routeClass: "api/test" }, capture); + assert.equal(calls.length, 1); + assert.equal(calls[0].err, err); +}); + +test("reportDiagnostic: SENTRY_SCOPE=uncaught calls the injected capture zero times", () => { + const { calls, capture } = recorder(); + reportDiagnostic(ENV_UNCAUGHT, new Error("boom"), { context: "test-site", routeClass: "api/test" }, capture); + assert.equal(calls.length, 0); +}); + +test("reportDiagnostic: absent SENTRY_SCOPE defaults to full (calls the capture)", () => { + const { calls, capture } = recorder(); + reportDiagnostic({ PREVIEW_HOST: "demos.handsontable.com" }, new Error("boom"), { context: "test-site", routeClass: "api/test" }, capture); + assert.equal(calls.length, 1); +}); + +test("reportDiagnostic: the captured context carries tags/fingerprint/level through", () => { + const { calls, capture } = recorder(); + reportDiagnostic(ENV_FULL, new Error("boom"), { + context: "test-site", + routeClass: "api/test", + tags: { upstream: "npm-registry" }, + sentryFingerprint: ["a", "b"], + level: "warning", + }, capture); + assert.deepEqual(calls[0].context, { + level: "warning", + tags: { upstream: "npm-registry" }, + fingerprint: ["a", "b"], + }); +}); + +test("reportDiagnostic: with no capture argument, does not throw (falls back to the real Sentry call)", () => { + // The default path (every real call site in index.ts/chat.ts/theme-ai.ts) + // — under `uncaught` scope the gate is closed before the real + // `Sentry.captureException` would ever run, so this is safe to exercise + // without an active Sentry client. + assert.doesNotThrow(() => reportDiagnostic(ENV_UNCAUGHT, new Error("boom"), { context: "test-site", routeClass: "api/test" })); +}); + +// The structured error line `reportDiagnostic` writes via `logErrorLine` +// must carry the contract-fingerprint under `hot.fingerprint` (contract +// §3 AE-only key) — otherwise the API worker's handled errors have no way +// to reach the §F.3 new-fingerprint registry once they arrive at the o11y +// worker as a worker-tenant OTLP export. +test("reportDiagnostic: the structured error line carries hot.fingerprint = fingerprint(context, message)", () => { + const lines = []; + const realConsoleError = console.error; + console.error = (...args) => lines.push(args.join(" ")); + try { + const { capture } = recorder(); + reportDiagnostic(ENV_UNCAUGHT, new Error("boom"), { context: "test-site", routeClass: "api/test" }, capture); + } finally { + console.error = realConsoleError; + } + assert.equal(lines.length, 1); + const parsed = JSON.parse(lines[0]); + assert.equal(parsed["log.kind"], "error"); + assert.equal(parsed.context, "test-site"); + assert.match(parsed["hot.fingerprint"], /^test-site:[0-9a-f]{16}$/); +}); + +// `index.ts`'s chat-gateway and theme-gateway `reportDiagnostic` calls +// live inside the main worker's `fetch` handler (a route match deep inside +// a ~2000-line switch), not something this suite can invoke directly +// without a full request/env — same constraint +// `pipeline/mcp-create.test.mjs`'s own "the update route calls +// isMcpCreated()" test documents for the same file. Structural, same +// style: the route source is read as text and the exact +// `sentryFingerprint` shape is asserted for each call site — a passing +// `reportDiagnostic` fingerprint-passthrough test elsewhere proves the +// function honours `sentryFingerprint` when given one; this proves each +// call site actually passes one. +test("the chat-gateway and theme-gateway reportDiagnostic calls set a status-grouped sentryFingerprint", () => { + const root = join(import.meta.dirname, ".."); + const source = readFileSync(join(root, "workers/api/src/index.ts"), "utf8"); + + const chatStart = source.indexOf('context: "chat-gateway"'); + assert.ok(chatStart > -1, "the chat-gateway reportDiagnostic call exists in index.ts"); + const chatCall = source.slice(chatStart, source.indexOf("});", chatStart)); + assert.match( + chatCall, + /sentryFingerprint:\s*\["litellm-gateway",\s*String\(err\.status\)\]/, + "chat-gateway must fingerprint by gateway + status, not by the default message-based grouping (which carries a unique request_id)", + ); + + const themeStart = source.indexOf('context: "theme-gateway"'); + assert.ok(themeStart > -1, "the theme-gateway reportDiagnostic call exists in index.ts"); + const themeCall = source.slice(themeStart, source.indexOf("});", themeStart)); + assert.match( + themeCall, + /sentryFingerprint:\s*\["litellm-gateway",\s*String\(err\.status\)\]/, + "theme-gateway must fingerprint by gateway + status too", + ); +}); + +// The scratch-directory copy+import above must not be a plain top-level +// sequence ending in a bare `rmSync` — a failing import would leave +// `workers/api/.hot-diagnostic-*` behind, ungitignored, for `git add -A` +// to pick up. Structural (the fix is the shape of this file's own +// top-level code, not something a runtime assertion can observe after the +// fact — the directory from a real run is already gone by the time any +// test() body runs, success or failure). +test("scratch-dir cleanup is wrapped in try/finally, and workers/api/.hot-* is gitignored", () => { + const selfSource = readFileSync(fileURLToPath(import.meta.url), "utf8"); + assert.match( + selfSource, + /try\s*\{[\s\S]*await import\(join\(dir, "diagnostic\.ts"\)\)[\s\S]*\}\s*finally\s*\{\s*rmSync\(dir,/, + "the copy+import block must be wrapped in try/finally, with rmSync(dir, ...) in the finally", + ); + + const gitignore = readFileSync(join(import.meta.dirname, "..", ".gitignore"), "utf8"); + assert.match(gitignore, /^workers\/api\/\.hot-\*$/m, "runner/.gitignore must ignore workers/api/.hot-* scratch dirs"); +}); diff --git a/runner/pipeline/api-telemetry-pool-gauge.test.mjs b/runner/pipeline/api-telemetry-pool-gauge.test.mjs new file mode 100644 index 0000000000..9c45b34654 --- /dev/null +++ b/runner/pipeline/api-telemetry-pool-gauge.test.mjs @@ -0,0 +1,111 @@ +// `pool.gauge` (`workers/api/src/telemetry/cron.ts`, reason `live`) must not +// count every `session-meter:` key in KV: a meter key outlives the +// container it fronts by `KV_METER_TTL_SECONDS` (24h, `budget.ts`), so a +// single stale 24h tail would read as pool pressure. This file proves +// `countLiveSessionMeters` shares the same awake/slept split +// `admin.ts#liveSessions`'s `awakeCount` uses (`session-listing.ts +// #classifyMeter`, via `admin.ts#readMeters`, the same KV scan the panel +// runs) rather than trusting key existence or inventing a second window. +// Run: node --experimental-strip-types --test pipeline/api-telemetry-pool-gauge.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import fs from "node:fs"; +import { fileURLToPath } from "node:url"; +import { fakeKV } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { countLiveSessionMeters, LIVE_POOL_MAX_INSTANCES } = await import("../workers/api/src/telemetry/cron.ts"); +const { AWAKE_WINDOW_SECONDS } = await import("../workers/api/src/session-listing.ts"); + +const now = 1_800_000_000_000; +const sec = (n) => n * 1000; + +/** Writes a meter the way `budget.ts#startSessionMeter`/`meterSessionUnsafe` do: + * value plus mirrored list metadata, so `admin.ts#readMeters` takes the fast + * (metadata-only) path instead of its legacy per-row `get` fallback. */ +async function seedMeter(cache, sessionId, startedAt, meteredThrough) { + await cache.put( + `session-meter:${sessionId}`, + JSON.stringify({ startedAt, meteredThrough, instanceType: "standard-1" }), + { metadata: { s: startedAt, m: meteredThrough, i: "standard-1" } }, + ); +} + +test("countLiveSessionMeters: a stale 24h meter plus one awake meter counts only the awake one", async () => { + const cache = fakeKV(); + // Stale: last ticked an hour ago, long past the idle window — the KV key is + // still alive (well inside its 24h TTL) but the container behind it slept. + await seedMeter(cache, "vue-stale00001", now - sec(20 * 3600), now - sec(3600)); + // Awake: the one live Tier-2 lane, ticked 30s ago. + await seedMeter(cache, "vue-awake00001", now - sec(120), now - sec(30)); + const env = { CACHE: cache }; + assert.equal(await countLiveSessionMeters(env, now), 1); +}); + +// Boundary: one meter exactly at the idle window (must count, inclusive — +// matches `classifyMeter`'s own `quietSeconds <= AWAKE_WINDOW_SECONDS` rule) +// and one meter one second past it (must not). A key-count-only +// implementation answers 2; a classifier with the boundary flipped to +// exclusive (`<` instead of `<=`) answers 0. Only the fix under test +// answers 1. +test("countLiveSessionMeters: the idle-window boundary is inclusive, same rule as classifyMeter", async () => { + const cache = fakeKV(); + await seedMeter(cache, "astro-atwindow1", now - sec(900), now - sec(AWAKE_WINDOW_SECONDS)); + await seedMeter(cache, "astro-pastwindow", now - sec(900), now - sec(AWAKE_WINDOW_SECONDS + 1)); + const env = { CACHE: cache }; + assert.equal(await countLiveSessionMeters(env, now), 1); +}); + +// --------------------------------------------------------------------------- +// `pool.gauge`'s `cap` (`LIVE_POOL_MAX_INSTANCES`) is a hard-coded constant, +// not read from config at runtime (wrangler does not expose +// `containers[].max_instances` to `env`), so it can drift silently from +// `Sandbox.max_instances` in wrangler.jsonc — the "Pool gauge vs cap" panel +// would then compare live sessions against the wrong ceiling with no error +// anywhere. This test parses the real wrangler.jsonc and pins the two +// together: change either number alone and this goes red. +// --------------------------------------------------------------------------- + +function stripLineComments(text) { + let out = ""; + let inString = false; + let escape = false; + for (let i = 0; i < text.length; i += 1) { + const c = text[i]; + if (inString) { + out += c; + if (escape) escape = false; + else if (c === "\\") escape = true; + else if (c === "\"") inString = false; + continue; + } + if (c === "\"") { + inString = true; + out += c; + continue; + } + if (c === "/" && text[i + 1] === "/") { + while (i < text.length && text[i] !== "\n") i += 1; + out += "\n"; + continue; + } + out += c; + } + return out; +} + +test("drift: LIVE_POOL_MAX_INSTANCES equals workers/api/wrangler.jsonc's Sandbox max_instances", () => { + const workersApiDir = fileURLToPath(new URL("../workers/api/", import.meta.url)); + const wranglerJsonc = fs.readFileSync(`${workersApiDir}wrangler.jsonc`, "utf8"); + const wrangler = JSON.parse(stripLineComments(wranglerJsonc)); + const sandbox = wrangler.containers.find((c) => c.class_name === "Sandbox"); + assert.ok(sandbox, "wrangler.jsonc must declare a Sandbox container"); + assert.equal( + LIVE_POOL_MAX_INSTANCES, + sandbox.max_instances, + "cron.ts's LIVE_POOL_MAX_INSTANCES must be updated in the same commit as wrangler.jsonc's Sandbox max_instances", + ); +}); diff --git a/runner/pipeline/api-telemetry-signals.test.mjs b/runner/pipeline/api-telemetry-signals.test.mjs new file mode 100644 index 0000000000..44fce6ba4f --- /dev/null +++ b/runner/pipeline/api-telemetry-signals.test.mjs @@ -0,0 +1,147 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { demoIdFromPath, routeClassOf, validSessionId } from "../workers/api/src/telemetry/route-class.ts"; +import { serviceEnvironment, serviceVersion } from "../workers/api/src/telemetry/resource.ts"; +import { sentryScopeIsFull } from "../workers/api/src/telemetry/scope.ts"; + +// Route classification (blob10 `route_class`, contract §4) and the +// service.version / deployment.environment.name resource attrs (contract §3). +// +// `route-class.ts` and `resource.ts` are deliberately import-free of their +// sibling `.ts` files (see resource.ts's own doc comment) so this file can +// import them directly under `--experimental-strip-types` — a sibling `.ts` +// file's compiled `.js` specifier does not resolve that way. + +test("routeClassOf: the exact case the acceptance criteria names", () => { + assert.equal(routeClassOf("GET", "/api/versions"), "api/versions"); +}); + +test("routeClassOf: dynamic ids collapse, not the whole path", () => { + assert.equal(routeClassOf("POST", "/api/session"), "api/session"); + assert.equal(routeClassOf("DELETE", "/api/session/react-18-abc123"), "api/session/:id"); + assert.equal(routeClassOf("GET", "/api/session/react-18-abc123/status"), "api/session/:id/status"); + assert.equal(routeClassOf("POST", "/api/session/react-18-abc123/file"), "api/session/:id/file"); +}); + +test("routeClassOf: shared demos and embeds", () => { + assert.equal(routeClassOf("GET", "/d/abc123"), "d/:id"); + assert.equal(routeClassOf("GET", "/embed/abc123"), "embed/:id"); +}); + +test("routeClassOf: versions/exists is distinct from versions", () => { + assert.equal(routeClassOf("GET", "/api/versions/exists"), "api/versions/exists"); +}); + +test("routeClassOf: chat/event is distinct from chat", () => { + assert.equal(routeClassOf("POST", "/api/chat"), "api/chat"); + assert.equal(routeClassOf("POST", "/api/chat/event"), "api/chat/event"); +}); + +test("routeClassOf: root and an unmatched top-level path", () => { + assert.equal(routeClassOf("GET", "/"), "root"); + assert.equal(routeClassOf("GET", "/robots.txt"), "other"); +}); + +test("routeClassOf: never throws on a malformed path", () => { + assert.doesNotThrow(() => routeClassOf("GET", "")); + assert.doesNotThrow(() => routeClassOf("GET", "//api//")); +}); + +test("demoIdFromPath: extracted for the routes that name one, empty otherwise", () => { + assert.equal(demoIdFromPath("/d/abc123"), "abc123"); + assert.equal(demoIdFromPath("/embed/abc123"), "abc123"); + assert.equal(demoIdFromPath("/api/versions"), ""); + assert.equal(demoIdFromPath("/api/session/sess-1/status"), ""); +}); + +// A raw, unresolved path segment must never reach `hot.demo_id` on the +// per-request log line — a crawler probing `/d/<garbage>` must not stuff +// arbitrary strings straight into Loki structured metadata. The +// `DEMO_ID_SHAPE_RE` check in `route-class.ts#demoIdFromPath` must reject +// it, not return the raw segment unconditionally. +test("demoIdFromPath: rejects a path segment that isn't shaped like a real demo id", () => { + assert.equal(demoIdFromPath("/d/';DROP TABLE demos;--"), ""); + assert.equal(demoIdFromPath("/d/<script>alert(1)</script>"), ""); + assert.equal(demoIdFromPath("/d/has spaces"), ""); + assert.equal(demoIdFromPath(`/d/${"a".repeat(200)}`), "", "implausibly long guesses are rejected"); + assert.equal(demoIdFromPath("/embed/../../etc/passwd"), "", "a literal '..' segment (path-traversal-shaped) is rejected"); +}); + +test("demoIdFromPath: still accepts every real id shape (shortId's own alphabet, and a hyphenated legacy fixed id)", () => { + assert.equal(demoIdFromPath("/d/abc123defg"), "abc123defg", "share.ts#shortId()'s own lowercase-base36 alphabet"); + assert.equal(demoIdFromPath("/d/react-18-legacy-id"), "react-18-legacy-id", "a hyphenated legacy fixed id"); + assert.equal(demoIdFromPath("/api/demos/abc123"), "abc123"); + assert.equal(demoIdFromPath("/api/mcp/demos/abc123"), "abc123"); +}); + +// `validSessionId` is what `index.ts`'s per-request log line calls instead +// of trusting the raw `x-hot-session` header. The `SESSION_ID_RE` check in +// `telemetry/lines.ts#validSessionId` must reject an arbitrary header +// value, not pass it through unchanged (`raw ?? ""`). +test("validSessionId: accepts a real crypto.randomUUID() page-load id", () => { + assert.equal(validSessionId("3fa85f64-5717-4562-b3fc-2c963f66afa6"), "3fa85f64-5717-4562-b3fc-2c963f66afa6"); +}); + +test("validSessionId: accepts the defensive plid-<base36>-<base36> fallback shape", () => { + assert.equal(validSessionId("plid-l3x9k2-9f2h1q"), "plid-l3x9k2-9f2h1q"); +}); + +test("validSessionId: rejects null, empty, and arbitrary client-controlled values", () => { + assert.equal(validSessionId(null), ""); + assert.equal(validSessionId(""), ""); + assert.equal(validSessionId("not-a-real-session-id"), ""); + assert.equal(validSessionId("'; DROP TABLE sessions;--"), ""); + assert.equal(validSessionId("a".repeat(500)), "", "an implausibly long header value is rejected"); +}); + +test("serviceEnvironment: production only under the real host", () => { + assert.equal(serviceEnvironment({ PREVIEW_HOST: "demos.handsontable.com" }), "production"); + assert.equal(serviceEnvironment({ PREVIEW_HOST: "localhost:8787" }), "local"); + assert.equal(serviceEnvironment({}), "local"); +}); + +test("serviceEnvironment: the check is equality, not a prefix or a suffix test", () => { + // Same rule sentry-gate.ts's apiSentryDsn already pins, both directions: a + // `.startsWith(PRODUCTION_HOST)` relaxation would still fail this first + // case, and (measured, not assumed — this exact case did NOT catch a + // `.endsWith(PRODUCTION_HOST)` mutation on a first draft of this test, + // caught only by adding the second case) an `.endsWith(PRODUCTION_HOST)` + // relaxation would pass the first case but fail the second. + assert.equal(serviceEnvironment({ PREVIEW_HOST: "demos.handsontable.com.evil.test" }), "local"); + assert.equal(serviceEnvironment({ PREVIEW_HOST: "evil-demos.handsontable.com" }), "local"); +}); + +test("serviceVersion: SERVICE_VERSION wins, then CF_VERSION_METADATA, then a literal fallback", () => { + assert.equal(serviceVersion({ SERVICE_VERSION: "abc123" }), "abc123"); + assert.equal(serviceVersion({ CF_VERSION_METADATA: { id: "v-id", tag: "" } }), "v-id"); + assert.equal(serviceVersion({}), "dev"); +}); + +test("serviceVersion: an empty --var (wrangler's own empty-string shape) still falls through", () => { + // `wrangler deploy --var SERVICE_VERSION:` yields "", exactly like + // SENTRY_ENVIRONMENT's documented empty-string case in sentry-gate.ts. + assert.equal(serviceVersion({ SERVICE_VERSION: "", CF_VERSION_METADATA: { id: "v-id", tag: "" } }), "v-id"); +}); + +// ── contract §11: the Sentry scope switch's decision ───────────────────────── +// +// `diagnostic.ts#reportDiagnostic` gates its `Sentry.captureException` call +// on this function's result +// (`if (sentryScopeIsFull(env)) { Sentry.captureException(...) }` — read +// directly in the source, since `diagnostic.ts` itself pulls in the real +// `@sentry/cloudflare` package). This is the decision alone; a live +// transport-spy integration check needs a running `wrangler dev`, where +// Sentry deliberately never initialises at all (sentry-gate.ts's own +// local-dev gate). + +test("sentryScopeIsFull: full (the default) is true", () => { + assert.equal(sentryScopeIsFull({ SENTRY_SCOPE: "full" }), true); +}); + +test("sentryScopeIsFull: uncaught is false", () => { + assert.equal(sentryScopeIsFull({ SENTRY_SCOPE: "uncaught" }), false); +}); + +test("sentryScopeIsFull: absent means full, exactly like leaving the wrangler.jsonc var out", () => { + assert.equal(sentryScopeIsFull({}), true); +}); diff --git a/runner/pipeline/browser-metrics.test.mjs b/runner/pipeline/browser-metrics.test.mjs new file mode 100644 index 0000000000..abc612e73b --- /dev/null +++ b/runner/pipeline/browser-metrics.test.mjs @@ -0,0 +1,531 @@ +// Observability contract §5 browser metric catalogue (ADR-0041 §F.2). +// Drives `apps/authoring/src/telemetry/metrics.ts` against a fake +// `DemoRuntime` and a `recordingTelemetry()`, replaying every call through +// the real `toAePoint` since `recordingTelemetry` validates nothing on its +// own. Real timing hooks are exercised separately in +// `sandpack-reload.test.mjs`/`session-start-failure.test.mjs`. +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. + +import test from "node:test"; +import assert from "node:assert/strict"; +import { recordingTelemetry, toAePoint } from "../packages/runtime/dist/telemetry/index.js"; +import { + COMPILE_TIMING_SETTLE_MS, + emitBucketResolve, + emitVersionSwitch, + flushCompileTimings, + htMajorOf, + startClock, + trackPreviewReady, + wireRuntimeMetrics, +} from "../apps/authoring/src/telemetry/metrics.ts"; + +const SERVICE = { service_name: "demos-authoring", service_version: "abc123", environment: "production" }; + +/** Replay every recorded `.metric()` call through the real `toAePoint` — the + * producer-contract check `recordingTelemetry` itself does not perform. Throws + * (failing the test) on a misspelled outcome, an attribute outside its metric's + * closed set, or an attribute with no AE slot. */ +function assertValidAgainstRegistry(telemetry) { + for (const { name, values, attrs } of telemetry.metrics) { + toAePoint(name, values, { ...SERVICE, ...attrs }); + } +} + +/** A minimal `DemoRuntime` stand-in: only `onReady`/`onError`, which is all + * `trackPreviewReady` reads. Exposes `fireReady`/`fireError` for the test to + * drive it, matching the real runtimes' "replay ready to a late subscriber" + * behaviour is NOT modelled here on purpose — `trackPreviewReady` subscribes + * once, synchronously, before either engine could have already settled. */ +function fakeRuntime() { + const readyCbs = []; + const errorCbs = []; + return { + onReady(cb) { + readyCbs.push(cb); + }, + onError(cb) { + errorCbs.push(cb); + }, + fireReady() { + for (const cb of readyCbs) cb(); + }, + fireError(e) { + for (const cb of errorCbs) cb(e); + }, + }; +} + +const CTX = { surface: "authoring", tier: 1, framework: "react", versionRef: "18.1.0", bucket: "18.1" }; + +test("preview.ready_ms: one point on ready, with the right attributes", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + runtime.fireReady(); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "preview.ready_ms"); + assert.equal(call.attrs.outcome, "ready"); + assert.equal(call.attrs.surface, "authoring"); + assert.equal(call.attrs.tier, "1"); + assert.equal(call.attrs.framework, "react"); + assert.equal(call.attrs.ht_major, "18"); + assert.equal(call.attrs.bucket, "18.1"); + assert.ok(typeof call.values.duration_ms === "number" && call.values.duration_ms >= 0); + assertValidAgainstRegistry(telemetry); +}); + +test("preview.ready_ms: a second onReady (a later recompile) does not re-emit", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + runtime.fireReady(); + runtime.fireReady(); + runtime.fireReady(); + + assert.equal(telemetry.metrics.length, 1, "guard: settled must latch after the first ready"); +}); + +test("preview.ready_ms: abandon() before ready emits outcome abandoned", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + const tracker = trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + tracker.abandon(); + + assert.equal(telemetry.metrics.length, 1); + assert.equal(telemetry.metrics[0].attrs.outcome, "abandoned"); + assertValidAgainstRegistry(telemetry); +}); + +test("preview.ready_ms: abandon() after ready is a no-op (no second point, no outcome flip)", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + const tracker = trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + runtime.fireReady(); + tracker.abandon(); + + assert.equal(telemetry.metrics.length, 1, "guard: abandon() must not fire once ready already settled"); + assert.equal(telemetry.metrics[0].attrs.outcome, "ready"); +}); + +test("preview.ready_ms: a mount() rejection observed via observe() reports outcome error, not abandoned", async () => { + // The case that motivates observe() at all (DEV-2130 / ContainerRuntime's + // dispose()-before-rethrow): a rejection that never reaches onError. + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + const tracker = trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + const mountPromise = Promise.reject(new Error("Setup failed")); + tracker.observe(mountPromise); + await mountPromise.catch(() => {}); + // Let the .catch() microtask inside trackPreviewReady settle too. + await Promise.resolve(); + + assert.equal(telemetry.metrics.length, 1); + assert.equal(telemetry.metrics[0].attrs.outcome, "error"); + + // A cleanup that runs after the rejection (the effect unmounting, or a version + // switch) must not turn this into "abandoned" — the guard is what this proves. + tracker.abandon(); + assert.equal(telemetry.metrics.length, 1, "guard: observe()'s error must latch before abandon() can fire"); + assertValidAgainstRegistry(telemetry); +}); + +test("preview.ready_ms: onError reports outcome error", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 50_000 }); + + runtime.fireError(new Error("boom")); + + assert.equal(telemetry.metrics.length, 1); + assert.equal(telemetry.metrics[0].attrs.outcome, "error"); +}); + +test("preview.ready_ms: a timeout reports outcome timeout, and a later ready is ignored", async () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + trackPreviewReady(runtime, CTX, telemetry, { timeoutMs: 5 }); + + await new Promise((resolve) => setTimeout(resolve, 40)); + assert.equal(telemetry.metrics.length, 1); + assert.equal(telemetry.metrics[0].attrs.outcome, "timeout"); + + runtime.fireReady(); + assert.equal(telemetry.metrics.length, 1, "guard: a ready arriving after the timeout must not re-emit"); +}); + +test("preview.ready_ms: a version with no ref attached reads ht_major as none", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeRuntime(); + trackPreviewReady(runtime, { ...CTX, versionRef: "" }, telemetry, { timeoutMs: 50_000 }); + + runtime.fireReady(); + + assert.equal(telemetry.metrics[0].attrs.ht_major, "none"); +}); + +// ---- htMajorOf -------------------------------------------------------------------- + +test("htMajorOf reads a release version's major, a next prerelease as next, and a pkg.pr.new ref as next", () => { + assert.equal(htMajorOf("18.1.0"), "18"); + assert.equal(htMajorOf("19.0.0-next.1"), "next"); + assert.equal(htMajorOf("0.0.0-next-abc123-20260101"), "next"); + assert.equal(htMajorOf("https://pkg.pr.new/handsontable/handsontable@7940"), "next"); + assert.equal(htMajorOf(null), "none"); + assert.equal(htMajorOf(undefined), "none"); +}); + +// ---- sandpack.compile_ms / compile_error / bundler_unreachable -------------------- + +function fakeSandpackRuntime() { + const compileTimingCbs = []; + const compileErrorCbs = []; + const bundlerUnreachableCbs = []; + return { + onCompileTiming(cb) { + compileTimingCbs.push(cb); + }, + onCompileError(cb) { + compileErrorCbs.push(cb); + }, + onBundlerUnreachable(cb) { + bundlerUnreachableCbs.push(cb); + }, + fireCompileTiming(e) { + for (const cb of compileTimingCbs) cb(e); + }, + fireCompileError(e) { + for (const cb of compileErrorCbs) cb(e); + }, + fireBundlerUnreachable(e) { + for (const cb of bundlerUnreachableCbs) cb(e); + }, + }; +} + +const SANDPACK_CTX = { framework: "vue", versionRef: "17.1.0" }; + +test("sandpack.compile_ms: reports the hook's own duration and outcome, tier fixed to 1", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileTiming({ durationMs: 900, outcome: "ok" }); // the mount + runtime.fireCompileTiming({ durationMs: 123, outcome: "ok" }); + clock.advance(COMPILE_TIMING_SETTLE_MS); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "sandpack.compile_ms"); + assert.equal(call.values.duration_ms, 123); + assert.equal(call.attrs.tier, "1"); + assert.equal(call.attrs.outcome, "ok"); + assert.equal(call.attrs.ht_major, "17"); + assertValidAgainstRegistry(telemetry); +}); + +/** A manual clock for the held compile point: `advance(ms)` fires what is due. */ +function manualTimers() { + let now = 0; + const timers = new Map(); + let nextId = 1; + return { + setTimer(fn, ms) { + const id = nextId++; + timers.set(id, { fn, at: now + ms }); + return id; + }, + clearTimer(id) { + timers.delete(id); + }, + advance(ms) { + now += ms; + for (const [id, t] of [...timers]) { + if (t.at <= now) { + timers.delete(id); + t.fn(); + } + } + }, + }; +} + +const compileTimes = (telemetry) => + telemetry.metrics.filter((m) => m.name === "sandpack.compile_ms").map((m) => [m.values.duration_ms, m.attrs.outcome]); + +test("sandpack.compile_ms: the compiles of one edit burst send one point, the burst's last", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileTiming({ durationMs: 900, outcome: "ok" }); // the mount + for (let i = 1; i <= 20; i += 1) { + runtime.fireCompileTiming({ durationMs: 100 + i, outcome: i === 7 ? "error" : "ok" }); + clock.advance(200); // one keystroke's compile every 200 ms + } + assert.deepEqual(compileTimes(telemetry), [], "the mount sends nothing, and the burst is held while compiles keep coming"); + + clock.advance(COMPILE_TIMING_SETTLE_MS); + assert.deepEqual(compileTimes(telemetry), [[120, "ok"]]); + assertValidAgainstRegistry(telemetry); + + runtime.fireCompileTiming({ durationMs: 77, outcome: "error" }); + clock.advance(COMPILE_TIMING_SETTLE_MS); + assert.deepEqual(compileTimes(telemetry).at(-1), [77, "error"], "a later burst gets its own point"); + assert.equal(compileTimes(telemetry).length, 2); +}); + +test("sandpack.compile_ms: keystrokes that fail to transpile keep the burst open", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileTiming({ durationMs: 900, outcome: "ok" }); + runtime.fireCompileTiming({ durationMs: 31, outcome: "ok" }); + for (let i = 0; i < 5; i += 1) { + clock.advance(1500); + runtime.fireCompileError({ message: "Unexpected token (1:7)", origin: "transpile" }); + } + clock.advance(1500); + runtime.fireCompileTiming({ durationMs: 42, outcome: "ok" }); + clock.advance(COMPILE_TIMING_SETTLE_MS); + assert.deepEqual(compileTimes(telemetry), [[42, "ok"]]); +}); + +test("sandpack.compile_ms: a held point is sent at once when the page is hidden, and only once", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileTiming({ durationMs: 900, outcome: "ok" }); + runtime.fireCompileTiming({ durationMs: 55, outcome: "ok" }); + flushCompileTimings(); + assert.deepEqual(compileTimes(telemetry), [[55, "ok"]]); + + clock.advance(COMPILE_TIMING_SETTLE_MS); + flushCompileTimings(); + assert.equal(compileTimes(telemetry).length, 1); +}); + +test("sandpack.compile_ms: the mount's compile sends no point, whatever its outcome", () => { + for (const outcome of ["ok", "error"]) { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileTiming({ durationMs: 900, outcome }); + clock.advance(COMPILE_TIMING_SETTLE_MS * 5); + flushCompileTimings(); + assert.deepEqual(compileTimes(telemetry), [], outcome); + } +}); + +test("sandpack.compile_ms: after a mount that failed the pre-transpile, the first compile is an edit's", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + const clock = manualTimers(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry, clock); + + runtime.fireCompileError({ message: "Unexpected token (1:7)", origin: "transpile" }); + runtime.fireCompileTiming({ durationMs: 64, outcome: "ok" }); + clock.advance(COMPILE_TIMING_SETTLE_MS); + assert.deepEqual(compileTimes(telemetry), [[64, "ok"]]); +}); + +test("sandpack.compile_error: fingerprinted, no authored text in the recorded attrs", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry); + + const message = "SyntaxError: Unexpected token (2:7) in /src/App.vue"; + runtime.fireCompileError({ message }); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "sandpack.compile_error"); + assert.match(call.attrs.fingerprint, /^sandpack\.compile_error:[0-9a-f]{16}$/); + // `HotAttrs` has no free-text field, so the message itself cannot travel even by + // accident — asserted anyway, against every attr value, as the guard for it. + for (const value of Object.values(call.attrs)) { + assert.ok(!String(value).includes("Unexpected token"), "no authored code in the recorded attrs"); + } + assertValidAgainstRegistry(telemetry); +}); + +test("sandpack.compile_error: the same fingerprint (a keystroke ladder) reports once, not once per keystroke", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry); + + runtime.fireCompileError({ message: "'t' is not defined" }); + runtime.fireCompileError({ message: "'tr' is not defined" }); + runtime.fireCompileError({ message: "'tru' is not defined" }); + + assert.equal( + telemetry.metrics.length, + 1, + "guard: the demo-runtime ladder collapse must dedupe these to one fingerprint", + ); +}); + +test("sandpack.compile_error: a genuinely different message gets its own point", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry); + + runtime.fireCompileError({ message: "'t' is not defined" }); + runtime.fireCompileError({ message: "Unexpected token }" }); + + assert.equal(telemetry.metrics.length, 2); + assert.notEqual(telemetry.metrics[0].attrs.fingerprint, telemetry.metrics[1].attrs.fingerprint); +}); + +test("sandpack.bundler_unreachable: reports duration, ht_major only", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeSandpackRuntime(); + wireRuntimeMetrics(runtime, SANDPACK_CTX, telemetry); + + runtime.fireBundlerUnreachable({ durationMs: 9001 }); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "sandpack.bundler_unreachable"); + assert.equal(call.values.duration_ms, 9001); + assert.equal(call.attrs.ht_major, "17"); + assertValidAgainstRegistry(telemetry); +}); + +// ---- session.start_ms / hmr.roundtrip_ms ----------------------------------------- + +function fakeContainerRuntime() { + const sessionStartCbs = []; + const hmrCbs = []; + return { + onSessionStart(cb) { + sessionStartCbs.push(cb); + }, + onHmr(cb) { + hmrCbs.push(cb); + }, + fireSessionStart(e) { + for (const cb of sessionStartCbs) cb(e); + }, + fireHmr(e) { + for (const cb of hmrCbs) cb(e); + }, + }; +} + +const CONTAINER_CTX = { framework: "next", versionRef: "18.2.0" }; + +test("session.start_ms: reports elapsed/outcome, and never sets reason (no cold/warm signal exists)", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeContainerRuntime(); + wireRuntimeMetrics(runtime, CONTAINER_CTX, telemetry); + + runtime.fireSessionStart({ elapsedMs: 4200, outcome: "ready" }); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "session.start_ms"); + assert.equal(call.values.duration_ms, 4200); + assert.equal(call.attrs.outcome, "ready"); + assert.equal(call.attrs.reason, undefined, "guard: reason must stay unset, not a guessed cold/warm"); + assertValidAgainstRegistry(telemetry); +}); + +test("session.start_ms: every session.start outcome value round-trips through toAePoint", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeContainerRuntime(); + wireRuntimeMetrics(runtime, CONTAINER_CTX, telemetry); + + for (const outcome of ["ready", "at_capacity", "container_starting", "boot_timeout", "budget_denied", "error"]) { + runtime.fireSessionStart({ elapsedMs: 1, outcome }); + } + + assert.equal(telemetry.metrics.length, 6); + assertValidAgainstRegistry(telemetry); +}); + +test("hmr.roundtrip_ms: reports the hook's own duration", () => { + const telemetry = recordingTelemetry(); + const runtime = fakeContainerRuntime(); + wireRuntimeMetrics(runtime, CONTAINER_CTX, telemetry); + + runtime.fireHmr({ durationMs: 87 }); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "hmr.roundtrip_ms"); + assert.equal(call.values.duration_ms, 87); + assert.equal(call.attrs.framework, "next"); + assert.equal(call.attrs.ht_major, "18"); + assertValidAgainstRegistry(telemetry); +}); + +// ---- version.switch / bucket.resolve_ms ------------------------------------------ + +test("version.switch: both to- and from-version go through the ht_major closed set, not the raw ref", () => { + const telemetry = recordingTelemetry(); + emitVersionSwitch(telemetry, { framework: "react", toRef: "18.1.0", fromRef: "17.1.0", bucket: "18.1" }); + + assert.equal(telemetry.metrics.length, 1); + const [call] = telemetry.metrics; + assert.equal(call.name, "version.switch"); + assert.equal(call.attrs.ht_major, "18"); + assert.equal(call.attrs.reason, "17"); + assert.equal(call.attrs.bucket, "18.1"); + assertValidAgainstRegistry(telemetry); +}); + +test("version.switch: a pkg.pr.new fromRef never lands raw in the reason blob (guard against unbounded AE data)", () => { + const telemetry = recordingTelemetry(); + emitVersionSwitch(telemetry, { + framework: "react", + toRef: "18.1.0", + fromRef: "https://pkg.pr.new/handsontable/handsontable@7940", + }); + + assert.equal(telemetry.metrics[0].attrs.reason, "next"); + assertValidAgainstRegistry(telemetry); +}); + +test("version.switch: an absent fromRef reads reason as none", () => { + const telemetry = recordingTelemetry(); + emitVersionSwitch(telemetry, { framework: "react", toRef: "18.1.0" }); + + assert.equal(telemetry.metrics[0].attrs.reason, "none"); + assertValidAgainstRegistry(telemetry); +}); + +test("bucket.resolve_ms: ok and error outcomes both round-trip", () => { + const telemetry = recordingTelemetry(); + emitBucketResolve(telemetry, { bucket: "18.1", outcome: "ok", durationMs: 12 }); + emitBucketResolve(telemetry, { bucket: "18.1", outcome: "error", durationMs: 34 }); + + assert.equal(telemetry.metrics.length, 2); + assert.equal(telemetry.metrics[0].values.duration_ms, 12); + assert.equal(telemetry.metrics[1].attrs.outcome, "error"); + assertValidAgainstRegistry(telemetry); +}); + +// ---- startClock --------------------------------------------------------------- + +test("startClock reports elapsed time against an injected clock", () => { + let now = 1000; + const elapsed = startClock(() => now); + now = 1042; + assert.equal(elapsed(), 42); +}); diff --git a/runner/pipeline/chat-answer-network-error.test.mjs b/runner/pipeline/chat-answer-network-error.test.mjs new file mode 100644 index 0000000000..64d5d902a5 --- /dev/null +++ b/runner/pipeline/chat-answer-network-error.test.mjs @@ -0,0 +1,88 @@ +// A network-level throw from the LiteLLM fetch in `/api/chat` must not +// vanish: `requestAnswer` (chat.ts) can reject with a real `TypeError` (DNS, +// refused, reset — never a `ChatUnavailableError`), and the route's catch +// must emit the `chat.answer` point for that throw too, not only inside its +// `if (err instanceof ChatUnavailableError)` branch — contract §5 promises a +// point on every outcome, `error` included, matching `theme.ai`'s twin catch. +// +// Driven through the real router (`workers/api/src/index.ts`'s default +// export) — a re-declared copy of the catch would not catch this regressing. +// Run: node --experimental-strip-types --test pipeline/chat-answer-network-error.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const HOST = "https://demos.handsontable.com"; + +// blob/double slot positions, §4 (AE_COLUMNS, packages/runtime/src/telemetry/metrics.ts): +// model=blob13 (index 12), outcome=blob8 (index 7), count=double1 (index 0). +const MODEL_SLOT = 12; +const OUTCOME_SLOT = 7; +const COUNT_SLOT = 0; + +function envWithPointCapture() { + const points = []; + const { env, ...rest } = makeEnv(); + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + // Routes AE writes through the in-memory sink instead of the local-ClickHouse + // HTTP fallback `serviceEnvironment` selects for a non-production host — see + // the identical comment on `snapshot-build-point.test.mjs`'s own helper. + env.PREVIEW_HOST = "demos.handsontable.com"; + env.LITELLM_API_KEY = "test-key"; + return { env, points, ...rest }; +} + +function chatAnswerPoints(points) { + return points.filter((p) => p.indexes[0] === "chat.answer"); +} + +test("a network-level throw from the LiteLLM fetch still emits chat.answer outcome=error", async () => { + const { env, points } = envWithPointCapture(); + + const realFetch = globalThis.fetch; + globalThis.fetch = async (input) => { + const url = typeof input === "string" ? input : input.url; + if (url.includes("litellm")) { + // The exact failure mode this test exists for: `fetch()` itself + // rejects (connection refused / DNS / reset), never resolving to a + // Response — so `requestAnswer`'s `if (!res.ok)` branch (the one that + // throws `ChatUnavailableError`) is never reached at all. + throw new TypeError("fetch failed: network connection lost"); + } + throw new Error(`unexpected network fetch in chat-answer-network-error.test.mjs: ${url}`); + }; + + try { + const res = await worker.fetch( + new Request(`${HOST}/api/chat`, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + messages: [{ role: "user", content: "what does this option do?" }], + framework: "react", + files: { "App.jsx": "export default function App() { return null; }" }, + }), + }), + env, + ctx, + ); + // The raw throw is not a ChatUnavailableError, so it still falls through + // to the generic fetch catch-all (a 500) — that part of the behaviour is + // pre-existing and out of scope here. The point is what to fix. + assert.equal(res.status, 500); + + const points_ = chatAnswerPoints(points); + assert.equal(points_.length, 1, `expected exactly 1 chat.answer point, got ${points_.length}`); + assert.equal(points_[0].blobs[OUTCOME_SLOT], "error"); + assert.equal(points_[0].blobs[MODEL_SLOT], "unknown"); + assert.equal(points_[0].doubles[COUNT_SLOT], 1); + } finally { + globalThis.fetch = realFetch; + } +}); diff --git a/runner/pipeline/ci-workflow.test.mjs b/runner/pipeline/ci-workflow.test.mjs new file mode 100644 index 0000000000..3ec4498506 --- /dev/null +++ b/runner/pipeline/ci-workflow.test.mjs @@ -0,0 +1,69 @@ +// Structural pins for `ci.yml`'s `e2e-telemetry` job: the telemetry leak +// checks and actionlint must run in PR CI, not only post-merge, plus the +// negative control that proves `check:telemetry-leak` can actually fail. +// Same rationale/pattern as `pipeline/master-workflow.test.mjs`'s +// structural pins on `master.yml`. +// Run: node --experimental-strip-types --test pipeline/ci-workflow.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const workflowPath = join(__dirname, "..", "..", ".github", "workflows", "ci.yml"); +const source = readFileSync(workflowPath, "utf8"); + +function jobBody(name) { + const start = source.indexOf(`\n ${name}:`); + assert.ok(start > -1, `the ${name} job must exist`); + // Next top-level job starts at a line beginning with exactly two spaces + // then a word then ':' — find the next such line after this job's header. + const rest = source.slice(start + 1); + const nextMatch = /\n [a-zA-Z0-9_-]+:\n/.exec(rest.slice(1)); + return nextMatch ? rest.slice(0, nextMatch.index + 1) : rest; +} + +const e2eTelemetry = jobBody("e2e-telemetry"); + +test("ci.yml: e2e-telemetry runs actionlint", () => { + // B-4: matching bare /actionlint/i passes even if the real step is + // deleted, because the step's own preceding comment ("actionlint here + // too, so a workflow-YAML mistake...") also contains the word + // "actionlint" — this only pins the STEP itself (its `uses:` line), + // which a comment can never satisfy. + assert.match( + e2eTelemetry, + /uses:\s*reviewdog\/action-actionlint@v1/, + "the job must run the reviewdog/action-actionlint step, not just mention it in a comment", + ); +}); + +test("ci.yml: e2e-telemetry builds a production-mode bundle and runs both leak checks against it, before building the flag bundle", () => { + const prodBuildIdx = e2eTelemetry.indexOf("Build authoring in production mode"); + assert.ok(prodBuildIdx > -1, "a production-mode build step must exist"); + + const devBypassIdx = e2eTelemetry.indexOf("dev-login bypass must not reach the production bundle"); + assert.ok(devBypassIdx > prodBuildIdx, "the AGENTS.md dev-bypass leak check must run after the production build"); + + const telemetryLeakIdx = e2eTelemetry.indexOf("pnpm check:telemetry-leak"); + assert.ok(telemetryLeakIdx > prodBuildIdx, "check:telemetry-leak must run against the production build"); + + const flagBuildIdx = e2eTelemetry.indexOf("VITE_TELEMETRY_LOCAL: '1'"); + assert.ok(flagBuildIdx > telemetryLeakIdx, "the flag build must come after the production-mode leak checks"); +}); + +test("ci.yml: e2e-telemetry has a negative control asserting check:telemetry-leak FAILS against the flag build", () => { + const flagBuildIdx = e2eTelemetry.indexOf("VITE_TELEMETRY_LOCAL: '1'"); + const negativeControlIdx = e2eTelemetry.indexOf("Leak check negative control"); + assert.ok(negativeControlIdx > flagBuildIdx, "the negative control step must run after the flag build"); + + const stepEnd = e2eTelemetry.indexOf("\n - name:", negativeControlIdx + 1); + const step = e2eTelemetry.slice(negativeControlIdx, stepEnd > -1 ? stepEnd : undefined); + assert.match( + step, + /if pnpm check:telemetry-leak; then[\s\S]*exit 1/, + "the negative control must fail the job (exit 1) if check:telemetry-leak PASSES against the flag build", + ); +}); diff --git a/runner/pipeline/container-boot-ms.test.mjs b/runner/pipeline/container-boot-ms.test.mjs new file mode 100644 index 0000000000..900faf33da --- /dev/null +++ b/runner/pipeline/container-boot-ms.test.mjs @@ -0,0 +1,165 @@ +// `container.boot_ms` must be emitted from `POST /api/session`'s own +// create path, not only for `outcome: "window_exceeded"` from a DO fetch +// override reachable on a later proxied preview request — otherwise the +// `tier2-sessions` dashboard panel ("container.boot_ms p95 by outcome") +// stays permanently empty even on a healthy deploy, since its +// `blob6 IN (${framework:sqlstring})` filter has nothing to match when +// every point carries `blob6 = ''`. +// +// This file pins the create path's two outcomes — `ready` (the +// `withSpan("container.boot", …)` block resolves) and `error` (it throws, +// past the `at_capacity`/`container_starting` refusals, which already have +// their own `session.start` outcome and must not be double-counted here) — +// and that a refusal or an early throw emits neither. +// +// Route-tested with the same harness `session-create-container-starting +// .test.mjs` established for `POST /api/session`. `container.boot_ms` +// requires flipping `getSink()` to its `bindingSink` branch (see +// `pipeline/lite-inject.test.mjs#makeCountingEnv`'s own doc comment): +// `worker-harness.mjs#makeEnv` otherwise routes Analytics Engine points at +// a local ClickHouse HTTP fetch with nothing listening, which `emitPoint` +// — by design — swallows on failure. + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; + +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; +import { setSandboxFactory } from "./fixtures/cloudflare-sandbox-stub.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const FILES = { + "package.json": JSON.stringify({ name: "demo" }), + "src/App.jsx": "export default function App() { return null; }", +}; + +const sessionRequest = (body) => + new Request("https://demos.handsontable.com/api/session", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body), + }); + +/** See `pipeline/lite-inject.test.mjs#makeCountingEnv`: flips `getSink()` to + * its `bindingSink` branch (a real `RUNNER_EVENTS` fake) instead of the + * local-mode ClickHouse HTTP fetch, which has nothing to talk to here. */ +function countingEnv() { + const { env } = makeEnv(); + const points = []; + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, points }; +} + +const bootPoints = (points) => points.filter((p) => p.indexes[0] === "container.boot_ms"); + +function fakeSandbox({ startProcessError, mkdirError, writeFileError } = {}) { + const calls = { mkdir: 0, writeFile: 0, startProcess: 0, exposePort: 0, setFramework: 0 }; + const frameworks = []; + return { + calls, + frameworks, + async mkdir() { + calls.mkdir += 1; + if (mkdirError) throw mkdirError; + }, + async writeFile() { + calls.writeFile += 1; + if (writeFileError) throw writeFileError; + }, + deleteFile: async () => {}, + exec: async () => ({ success: true, stdout: "", stderr: "" }), + async startProcess() { + calls.startProcess += 1; + if (startProcessError) throw startProcessError; + }, + async exposePort() { + calls.exposePort += 1; + return { url: "https://preview.test/session" }; + }, + async setFramework(framework) { + calls.setFramework += 1; + frameworks.push(framework); + }, + destroy: async () => {}, + }; +} + +test("the happy path emits exactly one container.boot_ms ready, with the framework", async () => { + const { env, points } = countingEnv(); + const sandbox = fakeSandbox(); + setSandboxFactory(() => sandbox); + + const res = await worker.fetch(sessionRequest({ framework: "react-js", files: FILES }), env, ctx); + assert.equal(res.status, 200); + + const boot = bootPoints(points); + assert.equal(boot.length, 1, "expected exactly one container.boot_ms point"); + assert.equal(boot[0].blobs[7], "ready", "blob8 outcome"); + assert.equal(boot[0].blobs[5], "react-js", "blob6 framework — what the tier2-sessions panel filters on"); + assert.ok(boot[0].doubles[1] >= 0, "double2 duration_ms"); +}); + +test("a startProcess throw emits exactly one container.boot_ms error, with the framework", async () => { + const { env, points } = countingEnv(); + const sandbox = fakeSandbox({ startProcessError: new Error("boom: disk full") }); + setSandboxFactory(() => sandbox); + + const res = await worker.fetch(sessionRequest({ framework: "react-js", files: FILES }), env, ctx); + assert.equal(res.status, 500, "an unrecognised throw is not degraded"); + + const boot = bootPoints(points); + assert.equal(boot.length, 1, "expected exactly one container.boot_ms point"); + assert.equal(boot[0].blobs[7], "error", "blob8 outcome"); + assert.equal(boot[0].blobs[5], "react-js", "blob6 framework"); +}); + +test("a throw before the boot span even starts (writeFiles) emits no container.boot_ms point", async () => { + const { env, points } = countingEnv(); + // `writeFile` itself throws — still upstream of + // `bootStartedAt = Date.now()` / `withSpan("container.boot", …)`, which + // only run once every file write has succeeded. + const sandbox = fakeSandbox({ writeFileError: new Error("disk quota exceeded") }); + setSandboxFactory(() => sandbox); + + const res = await worker.fetch(sessionRequest({ framework: "react-js", files: FILES }), env, ctx); + assert.equal(res.status, 500); + assert.equal(sandbox.calls.startProcess, 0, "the boot span must never have started"); + + assert.equal(bootPoints(points).length, 0, "a pre-boot failure is not a container.boot_ms outcome"); +}); + +test("a container-starting refusal emits session.start's own outcome, not a second container.boot_ms error", async () => { + const { env, points } = countingEnv(); + // The SDK's own retry-exhausted 503, surfaced through mkdir (same fixture + // shape as session-create-container-starting.test.mjs). + const sandbox = fakeSandbox({ + mkdirError: new Error("Container is starting. Please retry in a moment."), + }); + setSandboxFactory(() => sandbox); + + const res = await worker.fetch(sessionRequest({ framework: "react-js", files: FILES }), env, ctx); + assert.equal(res.status, 503); + + const sessionStart = points.filter((p) => p.indexes[0] === "session.start"); + assert.equal(sessionStart.length, 1); + assert.equal(sessionStart[0].blobs[7], "container_starting"); + + // The at-capacity/container-starting refusals are classified from the + // create's OWN try, before `withSpan("container.boot", …)` — `bootStartedAt` + // is still null, so no `container.boot_ms` point of either outcome exists. + // Double-counting a refusal already carried by `session.start` would corrupt + // the panel's error rate. + assert.equal(bootPoints(points).length, 0, "container-starting must not also emit container.boot_ms"); +}); + +// A budget denial (`budgetGate`) returns before `startSessionMeter`/ +// `writeFiles` even run — earlier than the mkdir/EACCES throw above, which +// already proves the general invariant this depends on: `bootStartedAt` is +// only ever set immediately before `withSpan("container.boot", …)`, so any +// return/throw upstream of it — a denial included — emits no +// `container.boot_ms` point of either outcome. Not pinned as its own case: +// out of scope here. diff --git a/runner/pipeline/demo-event-collapse.test.mjs b/runner/pipeline/demo-event-collapse.test.mjs new file mode 100644 index 0000000000..8973bb9d8b --- /dev/null +++ b/runner/pipeline/demo-event-collapse.test.mjs @@ -0,0 +1,477 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + createDemoEventCollapse, + DEMO_EDIT_SETTLE_MS, + DEMO_COLLAPSE_CEILING, +} from "../apps/authoring/src/demoEventCollapse.ts"; +import { demoEventReport } from "../apps/authoring/src/demoEventReport.ts"; +import { fingerprint, fingerprintShape } from "../packages/runtime/dist/telemetry/index.js"; + +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build` (for +// the real §7 `fingerprint`/`fingerprintShape`, so the ladder below is keyed +// exactly as `sentry.ts` keys it). +// +// Typing one throwing line into a Tier-1 editor must not relay ~20 +// `preview.runtime_error` points (one per half-typed prefix) — the collapse +// turns that into one point per edit burst, while a non-edit error still +// counts immediately. + +/** A hand-driven timer: `advance(ms)` fires whatever came due. */ +function fakeTimers() { + let now = 0; + let nextId = 1; + const timers = new Map(); + return { + setTimer(fn, ms) { + const id = nextId++; + timers.set(id, { at: now + ms, fn }); + return id; + }, + clearTimer(id) { + timers.delete(id); + }, + advance(ms) { + now += ms; + for (const [id, t] of [...timers]) { + if (t.at <= now) { + timers.delete(id); + t.fn(); + } + } + }, + pending: () => timers.size, + }; +} + +function harness(extra = {}) { + const clock = fakeTimers(); + const emitted = []; + const collapse = createDemoEventCollapse({ + emit: (item) => emitted.push(item), + setTimer: clock.setTimer, + clearTimer: clock.clearTimer, + ...extra, + }); + /** Relay a message the way `sentry.ts` does: keyed by its real fingerprint. */ + const relay = (message) => collapse.report(fingerprint("demo-runtime", message), message); + return { clock, emitted, collapse, relay }; +} + +// The real ladder shape for `setTimeout(() => { throw new Error('R5RUNTIME'); }, 100);` +// typed one key at a time: several DIFFERENT fingerprints (a ReferenceError +// ladder, syntax errors, the final throw), which is why fingerprint dedupe +// alone cannot collapse it. +const LADDER = [ + "s is not defined", + "se is not defined", + "set is not defined", + "setT is not defined", + "setTi is not defined", + "setTim is not defined", + "setTime is not defined", + "setTimeo is not defined", + "setTimeou is not defined", + "Unexpected end of input", + "Unexpected token ')'", + "Unexpected end of input", + "Unterminated string constant", + "missing ) after argument list", + "Unexpected end of input", +]; +const FINAL = "R5RUNTIME boom"; + +test("a keystroke prefix ladder emits exactly one point: the error the finished line throws", () => { + const { clock, emitted, collapse, relay } = harness(); + for (const message of LADDER) { + collapse.noteEdit(); + clock.advance(120); // typing speed, well inside the settle window + relay(message); + } + collapse.noteEdit(); // the last keystroke + clock.advance(150); + relay(FINAL); + assert.equal(emitted.length, 0, "nothing is emitted while the user is still typing"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL]); + // The distinct fingerprints above prove it is the burst rule, not fingerprint + // dedupe, that removed the rungs. + assert.ok(new Set(LADDER.map((m) => fingerprint("demo-runtime", m))).size > 3); +}); + +test("a ladder whose finished line is clean emits nothing", () => { + const { clock, emitted, collapse, relay } = harness(); + for (const message of LADDER) { + collapse.noteEdit(); + clock.advance(100); + relay(message); + } + collapse.noteEdit(); // the keystroke that completes a valid line: no error follows + clock.advance(DEMO_EDIT_SETTLE_MS * 3); + assert.deepEqual(emitted, []); +}); + +test("a first-load error (no edit) emits one point immediately, and repeats of it do not add more", () => { + const { clock, emitted, relay } = harness(); + relay("Cannot read properties of undefined (reading 'getData')"); + assert.equal(emitted.length, 1, "no settle wait outside an edit burst"); + relay("Cannot read properties of undefined (reading 'getData')"); + clock.advance(DEMO_EDIT_SETTLE_MS * 2); + relay("Cannot read properties of undefined (reading 'getData')"); + assert.equal(emitted.length, 1); +}); + +test("two distinct persistent errors emit two points — one burst each", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + relay("first broken thing"); + clock.advance(DEMO_EDIT_SETTLE_MS); + collapse.noteEdit(); + relay("second broken thing"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["first broken thing", "second broken thing"]); +}); + +test("two distinct persistent errors from the same final run emit two points", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + relay("first broken thing"); + relay("second broken thing"); + relay("first broken thing"); // a re-render of the same fault + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["first broken thing", "second broken thing"]); +}); + +test("a persistent error is counted once per edit burst, again after the next burst", () => { + const { clock, emitted, collapse, relay } = harness(); + relay("still broken"); // first load + collapse.noteEdit(); // an edit elsewhere in the file, the fault survives it + relay("still broken"); + clock.advance(DEMO_EDIT_SETTLE_MS); + relay("still broken"); // a click re-throwing it, same burst window + assert.equal(emitted.length, 2); +}); + +test("an error landing after the burst closed (a slow Tier-2 rebuild) is emitted immediately, once", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + clock.advance(DEMO_EDIT_SETTLE_MS + 5000); + relay("vite build failed"); + relay("vite build failed"); + assert.deepEqual(emitted, ["vite build failed"]); +}); + +test("reset counts the outgoing preview's last run and re-arms first-load counting", () => { + const { clock, emitted, collapse, relay } = harness(); + relay("same shape"); // first load of example A + collapse.noteEdit(); + relay("pending at switch"); + collapse.reset(); // switch to example B before the burst closed + assert.deepEqual(emitted, ["same shape", "pending at switch"]); + assert.equal(clock.pending(), 0, "the settle timer is cancelled, not left to double-fire"); + relay("same shape"); // example B's first load, same fingerprint as A's + assert.equal(emitted.length, 3); +}); + +test("reset alone (no edit in between) re-arms first-load counting for the next preview", () => { + const { emitted, collapse, relay } = harness(); + relay("theme not found"); // example A's first load + relay("theme not found"); // A re-renders: same burst window, not counted again + collapse.reset(); // switch to example B + relay("theme not found"); // B's first load hits the same fault + assert.deepEqual(emitted, ["theme not found", "theme not found"]); +}); + +test("the ceiling bounds a demo posting ever-different payloads with no edit", () => { + const { emitted, relay } = harness(); + for (let i = 0; i < DEMO_COLLAPSE_CEILING + 30; i++) relay(`crafted ${"x".repeat(i)}`); + assert.equal(emitted.length, DEMO_COLLAPSE_CEILING); +}); + +test("fingerprintShape is what fingerprint() hashes, so the Faro record and the metric agree", () => { + for (const message of [ + ...LADDER, + FINAL, + "Cannot read properties of undefined (reading 'getData') at https://x.test/a.js?t=1", + "Invalid language tag: zh-c", + "hot.getData is not a function", + "Unexpected token (2:11)\n 1 | function f() {\n> 2 | return x +;\n | ^\n 3 | }", + ]) { + const shape = fingerprintShape(message); + assert.equal(fingerprint("demo-runtime", shape), fingerprint("demo-runtime", message), message); + } + // The shape is not the raw message: quoted text, numbers, URLs and code + // frames (authored code, contract §3) are gone. + const shape = fingerprintShape( + "Unexpected token (2:11)\n> 2 | return secretVar +;\n | ^ at 'literal' https://x.test/?k=1", + ); + assert.ok(!shape.includes("secretVar"), shape); + assert.ok(!shape.includes("literal"), shape); + assert.ok(!shape.includes("x.test"), shape); +}); + +test("demoEventReport names the Faro record by kind, and gives a console warning none", () => { + const base = { message: "m", tier: 1, framework: "react", htMajor: "18" }; + assert.equal(demoEventReport({ ...base, kind: "error" }).recordName, "DemoError"); + assert.equal(demoEventReport({ ...base, kind: "rejection" }).recordName, "DemoUnhandledRejection"); + assert.equal(demoEventReport({ ...base, kind: "console-error" }).recordName, "DemoConsoleError"); + assert.equal(demoEventReport({ ...base, kind: "console-warn" }).recordName, null); + assert.equal(demoEventReport({ ...base, kind: "network" }).recordName, "DemoNetworkError"); + assert.equal(demoEventReport({ ...base, kind: "stderr" }).recordName, "DemoStderr"); +}); + +// ---- a compile failure replaces the burst's run ---------------------------- +// +// The key `sentry.ts#collapseCompileError` uses: by kind, not by message. +const COMPILE_KEY = "compile:sandpack.compile_error"; + +function compileHarness() { + const h = harness(); + /** A compile failure of the newest edit, the way `collapseCompileError` reports it. */ + const compileError = (diagnostic) => h.collapse.report(COMPILE_KEY, `compile: ${diagnostic}`, { replacesRun: true }); + return { ...h, compileError }; +} + +test("a typed syntax-error ladder is one compile error and no runtime error, stale relays included", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + // `const X = ;` typed key by key. `c`..`cons` parse and run (and throw); + // from `const` on, every prefix fails the pre-transpile. + for (const prefix of ["c", "co", "con", "cons"]) { + collapse.noteEdit(); + relay(`${prefix} is not defined`); + } + collapse.noteEdit(); // `const` + // The `cons` run's relay was still in flight at this keystroke (compile + // slower than the typist) — a known imprecision. + relay("cons is not defined"); + compileError("Unexpected token (1:5)"); + for (const diagnostic of ["Unexpected token (1:6)", "Missing initializer in const declaration", "Unexpected token (1:12)"]) { + collapse.noteEdit(); + compileError(diagnostic); + // A re-render warning / late rung that lands after the compile failure. + relay('Theme "main" is already registered.'); + } + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: Unexpected token (1:12)"], "the final state's compile error, alone"); +}); + +test("a compile error replaces an earlier one of the same burst, so the final state's diagnostic is the one counted", () => { + const { clock, emitted, collapse, compileError } = compileHarness(); + collapse.noteEdit(); + compileError("stale diagnostic from the previous push"); + compileError("the newest push's diagnostic"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: the newest push's diagnostic"]); +}); + +test("a burst that ends compiling cleanly counts its run's runtime error, not the earlier compile error", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); + compileError("Unexpected token"); + collapse.noteEdit(); // the line is finished and parses; it throws when it runs + relay(FINAL); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL]); +}); + +test("a runtime SyntaxError (JSON.parse) stays a runtime error — only the compile signal replaces a run", () => { + const { clock, emitted, collapse, relay } = compileHarness(); + const jsonParse = "SyntaxError: Unexpected token } in JSON at position 1"; + collapse.noteEdit(); + relay(jsonParse); + // The same run's next fault: were the SyntaxError taken for a compile + // failure (message-shape detection), it would suppress this one. + relay(FINAL); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [jsonParse, FINAL]); + // And outside a burst (a click that parses bad JSON), at once. + relay("SyntaxError: Unexpected end of JSON input"); + assert.deepEqual(emitted, [jsonParse, FINAL, "SyntaxError: Unexpected end of JSON input"]); +}); + +test("a first-load compile failure counts at once, and only once until the next edit", () => { + const { emitted, collapse, compileError } = compileHarness(); + compileError("Unexpected token"); + assert.deepEqual(emitted, ["compile: Unexpected token"], "no burst open: not held back"); + compileError("Unexpected token"); // a refresh of the same broken demo + assert.equal(emitted.length, 1); + collapse.reset(); // the next preview mount + compileError("Unexpected token"); + assert.equal(emitted.length, 2, "a new mount counts its own first-load failure"); +}); + +test("the next edit re-arms runtime reports after a compile failure", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); + compileError("Unexpected token"); + clock.advance(DEMO_EDIT_SETTLE_MS); + // Burst closed: a click in the stale preview that throws still counts. + relay("stale preview click"); + assert.deepEqual(emitted, ["compile: Unexpected token", "stale preview click"]); +}); + +test("a stale relay held before the final keystroke's compile failure is dropped by it", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); // the last keystroke of the line + relay("cons is not defined"); // the previous run, still in flight + compileError("Unexpected token (1:12)"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: Unexpected token (1:12)"]); +}); + +// An edit whose sandbox matches the running one (a closing `;`, a space, a +// trailing comma) re-runs nothing, so no later report replaces what that +// edit's `noteEdit` discarded. + +test("a burst ending on an edit that re-runs nothing counts the running sandbox's error once", () => { + const { clock, emitted, collapse, relay } = harness(); + for (const message of LADDER) { + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay(message); + } + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay(FINAL); // the finished line's run + collapse.noteEdit(); // the closing `;` + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL], "the final run's error, and none of the rungs"); +}); + +test("a compile failure undone back to the running sandbox counts that sandbox's error, not the compile error", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay(FINAL); + collapse.noteEdit(); // a stray `(` + compileError("Unexpected token"); + relay(FINAL); // the running sandbox, still throwing: suppressed while the newest edit is broken + collapse.noteEdit(); // deleted again: identical to what runs + collapse.pushOutcome("unchanged"); + relay("the running sandbox's next fault"); // no longer suppressed: the newest edit compiles + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL, "the running sandbox's next fault"]); +}); + +test("an edit that re-runs nothing does not count the running sandbox's already-counted error again", () => { + const { clock, emitted, collapse, relay } = harness(); + relay(FINAL); // first load, counted at once + collapse.noteEdit(); // a space + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay("next run's error"); + clock.advance(DEMO_EDIT_SETTLE_MS); + collapse.noteEdit(); + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL, "next run's error"]); +}); + +test("a rerun forgets the previous sandbox's reports, so 'unchanged' brings back only the new run's", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay("old run's error"); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); // the finished line runs clean + collapse.noteEdit(); + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, []); +}); + +test("a newest edit that fails to compile still counts one compile error, and 'unchanged' is never its outcome", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay("cons is not defined"); + collapse.noteEdit(); + compileError("Unexpected token (1:12)"); + relay("cons is not defined"); // the running sandbox's late relay + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: Unexpected token (1:12)"]); +}); + +test("reset forgets the outgoing preview's running sandbox", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay("outgoing preview's error"); + collapse.noteEdit(); // typed past, then the preview is switched away mid-burst + collapse.reset(); + collapse.noteEdit(); // the new preview's first edit matches its mounted sandbox + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, []); +}); + +test("a report already counted before a rerun is not brought back by a later unchanged edit", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + relay(FINAL); // the previous run's relay + clock.advance(DEMO_EDIT_SETTLE_MS); // counted: the burst closes before the new run starts (a slow install) + collapse.pushOutcome("rerun"); + relay(FINAL); // the new run's own copy: already counted + collapse.noteEdit(); // a space + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL]); +}); + +test("a bundler compile error of the running sandbox survives an unchanged edit and still replaces its run", () => { + const { clock, emitted, collapse, relay } = harness(); + const bundlerError = (d) => collapse.report(COMPILE_KEY, `compile: ${d}`, { replacesRun: true, fromBundler: true }); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + relay("imp is not defined"); // a stale rung that lands after the dispatch + bundlerError("Could not find module './missing.css'"); + relay("im is not defined"); // an older rung, later still + collapse.noteEdit(); // the closing `;` + collapse.pushOutcome("unchanged"); + relay("imp is not defined"); // still stale: the rejected sandbox never evaluated + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: Could not find module './missing.css'"]); +}); + +test("a rerun forgets the previous sandbox's bundler compile error, and records the new run's reports", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + collapse.report(COMPILE_KEY, "compile: bundler", { replacesRun: true, fromBundler: true }); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); // the fixed import builds, runs and throws + relay(FINAL); + collapse.noteEdit(); + collapse.pushOutcome("unchanged"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, [FINAL]); +}); + +// The bundler runs one compile at a time and signals `rerun` when it starts the next, +// so a run's relay can land after the next keystroke but before that keystroke's run. + +test("a previous run's relay that lands after the newest edit, before that edit's run starts, is not counted", () => { + const { clock, emitted, collapse, relay } = harness(); + collapse.noteEdit(); + collapse.pushOutcome("rerun"); + collapse.noteEdit(); // the last keystroke + relay("typed er"); // the previous run evaluates only now + collapse.pushOutcome("rerun"); // the bundler starts the finished line + relay("typed err"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["typed err"]); +}); + +test("the start of an older run does not drop the newest edit's compile failure", () => { + const { clock, emitted, collapse, relay, compileError } = compileHarness(); + collapse.noteEdit(); // `cons`, dispatched + collapse.noteEdit(); // `const`, which does not parse + compileError("Unexpected token (1:5)"); + collapse.pushOutcome("rerun"); // the bundler starts `cons` + relay("cons is not defined"); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(emitted, ["compile: Unexpected token (1:5)"]); +}); diff --git a/runner/pipeline/demo-event-report.test.mjs b/runner/pipeline/demo-event-report.test.mjs new file mode 100644 index 0000000000..976e6f3541 --- /dev/null +++ b/runner/pipeline/demo-event-report.test.mjs @@ -0,0 +1,141 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { demoEventReport } from "../apps/authoring/src/demoEventReport.ts"; +import { fingerprint } from "../packages/runtime/dist/telemetry/index.js"; + +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build` (for +// the `fingerprint` import used to prove the ladder-collapsing claim below). +// +// ADR §E.1 "Moves to the new stack only": demo-runtime preview events must +// leave Sentry entirely and become one `preview.runtime_error` count +// through the facade. `demoEventReport.ts` is the pure decision +// `sentry.ts#reportDemoEvent` delegates to — see its own header for why it +// is import-free. + +test("error and rejection both map to reason 'uncaught'", () => { + const facts = { kind: "error", message: "boom", tier: 1, framework: "react", htMajor: "18" }; + assert.equal(demoEventReport(facts).reason, "uncaught"); + assert.equal(demoEventReport({ ...facts, kind: "rejection" }).reason, "uncaught"); +}); + +test("console-error maps to reason 'console'; a console warning is not a runtime error at all", () => { + const facts = { kind: "console-error", message: "a real console.error", tier: 1, framework: "react", htMajor: "18" }; + assert.equal(demoEventReport(facts).reason, "console"); + // Handsontable's load-time notices (18: theme already registered; 17: `date` deprecation). + for (const message of ['Theme "main" is already registered. Registration skipped.', "Deprecated: The `date` cell type ..."]) { + assert.equal(demoEventReport({ ...facts, kind: "console-warn", message }).reason, null, message); + } +}); + +test("network maps to 'network', stderr maps to 'stderr'", () => { + assert.equal( + demoEventReport({ kind: "network", message: "m", tier: 1, framework: "react", htMajor: "18" }).reason, + "network", + ); + assert.equal( + demoEventReport({ kind: "stderr", message: "m", tier: 2, framework: "angular", htMajor: "17" }).reason, + "stderr", + ); +}); + +test("only console-warn is budgeted against the looser breadcrumb cap", () => { + const kinds = ["error", "rejection", "console-error", "network", "stderr"]; + for (const kind of kinds) { + assert.equal( + demoEventReport({ kind, message: "m", tier: 1, framework: "react", htMajor: "18" }).budget, + "relay", + `kind=${kind}`, + ); + } + assert.equal( + demoEventReport({ kind: "console-warn", message: "m", tier: 1, framework: "react", htMajor: "18" }).budget, + "breadcrumb", + ); +}); + +test("attrs carry surface=demo-runtime, the stringified tier, framework, and ht_major", () => { + const report = demoEventReport({ kind: "error", message: "m", tier: 2, framework: "vue", htMajor: "17" }); + assert.deepEqual(report.attrs, { surface: "demo-runtime", tier: "2", framework: "vue", ht_major: "17" }); +}); + +// The `tier1-playground` dashboard's "runtime_error rate by reason" panel +// filters `blob7 IN (${ht_major:sqlstring})`; without `ht_major` on the +// attrs every row lands blob7 = '' and the panel is permanently empty. +test("ht_major carries the caller's actual value through to attrs (contract §5 / F10a)", () => { + for (const htMajor of ["15", "16", "17", "18", "19", "next", "none"]) { + const report = demoEventReport({ kind: "error", message: "m", tier: 1, framework: "react", htMajor }); + assert.equal(report.attrs.ht_major, htMajor, `htMajor=${htMajor}`); + } +}); + +test("demoId, when present, becomes attrs.demo_id — and is omitted when absent/null", () => { + const withId = demoEventReport({ + kind: "error", + message: "m", + tier: 1, + framework: "react", + htMajor: "18", + demoId: "abc123", + }); + assert.equal(withId.attrs.demo_id, "abc123"); + + const withoutId = demoEventReport({ kind: "error", message: "m", tier: 1, framework: "react", htMajor: "18" }); + assert.equal("demo_id" in withoutId.attrs, false); + + const nullId = demoEventReport({ + kind: "error", + message: "m", + tier: 1, + framework: "react", + htMajor: "18", + demoId: null, + }); + assert.equal("demo_id" in nullId.attrs, false); +}); + +test("fingerprintContext is always 'demo-runtime' — the surface, never the kind", () => { + for (const kind of ["error", "rejection", "console-error", "console-warn", "network", "stderr"]) { + assert.equal( + demoEventReport({ kind, message: "m", tier: 1, framework: "react", htMajor: "18" }).fingerprintContext, + "demo-runtime", + ); + } +}); + +test("fingerprintMessage is the raw message, unnormalised — the caller runs fingerprint()", () => { + const report = demoEventReport({ + kind: "error", + message: "licenseKey is not defined", + tier: 1, + framework: "react", + htMajor: "18", + }); + assert.equal(report.fingerprintMessage, "licenseKey is not defined"); +}); + +// "A demo-runtime keystroke ladder becomes one deduplicated count in +// Faro" — proven here as "the same fingerprint," via the real contract +// `fingerprint()`, not a re-implementation. +test("a keystroke ladder collapses to one fingerprint (the actual contract dedupe)", () => { + const ladder = ["l is not defined", "li is not defined", "lic is not defined", "licenseKey is not defined"]; + const fingerprints = ladder.map((message) => { + const report = demoEventReport({ kind: "error", message, tier: 1, framework: "react", htMajor: "18" }); + return fingerprint(report.fingerprintContext, report.fingerprintMessage); + }); + assert.equal(new Set(fingerprints).size, 1, "every ladder rung must fingerprint identically"); + + // A genuinely different failure must NOT collapse into the same bucket — + // guards against a fingerprint function that has gone trivial (e.g. a + // constant), which would make the assertion above pass for the wrong reason. + const different = demoEventReport({ + kind: "error", + message: "Cannot read properties of undefined (reading 'foo')", + tier: 1, + framework: "react", + htMajor: "18", + }); + assert.notEqual( + fingerprint(different.fingerprintContext, different.fingerprintMessage), + fingerprints[0], + ); +}); diff --git a/runner/pipeline/demo-routes-version.test.mjs b/runner/pipeline/demo-routes-version.test.mjs index d1add1c80d..d261dd8a3d 100644 --- a/runner/pipeline/demo-routes-version.test.mjs +++ b/runner/pipeline/demo-routes-version.test.mjs @@ -203,7 +203,7 @@ test("a browser rebuild derives from the payload pin and replaces a stale sentin const { env, writes, artifacts } = makeEnv([demoRow({ ht_version: "latest" })]); const res = await worker.fetch(patchRequest("abc123", { files: filesWith("16.0.2") }), env, ctx); assert.equal(res.status, 200); - assert.deepEqual(await res.json(), { ok: true, htVersion: "16.0.2" }); + assert.deepEqual(await res.json(), { ok: true, htVersion: "16.0.2", exampleSaved: false }); const update = findVersionUpdate(writes); assert.ok(update, "the rebuild must update the demos row"); assert.equal(update.binds[0], "16.0.2", "the sentinel row is repaired to the derived ref"); diff --git a/runner/pipeline/demo-save-build-error.test.mjs b/runner/pipeline/demo-save-build-error.test.mjs new file mode 100644 index 0000000000..4209b998bc --- /dev/null +++ b/runner/pipeline/demo-save-build-error.test.mjs @@ -0,0 +1,205 @@ +// A Save or create whose build rejects the demo's own input is client input: 422 with +// the build error, an `api.request` 4xx, a `snapshot.build failed` point, and the stored +// demo untouched. A failure that is ours stays a 5xx. Driven through the real router +// with a scripted builder container. +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/demo-save-build-error.test.mjs + +import test, { after } from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { AUTHOR, SECRET, demoRow, makeEnv, seedCatalog } from "./fixtures/worker-harness.mjs"; +import { setSandboxFactory } from "./fixtures/cloudflare-sandbox-stub.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const REAL_FETCH = globalThis.fetch; +globalThis.fetch = async (input, init) => { + const url = typeof input === "string" ? input : input.url; + if (url.startsWith("https://login.invalid") && init?.headers?.Authorization === "Bearer test-token") { + return Response.json({ email: AUTHOR, sub: "u1" }); + } + throw new Error(`unexpected network fetch in demo-save-build-error.test.mjs: ${url}`); +}; +after(() => { + globalThis.fetch = REAL_FETCH; + setSandboxFactory(null); +}); + +const DEMO_ID = "abc123"; +const INDEX_HTML = + '<!doctype html><html><body><div id="root"></div>' + + '<script type="module" src="/src/index.tsx"></script></body></html>'; +const FILES = { + "/package.json": JSON.stringify({ name: "demo", dependencies: { handsontable: "16.0.2" }, devDependencies: { vite: "^5.4.0" } }), + "/index.html": INDEX_HTML, + "/src/index.tsx": "const X = ;\n", +}; + +/** The stderr of a `vite build` that rejects a syntax error. Its announced cause is + * only headings, so the useful line is the one after them. */ +const SYNTAX_ERROR_LOG = + "vite v7.1.0 building for production...\ntransforming...\nerror during build:\n" + + "Build failed with 1 error:\n/app/src/index.tsx:1:10: ERROR: Unexpected \";\"\n"; + +/** A builder whose install succeeds and whose build command answers `build`. */ +function builder(build) { + return () => ({ + mkdir: async () => {}, + writeFile: async () => {}, + readFile: async () => "", + destroy: async () => {}, + async exec(cmd) { + if (cmd.includes("pnpm install")) return { success: true, exitCode: 0, stdout: "", stderr: "" }; + return build(cmd); + }, + }); +} + +const rejectsCode = builder(() => ({ success: false, exitCode: 1, stdout: "", stderr: SYNTAX_ERROR_LOG })); + +/** The route's env with a build-cache miss (so the builder runs) and points in memory. */ +function setup(rows = [demoRow({ id: DEMO_ID, framework: "react", ht_version: "16.0.2" })]) { + const harness = makeEnv(rows, [], {}, { buildCacheHit: false }); + const { env } = harness; + const points = []; + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + const pending = []; + const ctx = { waitUntil: (p) => pending.push(Promise.resolve(p)), passThroughOnException() {} }; + /** Every point of `metric`, once the work scheduled past the response (which + * itself schedules more) has settled. */ + const pointsOf = async (metric) => { + for (let seen = -1; seen !== pending.length;) { + seen = pending.length; + await Promise.allSettled(pending); + } + return points.filter((p) => p.indexes[0] === metric); + }; + return { ...harness, ctx, pointsOf }; +} + +const authed = { "Content-Type": "application/json", Authorization: "Bearer test-token" }; +const mcpHeaders = { "Content-Type": "application/json", "X-MCP-Secret": SECRET, "X-Demo-Author": AUTHOR }; + +const request = (method, path, headers, body) => + new Request(`https://demos.handsontable.com${path}`, { method, headers, body: JSON.stringify(body) }); + +const ROUTES = [ + ["PATCH /api/demos/:id (editor Save)", () => request("PATCH", `/api/demos/${DEMO_ID}`, authed, { files: FILES, htVersion: "16.0.2" })], + ["POST /api/demos (fork, embed)", () => request("POST", "/api/demos", authed, { framework: "react", files: FILES, title: "Grid", htVersion: "16.0.2" })], + ["PATCH /api/mcp/demos/:id", () => request("PATCH", `/api/mcp/demos/${DEMO_ID}`, mcpHeaders, { files: FILES, htVersion: "16.0.2" })], + ["POST /api/mcp/demos", () => request("POST", "/api/mcp/demos", mcpHeaders, { framework: "react", files: FILES, title: "Grid", description: "A grid", htVersion: "16.0.2" })], +]; + +/** What `api.request` recorded for the one request: its outcome blob. */ +async function requestOutcomes(pointsOf) { + return (await pointsOf("api.request")).map((p) => p.blobs.find((b) => /^[2-5]xx$/.test(b))); +} + +for (const [name, makeRequest] of ROUTES) { + test(`${name}: code the build rejects is a 422 build_failed with the build error, recorded as 4xx`, async () => { + setSandboxFactory(rejectsCode); + const { env, ctx, pointsOf, writes, artifacts, demos } = setup(); + await seedCatalog(env); + const before = JSON.stringify(demos.get(DEMO_ID)); + + const res = await worker.fetch(makeRequest(), env, ctx); + + assert.equal(res.status, 422); + const detail = 'error during build: Build failed with 1 error: src/index.tsx:1:10: ERROR: Unexpected ";"'; + // `error` carries the diagnostic too: MCP clients (hot-mcp) read only that field. + assert.deepEqual(await res.json(), { error: `build failed: ${detail}`, code: "build_failed", detail }); + assert.deepEqual(await requestOutcomes(pointsOf), ["4xx"], "one api.request point, and it is not a 5xx"); + const builds = await pointsOf("snapshot.build"); + assert.equal(builds.length, 1); + assert.ok(builds[0].blobs.includes("failed"), "snapshot.build keeps its failed outcome"); + // The demo is unchanged: no artifact, no source snapshot, no row written. + assert.deepEqual(artifacts.puts.filter((p) => p.key.startsWith("demos/")), []); + assert.deepEqual(writes.filter((w) => /\bdemos\b/.test(w.sql) && !/build_cache/.test(w.sql)), []); + assert.equal(JSON.stringify(demos.get(DEMO_ID)), before); + }); +} + +/** A builder whose install answers `install` (the frozen install and its retry alike). */ +function installer(install) { + return () => ({ + mkdir: async () => {}, + writeFile: async () => {}, + readFile: async () => "", + destroy: async () => {}, + exec: async () => install(), + }); +} + +const failedInstall = (stderr) => installer(() => ({ success: false, exitCode: 1, stdout: "", stderr })); + +const USER_INSTALL_FAILURES = [ + ["ERR_PNPM_NO_MATCHING_VERSION", " ERR_PNPM_NO_MATCHING_VERSION No matching version found for dayjs@^99\n"], + ["ERR_PNPM_FETCH_404", " ERR_PNPM_FETCH_404 GET https://registry.npmjs.org/dayjss: Not Found - 404\n"], + ["ERR_PNPM_SPEC_NOT_SUPPORTED_BY_ANY_RESOLVER", " ERR_PNPM_SPEC_NOT_SUPPORTED_BY_ANY_RESOLVER dayjs@latest-ish isn't supported by any available resolver.\n"], + ["ERR_PNPM_BAD_PM_VERSION", " ERR_PNPM_BAD_PM_VERSION This project is configured to use v8 of pnpm. Your current pnpm is v10.34.5\n"], +]; + +for (const [code, stderr] of USER_INSTALL_FAILURES) { + test(`an install refused with ${code} (the author's dependency) is a 422 build_failed with the pnpm error`, async () => { + setSandboxFactory(failedInstall(stderr)); + const { env, ctx, pointsOf } = setup(); + await seedCatalog(env); + const res = await worker.fetch(ROUTES[0][1](), env, ctx); + assert.equal(res.status, 422); + const body = await res.json(); + assert.equal(body.code, "build_failed"); + assert.ok(body.detail.includes(code), body.detail); + assert.equal(body.error, `install failed: ${body.detail}`); + assert.deepEqual(await requestOutcomes(pointsOf), ["4xx"]); + }); +} + +test("an MCP client reading only `error` gets the build diagnostic", async () => { + setSandboxFactory(rejectsCode); + const { env, ctx } = setup(); + await seedCatalog(env); + for (const [, makeRequest] of ROUTES.filter(([name]) => name.includes("/api/mcp/"))) { + const res = await worker.fetch(makeRequest(), env, ctx); + assert.equal(res.status, 422); + assert.match((await res.json()).error, /src\/index\.tsx:1:10: ERROR: Unexpected ";"/); + } +}); + +const INFRA_FAILURES = [ + ["a build killed by a signal (OOM)", builder(() => ({ success: false, exitCode: 137, stdout: "", stderr: "Killed\n" }))], + ["a build command that is not executable (126)", builder(() => ({ success: false, exitCode: 126, stdout: "", stderr: "sh: vite: Permission denied\n" }))], + ["a build command that is not found (127)", builder(() => ({ success: false, exitCode: 127, stdout: "", stderr: "sh: vite: not found\n" }))], + ["a build that exits 1 on a network failure", builder(() => ({ + success: false, + exitCode: 1, + stdout: "", + stderr: "`next/font` error:\nFailed to fetch `Inter` from Google Fonts.\nTypeError: fetch failed\n", + }))], + ["a build that exits 1 after its worker ran out of memory", builder(() => ({ + success: false, + exitCode: 1, + stdout: "", + stderr: "FATAL ERROR: Reached heap limit Allocation failed - JavaScript heap out of memory\nerror during build:\nBuild failed\n", + }))], + ["an install that fails on a registry 5xx", failedInstall(" ERR_PNPM_FETCH_503 GET https://registry.npmjs.org/dayjs: Service Unavailable - 503\n")], + ["an install that fails on a reset connection", failedInstall(" ERR_PNPM_META_FETCH_FAIL GET https://registry.npmjs.org/dayjs: request to https://registry.npmjs.org/dayjs failed, reason: read ECONNRESET\n")], + ["an install that times out", failedInstall(" ERR_PNPM_META_FETCH_FAIL GET https://registry.npmjs.org/dayjs: request to https://registry.npmjs.org/dayjs failed, reason: connect ETIMEDOUT 104.16.0.35:443\n")], + ["a build result without an exit code", builder(() => ({ success: false, stdout: "", stderr: "error during build:\nsomething\n" }))], + ["an exec that throws (container lost)", builder(() => { throw new Error("container is not running"); })], +]; + +for (const [name, factory] of INFRA_FAILURES) { + test(`an editor Save that fails on ${name} stays a 5xx`, async () => { + setSandboxFactory(factory); + const { env, ctx, pointsOf } = setup(); + await seedCatalog(env); + const res = await worker.fetch(ROUTES[0][1](), env, ctx); + assert.equal(res.status, 500); + assert.notEqual((await res.json()).error, "build_failed"); + assert.deepEqual(await requestOutcomes(pointsOf), ["5xx"]); + }); +} diff --git a/runner/pipeline/demos-malformed-json.test.mjs b/runner/pipeline/demos-malformed-json.test.mjs new file mode 100644 index 0000000000..ba9f034e19 --- /dev/null +++ b/runner/pipeline/demos-malformed-json.test.mjs @@ -0,0 +1,102 @@ +// `POST /api/demos` and `PATCH /api/demos/:id` must not call +// `request.json()` with no `.catch()`: an unparseable body (or a JSON +// array where an object is expected) would throw a raw `SyntaxError`/hit +// `body.framework` on a non-object past the handler, into the generic +// fetch catch-all — a 500 on ordinary client garbage that pollutes the +// `api.request` 5xx rate and the `api-5xx-rate` alert, same class the +// `/api/session` fix (session-malformed-json.test.mjs) closed for the +// public, unauthenticated routes. These two are authenticated, but the +// body itself is still ordinary client input reachable by anything with a +// token. Reuses the same helper (`isPlainRecord`) and error shape as the +// session fix. +// +// Driven through the real router (`workers/api/src/index.ts`'s default +// export) — a hand-rolled re-check of the body would not catch a +// regression in the actual route. +// Run: node --experimental-strip-types --test pipeline/demos-malformed-json.test.mjs + +import test, { after } from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { AUTHOR, ctx, demoRow, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +// ---- the broker stub / network tripwire (same shape as demo-routes-version.test.mjs) ---- + +const REAL_FETCH = globalThis.fetch; + +globalThis.fetch = async (input, init) => { + const url = typeof input === "string" ? input : input.url; + if (url.startsWith("https://login.invalid") && init?.headers?.Authorization === "Bearer test-token") { + return Response.json({ email: AUTHOR, sub: "u1" }); + } + throw new Error(`unexpected network fetch in demos-malformed-json.test.mjs: ${url}`); +}; + +after(() => { + globalThis.fetch = REAL_FETCH; +}); + +const authHeaders = { + "Content-Type": "application/json", + Authorization: "Bearer test-token", +}; + +function malformedJsonRequest(method, path) { + return new Request(`https://demos.handsontable.com${path}`, { + method, + headers: authHeaders, + // Not valid JSON — `request.json()` rejects on this body. + body: "{ this is not json", + }); +} + +function arrayBodyRequest(method, path) { + return new Request(`https://demos.handsontable.com${path}`, { + method, + headers: authHeaders, + // Valid JSON — parses fine — but not a plain record: `isPlainRecord` + // must still reject it, same as the session fix's own array-body case. + body: JSON.stringify([1, 2, 3]), + }); +} + +// ---- POST /api/demos -------------------------------------------------------------- + +test("a malformed POST /api/demos body is a 400, not the fetch catch-all's 500", async () => { + const { env } = makeEnv(); + const res = await worker.fetch(malformedJsonRequest("POST", "/api/demos"), env, ctx); + assert.equal(res.status, 400, "must not reach the generic 500 catch-all"); + const body = await res.json(); + // Same error shape the /api/session fix uses (isPlainRecord's own 400). + assert.equal(body.error, "request body must be a plain record"); +}); + +test("a JSON array body on POST /api/demos is a 400, not a 500", async () => { + const { env } = makeEnv(); + const res = await worker.fetch(arrayBodyRequest("POST", "/api/demos"), env, ctx); + assert.equal(res.status, 400, "an array is valid JSON but not a plain record"); + const body = await res.json(); + assert.equal(body.error, "request body must be a plain record"); +}); + +// ---- PATCH /api/demos/:id ---------------------------------------------------------- + +test("a malformed PATCH /api/demos/:id body is a 400, not the fetch catch-all's 500", async () => { + const { env } = makeEnv([demoRow()]); + const res = await worker.fetch(malformedJsonRequest("PATCH", "/api/demos/abc123"), env, ctx); + assert.equal(res.status, 400, "must not reach the generic 500 catch-all"); + const body = await res.json(); + assert.equal(body.error, "request body must be a plain record"); +}); + +test("a JSON array body on PATCH /api/demos/:id is a 400, not a 500", async () => { + const { env } = makeEnv([demoRow()]); + const res = await worker.fetch(arrayBodyRequest("PATCH", "/api/demos/abc123"), env, ctx); + assert.equal(res.status, 400, "an array is valid JSON but not a plain record"); + const body = await res.json(); + assert.equal(body.error, "request body must be a plain record"); +}); diff --git a/runner/pipeline/dev-script.test.mjs b/runner/pipeline/dev-script.test.mjs new file mode 100644 index 0000000000..c6937cdc35 --- /dev/null +++ b/runner/pipeline/dev-script.test.mjs @@ -0,0 +1,2119 @@ +// Unit tests for `runner/scripts/dev-lib.mjs` (the shared logic behind +// `pnpm dev`/`dev:live`/`dev:full` and `pnpm o11y:dev`) plus a small +// CLI-level test for the Docker-missing failure path and a drift test +// pinning the script's own env-var/flag surface to +// docs/run-and-deploy.md's "Run locally" section. +// +// No real `wrangler`/`docker`/`vite` is spawned here — port resolution, +// bootstrap, and migration-tracking logic are exercised directly (with a +// real temp directory for file I/O, and injected stub functions for +// anything that would otherwise shell out), following this repo's own +// `pipeline/fixtures/stub-bin` pattern for the one place a real subprocess +// is worth spawning (the Docker-missing CLI message). +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, rmSync, writeFileSync, readFileSync, mkdirSync, existsSync, utimesSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { spawnSync } from "node:child_process"; + +import { + HELP_TEXT, + parseArgs, + resolvePorts, + PORT_DEFAULTS, + assertNoPortCollisions, + bootstrapDevVars, + o11yDevVarsPatch, + O11Y_DEVVARS_STRIP_KEYS, + fillEmptyDevVarsSecrets, + O11Y_DEVVARS_AUTOFILL_SECRET_KEYS, + resolveReplaySecret, + checkDevVarsPortDrift, + resolveDevVarsPortAdoption, + checkO11yDevVarsStaleness, + readDevVarsLine, + migrationRecordPath, + planMigrations, + applyMigrations, + readAppliedMigrations, + parseMigrationTargets, + isMigrationAlreadyApplied, + snapshotLocalSchema, + MigrationError, + formatMigrationError, + resetLocalD1, + isDockerAvailable, + DOCKER_NOT_RUNNING_MESSAGE, + isRuntimeDistStale, + isPnpmInstallNeeded, + PNPM_INSTALL_NEEDED_MESSAGE, + wranglerBuildErrorLine, + waitForServer, + buildPlan, + ephemeralSecret, + redactArgsForLog, + possiblyLeftoverContainers, + reportLeftoverContainers, + SHUTDOWN_SIGNALS, + o11yLocalPublicOrigin, +} from "../scripts/dev-lib.mjs"; +// dev-persist task's own additions — a separate import statement so a +// parallel edit to the block above merges cleanly. +import { + composeDownArgs, + o11yDevDataModeLine, + resetO11yLocalState, + bringUpO11yCompose, + o11yLedgerCommittedKeyCount, + findComposeVolume, + detectO11yStateDivergence, + formatO11yDivergenceWarning, + defaultComposeProjectName, + resolveComposeProjectName, +} from "../scripts/dev-lib.mjs"; +// dev-prepull task's own additions — a separate import statement so a +// parallel edit to the blocks above merges cleanly. +import { + readContainerDockerfilePaths, + parseDockerfileBaseImages, + containerWranglerConfigsForTier, + collectTierBaseImages, + shouldCheckContainerImages, + isImagePresent, + pullImageWithRetry, + ensureContainerImagesPresent, + formatImagePullFailure, +} from "../scripts/dev-lib.mjs"; +import { DatabaseSync } from "node:sqlite"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const RUNNER_ROOT = path.join(HERE, ".."); + +function withTmpDir(fn) { + const dir = mkdtempSync(path.join(tmpdir(), "dev-script-test-")); + const cleanup = () => rmSync(dir, { recursive: true, force: true }); + let result; + try { + result = fn(dir); + } catch (err) { + cleanup(); + throw err; + } + if (result && typeof result.then === "function") { + // `fn` is async — cleanup must wait for it to settle, or the temp dir + // gets deleted out from under a still-pending write (the `finally` + // version of this helper deletes as soon as `fn(dir)` RETURNS a + // promise, not once it resolves). + return result.then( + (value) => { + cleanup(); + return value; + }, + (err) => { + cleanup(); + throw err; + }, + ); + } + cleanup(); + return result; +} + +// --------------------------------------------------------------------------- +// arg parsing +// --------------------------------------------------------------------------- + +test("parseArgs: --tier=1|2|full are accepted", () => { + assert.equal(parseArgs(["--tier=1"]).tier, "1"); + assert.equal(parseArgs(["--tier=2"]).tier, "2"); + assert.equal(parseArgs(["--tier=full"]).tier, "full"); +}); + +test("parseArgs: an invalid --tier value is a parse error, not a silent fallback", () => { + const { tier, errors } = parseArgs(["--tier=3"]); + assert.equal(tier, null); + assert.ok(errors.some((e) => e.includes("--tier"))); +}); + +test("parseArgs: --tier is required unless --help is passed", () => { + const noTier = parseArgs([]); + assert.ok(noTier.errors.some((e) => e.includes("required"))); + const help = parseArgs(["--help"]); + assert.equal(help.help, true); + assert.deepEqual(help.errors, []); +}); + +test("parseArgs: -h is recognized the same as --help", () => { + assert.equal(parseArgs(["-h"]).help, true); +}); + +test("parseArgs: --replay is only valid with --tier=full", () => { + assert.deepEqual(parseArgs(["--tier=full", "--replay"]).errors, []); + const withTier1 = parseArgs(["--tier=1", "--replay"]); + assert.ok(withTier1.errors.some((e) => e.includes("--replay"))); +}); + +test("parseArgs: an unrecognized flag is an error", () => { + const { errors } = parseArgs(["--tier=1", "--bogus"]); + assert.ok(errors.some((e) => e.includes("--bogus"))); +}); + +// --------------------------------------------------------------------------- +// port resolution +// --------------------------------------------------------------------------- + +test("resolvePorts: tier=1 resolves only AUTHORING_DEV_PORT, at its documented default", () => { + const ports = resolvePorts("1", {}); + assert.deepEqual(ports, { AUTHORING_DEV_PORT: PORT_DEFAULTS.AUTHORING_DEV_PORT }); +}); + +test("resolvePorts: env overrides win over defaults", () => { + const ports = resolvePorts("2", { API_DEV_PORT: "6250" }); + assert.equal(ports.API_DEV_PORT, 6250); + assert.equal(ports.AUTHORING_DEV_PORT, PORT_DEFAULTS.AUTHORING_DEV_PORT); +}); + +test("resolvePorts: tier=full resolves a distinct inspector port for both api and o11y", () => { + const ports = resolvePorts("full", {}); + assert.notEqual(ports.API_DEV_INSPECTOR_PORT, ports.O11Y_DEV_INSPECTOR_PORT); + assert.notEqual(ports.API_DEV_INSPECTOR_PORT, ports.API_DEV_PORT); + assert.notEqual(ports.O11Y_DEV_INSPECTOR_PORT, ports.O11Y_DEV_PORT); +}); + +test("resolvePorts: throws on a port collision (two keys resolving to the same number)", () => { + assert.throws( + () => resolvePorts("full", { API_DEV_PORT: "6200", O11Y_DEV_PORT: "6200" }), + /collision/, + ); +}); + +test("resolvePorts: throws on a non-numeric port override", () => { + assert.throws(() => resolvePorts("1", { AUTHORING_DEV_PORT: "not-a-port" }), /invalid port/); +}); + +// --------------------------------------------------------------------------- +// .dev.vars bootstrap +// --------------------------------------------------------------------------- + +test("bootstrapDevVars: copies the example when .dev.vars is absent", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "DEV_AUTH_EMAIL=\"dev@handsontable.com\"\nPREVIEW_HOST=\"localhost:8787\"\n"); + const result = bootstrapDevVars({ examplePath, devVarsPath }); + assert.equal(result.created, true); + assert.equal(readFileSync(devVarsPath, "utf8"), readFileSync(examplePath, "utf8")); + }); +}); + +test("bootstrapDevVars: never overwrites an existing .dev.vars", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "DEV_AUTH_EMAIL=\"dev@handsontable.com\"\n"); + writeFileSync(devVarsPath, "DEV_AUTH_EMAIL=\"someone-else@handsontable.com\"\n# my own edits\n"); + const result = bootstrapDevVars({ examplePath, devVarsPath }); + assert.equal(result.created, false); + assert.equal(readFileSync(devVarsPath, "utf8"), "DEV_AUTH_EMAIL=\"someone-else@handsontable.com\"\n# my own edits\n"); + }); +}); + +test("bootstrapDevVars: throws a clear error when the example itself is missing", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + assert.throws(() => bootstrapDevVars({ examplePath, devVarsPath }), /missing/); + }); +}); + +test("bootstrapDevVars: patch fills in an empty placeholder line, only on fresh creation", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "DEV_ADMIN=\nO11Y_EXPORT_SECRET=\nO11Y_ENV=local\n"); + const result = bootstrapDevVars({ + examplePath, + devVarsPath, + patch: { DEV_ADMIN: "dev@handsontable.com" }, + }); + assert.equal(result.created, true); + assert.deepEqual(result.patched, ["DEV_ADMIN"]); + const text = readFileSync(devVarsPath, "utf8"); + assert.match(text, /^DEV_ADMIN=dev@handsontable\.com$/m); + // A secret this task must never auto-create stays untouched (empty). + assert.match(text, /^O11Y_EXPORT_SECRET=\s*$/m); + }); +}); + +test("bootstrapDevVars: patch never touches a NON-empty line (a real value already there)", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "DEV_ADMIN=someone@handsontable.com\n"); + const result = bootstrapDevVars({ examplePath, devVarsPath, patch: { DEV_ADMIN: "dev@handsontable.com" } }); + assert.deepEqual(result.patched, []); + assert.match(readFileSync(devVarsPath, "utf8"), /^DEV_ADMIN=someone@handsontable\.com$/m); + }); +}); + +test("bootstrapDevVars: stripKeys removes an empty declared line so a later --var isn't silently shadowed", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "O11Y_SESSION_SECRET=\nO11Y_ENV=local\n"); + const result = bootstrapDevVars({ examplePath, devVarsPath, stripKeys: ["O11Y_SESSION_SECRET"] }); + assert.deepEqual(result.stripped, ["O11Y_SESSION_SECRET"]); + const text = readFileSync(devVarsPath, "utf8"); + assert.doesNotMatch(text, /O11Y_SESSION_SECRET/); + assert.match(text, /O11Y_ENV=local/); + }); +}); + +test("bootstrapDevVars: stripKeys is a no-op when the key isn't declared in the example at all", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(examplePath, "O11Y_ENV=local\n"); + const result = bootstrapDevVars({ examplePath, devVarsPath, stripKeys: ["O11Y_SESSION_SECRET"] }); + assert.deepEqual(result.stripped, []); + assert.equal(readFileSync(devVarsPath, "utf8"), "O11Y_ENV=local\n"); + }); +}); + +test("o11yDevVarsPatch: never includes O11Y_EXPORT_SECRET, SENTRY_HOOK_SECRET, or O11Y_SESSION_SECRET (never auto-create a real secret)", () => { + const patch = o11yDevVarsPatch({ O11Y_SLACK_CAPTURE_PORT: 4210 }); + assert.equal("O11Y_EXPORT_SECRET" in patch, false); + assert.equal("SENTRY_HOOK_SECRET" in patch, false); + assert.equal("O11Y_SESSION_SECRET" in patch, false); +}); + +// ---- O11Y_EXPORT_SECRET/SENTRY_HOOK_SECRET must stay empty locally, or +// scripts/o11y-replay-fixtures.mjs 401s on every OTLP/deploy/Sentry +// fixture and the worker-tenant/Sentry panels never fill. ------------------- + +test("fillEmptyDevVarsSecrets: fills an empty declared line with a generated value", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, "DEV_ADMIN=dev@handsontable.com\nO11Y_EXPORT_SECRET=\nSENTRY_HOOK_SECRET=\nO11Y_ENV=local\n"); + const result = fillEmptyDevVarsSecrets({ + devVarsPath, + keys: O11Y_DEVVARS_AUTOFILL_SECRET_KEYS, + generate: () => "ephemeral-value", + }); + assert.deepEqual(result.filled, ["O11Y_EXPORT_SECRET", "SENTRY_HOOK_SECRET"]); + const text = readFileSync(devVarsPath, "utf8"); + assert.match(text, /^O11Y_EXPORT_SECRET=ephemeral-value$/m); + assert.match(text, /^SENTRY_HOOK_SECRET=ephemeral-value$/m); + // Untouched lines stay untouched. + assert.match(text, /^DEV_ADMIN=dev@handsontable\.com$/m); + assert.match(text, /^O11Y_ENV=local$/m); + }); +}); + +test("fillEmptyDevVarsSecrets: runs on a PRE-EXISTING file, unlike bootstrapDevVars's own patch", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + // Simulate a `.dev.vars` bootstrapped by an OLDER dev.mjs, before this + // fix shipped: real values everywhere except the two secrets. + writeFileSync( + devVarsPath, + "DEV_ADMIN=dev@handsontable.com\nAE_SQL_TOKEN=local-dev-token\nO11Y_EXPORT_SECRET=\nSENTRY_HOOK_SECRET=\nO11Y_ENV=local\n", + ); + // bootstrapDevVars alone is a no-op here (the file already exists) — + // pinning that this really is the gap `fillEmptyDevVarsSecrets` closes. + const bootstrapResult = bootstrapDevVars({ + examplePath: path.join(dir, ".dev.vars.example"), // never read; existsSync(devVarsPath) short-circuits + devVarsPath, + patch: o11yDevVarsPatch({ O11Y_SLACK_CAPTURE_PORT: 4210 }), + stripKeys: O11Y_DEVVARS_STRIP_KEYS, + }); + assert.equal(bootstrapResult.created, false); + assert.match(readFileSync(devVarsPath, "utf8"), /^O11Y_EXPORT_SECRET=\s*$/m, "still empty after bootstrapDevVars alone"); + + const result = fillEmptyDevVarsSecrets({ devVarsPath, keys: O11Y_DEVVARS_AUTOFILL_SECRET_KEYS }); + assert.deepEqual(result.filled, ["O11Y_EXPORT_SECRET", "SENTRY_HOOK_SECRET"]); + const text = readFileSync(devVarsPath, "utf8"); + assert.doesNotMatch(text, /^O11Y_EXPORT_SECRET=\s*$/m); + assert.doesNotMatch(text, /^SENTRY_HOOK_SECRET=\s*$/m); + }); +}); + +test("fillEmptyDevVarsSecrets: never touches a key that already holds a real (non-empty) value", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, "O11Y_EXPORT_SECRET=a-real-pasted-value\nSENTRY_HOOK_SECRET=\n"); + const result = fillEmptyDevVarsSecrets({ + devVarsPath, + keys: O11Y_DEVVARS_AUTOFILL_SECRET_KEYS, + generate: () => "generated", + }); + assert.deepEqual(result.filled, ["SENTRY_HOOK_SECRET"]); + const text = readFileSync(devVarsPath, "utf8"); + assert.match(text, /^O11Y_EXPORT_SECRET=a-real-pasted-value$/m); + assert.match(text, /^SENTRY_HOOK_SECRET=generated$/m); + }); +}); + +test("fillEmptyDevVarsSecrets: no-op (no throw, filled: []) when the file does not exist", () => { + withTmpDir((dir) => { + const result = fillEmptyDevVarsSecrets({ + devVarsPath: path.join(dir, "does-not-exist", ".dev.vars"), + keys: O11Y_DEVVARS_AUTOFILL_SECRET_KEYS, + }); + assert.deepEqual(result.filled, []); + }); +}); + +test("fillEmptyDevVarsSecrets: generates a real ephemeral value by default (not a fixed placeholder) — same shape as ephemeralSecret()", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, "O11Y_EXPORT_SECRET=\n"); + fillEmptyDevVarsSecrets({ devVarsPath, keys: ["O11Y_EXPORT_SECRET"] }); + const text = readFileSync(devVarsPath, "utf8"); + const m = /^O11Y_EXPORT_SECRET=([0-9a-f]{64})$/m.exec(text); + assert.ok(m, `expected a 32-byte hex value, got: ${text}`); + }); +}); + +test("resolveReplaySecret: env value wins when both env and .dev.vars have it", () => { + const value = resolveReplaySecret("from-env", "/irrelevant", "O11Y_EXPORT_SECRET", () => "from-devvars"); + assert.equal(value, "from-env"); +}); + +test("resolveReplaySecret: falls back to .dev.vars when env is unset (the standalone-invocation case)", () => { + const value = resolveReplaySecret(undefined, "/irrelevant", "O11Y_EXPORT_SECRET", () => "from-devvars"); + assert.equal(value, "from-devvars"); +}); + +test("resolveReplaySecret: falls back to .dev.vars when env is an empty string too", () => { + const value = resolveReplaySecret("", "/irrelevant", "O11Y_EXPORT_SECRET", () => "from-devvars"); + assert.equal(value, "from-devvars"); +}); + +test("resolveReplaySecret: empty string when neither source has a value (never undefined/null — callers compare it with a header string)", () => { + assert.equal(resolveReplaySecret(undefined, "/irrelevant", "O11Y_EXPORT_SECRET", () => undefined), ""); +}); + +test("resolveReplaySecret: reads the REAL workers/o11y/.dev.vars shape via readDevVarsLine (no injected readLine) — proves the wiring, not just the arithmetic", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, "O11Y_EXPORT_SECRET=abc123\nSENTRY_HOOK_SECRET=\n"); + assert.equal(resolveReplaySecret(undefined, devVarsPath, "O11Y_EXPORT_SECRET"), "abc123"); + assert.equal(resolveReplaySecret(undefined, devVarsPath, "SENTRY_HOOK_SECRET"), ""); + }); +}); + +test("readDevVarsLine / checkDevVarsPortDrift: detects a port mismatch and reports none when it matches", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, 'PREVIEW_HOST="localhost:8787"\n'); + assert.equal(checkDevVarsPortDrift(devVarsPath, "PREVIEW_HOST", 8787), null); + const drift = checkDevVarsPortDrift(devVarsPath, "PREVIEW_HOST", 6250); + assert.match(drift, /8787/); + assert.match(drift, /6250/); + }); +}); + +test("resolveDevVarsPortAdoption: PREVIEW_HOST port is ADOPTED when API_DEV_PORT was not explicitly set (bug 2's fix)", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + // The user's exact real .dev.vars: PREVIEW_HOST pinned to 8799 while + // this run's own default/resolved port is 8787. + writeFileSync(devVarsPath, 'PREVIEW_HOST="localhost:8799"\n'); + + const adoption = resolveDevVarsPortAdoption({ devVarsPath, key: "PREVIEW_HOST", currentPort: 8787, explicit: false }); + assert.equal(adoption.port, 8799, "the .dev.vars port is adopted, since .dev.vars always wins anyway"); + assert.equal(adoption.adopted, true); + assert.match(adoption.message, /8799/); + assert.match(adoption.message, /adopting/); + }); +}); + +test("resolveDevVarsPortAdoption: WARNS instead (does not override) when API_DEV_PORT was explicitly set and conflicts", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, 'PREVIEW_HOST="localhost:8799"\n'); + + const adoption = resolveDevVarsPortAdoption({ devVarsPath, key: "PREVIEW_HOST", currentPort: 6450, explicit: true }); + assert.equal(adoption.port, 6450, "an explicit override is never silently discarded"); + assert.equal(adoption.adopted, false); + assert.match(adoption.message, /8799/); + assert.match(adoption.message, /6450/); + }); +}); + +test("resolveDevVarsPortAdoption: no message and no change when the port already matches, or the key is undeclared", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + writeFileSync(devVarsPath, 'PREVIEW_HOST="localhost:8787"\n'); + assert.deepEqual(resolveDevVarsPortAdoption({ devVarsPath, key: "PREVIEW_HOST", currentPort: 8787, explicit: false }), { + port: 8787, + adopted: false, + message: null, + }); + assert.deepEqual(resolveDevVarsPortAdoption({ devVarsPath, key: "SLACK_WEBHOOK_URL", currentPort: 4210, explicit: false }), { + port: 4210, + adopted: false, + message: null, + }); + }); +}); + +test("assertNoPortCollisions: throws for a duplicate port value, passes for all-distinct ports", () => { + assert.doesNotThrow(() => assertNoPortCollisions({ A: 1, B: 2 })); + assert.throws(() => assertNoPortCollisions({ A: 1, B: 1 }), /port collision/); +}); + +test("checkO11yDevVarsStaleness: warns when a pre-existing .dev.vars declares DEV_ADMIN or O11Y_SESSION_SECRET empty", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + + // A healthy, freshly-bootstrapped file: no warnings. + writeFileSync(devVarsPath, "O11Y_ENV=local\nDEV_ADMIN=dev@handsontable.com\n"); + assert.deepEqual(checkO11yDevVarsStaleness(devVarsPath), []); + + // An old .dev.vars from before this file's own key set was added + // DEV_ADMIN/O11Y_SESSION_SECRET handling existed, both declared empty. + writeFileSync(devVarsPath, "O11Y_ENV=local\nDEV_ADMIN=\nO11Y_SESSION_SECRET=\n"); + const warnings = checkO11yDevVarsStaleness(devVarsPath); + assert.equal(warnings.length, 2, "both the bypass and the secret must each get their own warning"); + assert.ok(warnings.some((w) => w.includes("DEV_ADMIN"))); + assert.ok(warnings.some((w) => w.includes("O11Y_SESSION_SECRET"))); + + // O11Y_SESSION_SECRET simply ABSENT (the normal, freshly-stripped case) + // must never warn — only DECLARED-but-empty is the problem. + writeFileSync(devVarsPath, "O11Y_ENV=local\nDEV_ADMIN=dev@handsontable.com\n"); + assert.deepEqual(checkO11yDevVarsStaleness(devVarsPath), []); + }); +}); + +test("checkO11yDevVarsStaleness: warns when a pre-existing .dev.vars declares SLACK_WEBHOOK_URL or AE_SQL_TOKEN empty", () => { + withTmpDir((dir) => { + const devVarsPath = path.join(dir, ".dev.vars"); + + // A healthy, freshly-bootstrapped file: no warnings. + writeFileSync( + devVarsPath, + "O11Y_ENV=local\nDEV_ADMIN=dev@handsontable.com\nSLACK_WEBHOOK_URL=http://localhost:4210/slack\nAE_SQL_TOKEN=local-dev-token\n", + ); + assert.deepEqual(checkO11yDevVarsStaleness(devVarsPath), []); + + // The stale-bootstrap shape: an old .dev.vars from before + // o11yDevVarsPatch grew these two keys, both declared empty (exactly + // what workers/o11y/.dev.vars.example still declares them as before a + // fresh bootstrap patches them). + writeFileSync( + devVarsPath, + "O11Y_ENV=local\nDEV_ADMIN=dev@handsontable.com\nSLACK_WEBHOOK_URL=\nAE_SQL_TOKEN=\n", + ); + const warnings = checkO11yDevVarsStaleness(devVarsPath); + assert.equal(warnings.length, 2, "both the Slack webhook and the ClickHouse token must each get their own warning"); + assert.ok(warnings.some((w) => w.includes("SLACK_WEBHOOK_URL"))); + assert.ok(warnings.some((w) => w.includes("AE_SQL_TOKEN"))); + + // Absent entirely must never warn — only DECLARED-but-empty is the + // problem (same rule as DEV_ADMIN/O11Y_SESSION_SECRET above). + writeFileSync(devVarsPath, "O11Y_ENV=local\nDEV_ADMIN=dev@handsontable.com\n"); + assert.deepEqual(checkO11yDevVarsStaleness(devVarsPath), []); + }); +}); + +test("o11yDevVarsPatch + bootstrapDevVars: a FRESH bootstrap never leaves SLACK_WEBHOOK_URL or AE_SQL_TOKEN declared empty", () => { + withTmpDir((dir) => { + const examplePath = path.join(dir, ".dev.vars.example"); + const devVarsPath = path.join(dir, ".dev.vars"); + // The real workers/o11y/.dev.vars.example shape for these two keys. + writeFileSync(examplePath, "DEV_ADMIN=\nAE_SQL_TOKEN=\nSLACK_WEBHOOK_URL=\nO11Y_SESSION_SECRET=\nO11Y_ENV=local\n"); + + const result = bootstrapDevVars({ + examplePath, + devVarsPath, + patch: o11yDevVarsPatch({ O11Y_SLACK_CAPTURE_PORT: 4210 }), + stripKeys: O11Y_DEVVARS_STRIP_KEYS, + }); + assert.ok(result.patched.includes("AE_SQL_TOKEN")); + assert.ok(result.patched.includes("SLACK_WEBHOOK_URL")); + + // The staleness check must find nothing to warn about on this fresh file. + assert.deepEqual(checkO11yDevVarsStaleness(devVarsPath), []); + }); +}); + +test("ephemeralSecret: never the same value twice, and never written by bootstrapDevVars", () => { + assert.notEqual(ephemeralSecret(), ephemeralSecret()); + assert.equal(ephemeralSecret().length, 64); // 32 bytes, hex +}); + +test("redactArgsForLog: O11Y_SESSION_SECRET's --var value is redacted for dev.mjs's own log line", () => { + const secret = ephemeralSecret(); + const args = [ + "dev", + "--port", + "4200", + "--var", + `O11Y_SESSION_SECRET:${secret}`, + "--var", + "O11Y_LOCAL_MINIO_PORT:9000", + ]; + const redacted = redactArgsForLog(args); + assert.ok(!redacted.join(" ").includes(secret), "the secret value must never appear in the redacted args"); + assert.deepEqual(redacted, ["dev", "--port", "4200", "--var", "O11Y_SESSION_SECRET:<redacted>", "--var", "O11Y_LOCAL_MINIO_PORT:9000"]); + // The real spawn() args are untouched — only a copy for display is redacted. + assert.ok(args.join(" ").includes(secret), "the original args array passed to spawn() must be unaffected"); +}); + +// --------------------------------------------------------------------------- +// migration tracking +// --------------------------------------------------------------------------- + +function makeMigrationsDir(dir, files) { + const migrationsDir = path.join(dir, "migrations"); + mkdirSync(migrationsDir); + for (const f of files) writeFileSync(path.join(migrationsDir, f), "-- sql\n"); + return migrationsDir; +} + +test("planMigrations: with no record, every file on disk is pending", () => { + withTmpDir((dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql", "0002_more.sql"]); + const recordPath = path.join(dir, "record.json"); + const plan = planMigrations({ migrationsDir, recordPath }); + assert.deepEqual(plan.pending, ["0001_init.sql", "0002_more.sql"]); + }); +}); + +test("applyMigrations: a second run applies nothing once the first run recorded every file", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql", "0002_more.sql", "0003_third.sql"]); + const recordPath = migrationRecordPath(dir); + const calls = []; + const run = async (args) => { + calls.push(args); + }; + + const first = await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run }); + assert.deepEqual(first.applied, ["0001_init.sql", "0002_more.sql", "0003_third.sql"]); + assert.equal(calls.length, 3); + assert.deepEqual(calls[0], ["d1", "execute", "handsontable-demos", "--local", "--file=migrations/0001_init.sql", "-y"]); + + const second = await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run }); + assert.deepEqual(second.applied, []); + assert.equal(calls.length, 3, "no new wrangler calls on the second run"); + }); +}); + +test("applyMigrations: a new migration file added after the first run is the only one applied on the second run", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql"]); + const recordPath = migrationRecordPath(dir); + const calls = []; + const run = async (args) => { + calls.push(args); + }; + await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run }); + + writeFileSync(path.join(migrationsDir, "0002_new.sql"), "-- sql\n"); + const second = await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run }); + assert.deepEqual(second.applied, ["0002_new.sql"]); + }); +}); + +test("applyMigrations: records each file as it succeeds, so a failure partway through doesn't lose earlier progress", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql", "0002_boom.sql"]); + const recordPath = migrationRecordPath(dir); + const run = async (args) => { + if (args.includes("--file=migrations/0002_boom.sql")) throw new Error("simulated d1 failure"); + }; + await assert.rejects(() => applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run })); + const plan = planMigrations({ migrationsDir, recordPath }); + assert.deepEqual(plan.applied, ["0001_init.sql"]); + assert.deepEqual(plan.pending, ["0002_boom.sql"]); + }); +}); + +// --------------------------------------------------------------------------- +// migration schema probe — adopting a pre-existing local D1 with no record: +// a hand-migrated local D1, no dev-migrations-applied.json, re-applying +// from 0001 must not die on 0003's non-idempotent `ALTER TABLE ... ADD +// COLUMN` with a raw `duplicate column name` failure. +// --------------------------------------------------------------------------- + +/** A stub `query` (the injectable `applyMigrations`/`snapshotLocalSchema` + * takes in place of a real `wrangler d1 execute ... --json`) backed by an + * in-memory {tables: Set<string>, indexes: Set<string>, columns: {[table]: + * Set<string>}} — enough to answer both queries `snapshotLocalSchema` + * issues, shaped like wrangler's own `--json` output. */ +function stubD1Query(state) { + return async (args) => { + const command = args[args.indexOf("--command") + 1]; + if (command.includes("sqlite_master")) { + const results = [ + ...[...state.tables].map((name) => ({ type: "table", name })), + ...[...state.indexes].map((name) => ({ type: "index", name })), + ]; + return JSON.stringify([{ results, success: true, meta: { duration: 0 } }]); + } + const m = /PRAGMA table_info\((\w+)\)/.exec(command); + if (m) { + const cols = state.columns[m[1]] ?? new Set(); + return JSON.stringify([{ results: [...cols].map((name) => ({ name })), success: true, meta: { duration: 0 } }]); + } + throw new Error(`stubD1Query: unrecognized --command: ${command}`); + }; +} + +test("parseMigrationTargets: CREATE TABLE / CREATE INDEX / ALTER TABLE ADD COLUMN are checkable; any other statement makes the file non-checkable", () => { + assert.deepEqual(parseMigrationTargets("CREATE TABLE IF NOT EXISTS demos (id TEXT);\nCREATE INDEX IF NOT EXISTS idx_x ON demos(id);"), { + checkable: true, + targets: [ + { type: "table", name: "demos" }, + { type: "index", name: "idx_x" }, + ], + }); + assert.deepEqual(parseMigrationTargets("ALTER TABLE demos ADD COLUMN foo TEXT;"), { + checkable: true, + targets: [{ type: "column", table: "demos", name: "foo" }], + }); + assert.deepEqual(parseMigrationTargets("ALTER TABLE demos ADD foo TEXT;"), { + checkable: true, + targets: [{ type: "column", table: "demos", name: "foo" }], + }); + // DROP INDEX (0002_buildkey_nonunique.sql's real shape) is not a + // recognized "additive, checkable" statement — the whole file must fall + // through to "always run it" rather than risk skipping the DROP because + // the CREATE INDEX that follows it happens to already exist. + assert.deepEqual(parseMigrationTargets("DROP INDEX IF EXISTS idx_x;\nCREATE INDEX IF NOT EXISTS idx_x ON demos(id);"), { + checkable: false, + targets: [], + }); + // Empty file: never a vacuous "already applied". + assert.deepEqual(parseMigrationTargets("-- just a comment\n"), { checkable: false, targets: [] }); +}); + +test("parseMigrationTargets: pinned against every real workers/api/migrations/*.sql file", () => { + const migrationsDir = path.join(RUNNER_ROOT, "workers", "api", "migrations"); + const files = readFileSync(path.join(migrationsDir, "0001_init.sql"), "utf8"); // sanity: file exists + assert.ok(files.length > 0); + + const read = (name) => readFileSync(path.join(migrationsDir, name), "utf8"); + assert.deepEqual(parseMigrationTargets(read("0001_init.sql")), { + checkable: true, + targets: [ + { type: "table", name: "demos" }, + { type: "index", name: "idx_demos_framework" }, + { type: "index", name: "idx_demos_created_by" }, + { type: "index", name: "idx_demos_forked_from" }, + { type: "index", name: "idx_demos_buildkey" }, + { type: "table", name: "build_cache" }, + ], + }); + assert.equal(parseMigrationTargets(read("0002_buildkey_nonunique.sql")).checkable, false, "0002's DROP INDEX must not be treated as checkable"); + assert.deepEqual(parseMigrationTargets(read("0003_cost_ledger.sql")), { + checkable: true, + targets: [ + { type: "table", name: "cost_ledger" }, + { type: "index", name: "idx_cost_ledger_day" }, + { type: "table", name: "usage_daily" }, + { type: "index", name: "idx_usage_daily_day" }, + { type: "column", table: "demos", name: "artifacts_purged_at" }, + ], + }); + assert.deepEqual(parseMigrationTargets(read("0007_build_status.sql")), { + checkable: true, + targets: [ + { type: "column", table: "demos", name: "build_status" }, + { type: "column", table: "demos", name: "build_error" }, + ], + }); + // 0009_example_daily_downloaded.sql: the second real + // ALTER TABLE ... ADD COLUMN file in this migrations dir (after 0003/0007, + // both against `demos`) — pinned explicitly, not just swept into the + // "checkable with >=1 target" loop below, because it is the one that + // exercises a table OTHER than `demos` going through the same adoption + // path (see the `applyMigrations` adoption test further down). + assert.deepEqual(parseMigrationTargets(read("0009_example_daily_downloaded.sql")), { + checkable: true, + targets: [{ type: "column", table: "example_daily", name: "downloaded" }], + }); + // Every file must at least parse without throwing and either be checkable + // with >=1 target, or explicitly non-checkable — never checkable with zero + // targets (that would be silently skippable). + for (const name of ["0004_settings_and_analytics.sql", "0005_profiles.sql", "0006_api_tokens.sql", "0008_example_daily.sql"]) { + const parsed = parseMigrationTargets(read(name)); + assert.ok(parsed.checkable, `${name} expected checkable`); + assert.ok(parsed.targets.length > 0, `${name} expected at least one target`); + } +}); + +test("isMigrationAlreadyApplied: true only when every target is present; false for an empty/unchecked target list", () => { + const snapshot = { tableNames: new Set(["demos"]), indexNames: new Set(["idx_x"]), columns: { demos: new Set(["id", "artifacts_purged_at"]) } }; + assert.equal(isMigrationAlreadyApplied([{ type: "table", name: "demos" }], snapshot), true); + assert.equal(isMigrationAlreadyApplied([{ type: "column", table: "demos", name: "artifacts_purged_at" }], snapshot), true); + assert.equal(isMigrationAlreadyApplied([{ type: "column", table: "demos", name: "build_status" }], snapshot), false); + assert.equal(isMigrationAlreadyApplied([{ type: "table", name: "demos" }, { type: "table", name: "nope" }], snapshot), false); + assert.equal(isMigrationAlreadyApplied([], snapshot), false, "an empty target list must never read as already applied"); +}); + +test("snapshotLocalSchema: one sqlite_master query plus one PRAGMA per requested table, parsed from wrangler --json shape", async () => { + const calls = []; + const query = async (args) => { + calls.push(args); + return stubD1Query({ tables: new Set(["demos"]), indexes: new Set(["idx_x"]), columns: { demos: new Set(["id", "artifacts_purged_at"]) } })(args); + }; + const snapshot = await snapshotLocalSchema({ dbName: "handsontable-demos", tables: ["demos"], query }); + assert.deepEqual([...snapshot.tableNames], ["demos"]); + assert.deepEqual([...snapshot.indexNames], ["idx_x"]); + assert.deepEqual([...snapshot.columns.demos].sort(), ["artifacts_purged_at", "id"]); + assert.equal(calls.length, 2, "one sqlite_master query + one PRAGMA for the one requested table"); +}); + +test("applyMigrations: a hand-migrated local D1 with NO record — every pending file whose targets already exist is adopted, not re-applied", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = path.join(dir, "migrations"); + mkdirSync(migrationsDir); + writeFileSync(path.join(migrationsDir, "0001_init.sql"), "CREATE TABLE IF NOT EXISTS demos (id TEXT PRIMARY KEY);\n"); + writeFileSync( + path.join(migrationsDir, "0003_cost_ledger.sql"), + "CREATE TABLE IF NOT EXISTS cost_ledger (day TEXT);\nALTER TABLE demos ADD COLUMN artifacts_purged_at TEXT;\n", + ); + writeFileSync(path.join(migrationsDir, "0007_build_status.sql"), "ALTER TABLE demos ADD COLUMN build_status TEXT;\n"); + const recordPath = migrationRecordPath(dir); // no record file at all — the exact bug precondition + + const state = { + tables: new Set(["demos", "cost_ledger"]), // 0001, 0003's table: already there + indexes: new Set(), + columns: { demos: new Set(["id", "artifacts_purged_at"]) }, // 0003's column exists; 0007's build_status does NOT + }; + const runCalls = []; + const run = async (args) => { + runCalls.push(args); + // Applying 0007 for real adds the column this run's own snapshot didn't have yet. + if (args.includes("--file=migrations/0007_build_status.sql")) state.columns.demos.add("build_status"); + }; + const query = stubD1Query(state); + + const result = await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run, query, log: () => {} }); + + assert.deepEqual(result.adopted, ["0001_init.sql", "0003_cost_ledger.sql"], "both fully-pre-existing files are adopted, not re-run"); + assert.deepEqual(result.applied, ["0007_build_status.sql"], "the file whose column is genuinely missing still runs for real"); + assert.deepEqual(runCalls, [["d1", "execute", "handsontable-demos", "--local", "--file=migrations/0007_build_status.sql", "-y"]]); + assert.deepEqual(readAppliedMigrations(recordPath), ["0001_init.sql", "0003_cost_ledger.sql", "0007_build_status.sql"].sort()); + }); +}); + +// The same adoption path, exercised against the real +// 0008/0009_example_daily_downloaded.sql pair — a table (`example_daily`) +// that is not `demos`, proving `applyMigrations`' `alterTables` derivation +// is not hardcoded to the one table every earlier migration in this dir +// happens to alter. +test("applyMigrations: a local D1 that already has example_daily.downloaded (dev stack migrated by hand) adopts 0009 instead of re-running it", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = path.join(dir, "migrations"); + mkdirSync(migrationsDir); + const realMigrationsDir = path.join(RUNNER_ROOT, "workers", "api", "migrations"); + writeFileSync( + path.join(migrationsDir, "0008_example_daily.sql"), + readFileSync(path.join(realMigrationsDir, "0008_example_daily.sql"), "utf8"), + ); + writeFileSync( + path.join(migrationsDir, "0009_example_daily_downloaded.sql"), + readFileSync(path.join(realMigrationsDir, "0009_example_daily_downloaded.sql"), "utf8"), + ); + // 0008 was recorded as applied by an earlier run; 0009 is pending, and a + // developer's local D1 already carries the `downloaded` column (e.g. + // adopted by hand, or applied once before the applied-migrations record + // existed — the same class of drift the adjacent `demos` test above + // covers). + mkdirSync(path.dirname(migrationRecordPath(dir)), { recursive: true }); + writeFileSync(migrationRecordPath(dir), JSON.stringify(["0008_example_daily.sql"])); + + const state = { + tables: new Set(["example_daily"]), + indexes: new Set(["idx_example_daily_day"]), + columns: { example_daily: new Set(["day", "kind", "ref", "area", "framework", "ht_major", "opens", "engaged", "forked", "saved", "shared", "downloaded"]) }, + }; + const runCalls = []; + const run = async (args) => runCalls.push(args); + const query = stubD1Query(state); + + const result = await applyMigrations({ migrationsDir, recordPath: migrationRecordPath(dir), dbName: "handsontable-demos", run, query, log: () => {} }); + + assert.deepEqual(result.adopted, ["0009_example_daily_downloaded.sql"], "the column already exists — 0009 must be adopted, not re-run"); + assert.deepEqual(result.applied, [], "never a real d1 execute for a file whose only target is already present"); + assert.deepEqual(runCalls, [], "no wrangler d1 execute call at all — this is what avoids the 'duplicate column name' failure"); + assert.deepEqual(readAppliedMigrations(migrationRecordPath(dir)), ["0008_example_daily.sql", "0009_example_daily_downloaded.sql"].sort()); + }); +}); + +test("applyMigrations: without `query`, behavior is unchanged — every pending file is always re-run (no probing)", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql"]); + const recordPath = migrationRecordPath(dir); + const calls = []; + const run = async (args) => calls.push(args); + const result = await applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run }); + assert.deepEqual(result.applied, ["0001_init.sql"]); + assert.deepEqual(result.adopted, []); + assert.equal(calls.length, 1); + }); +}); + +test("applyMigrations: a genuinely failing migration throws a MigrationError with the file and the SQLite message, never swallowed as adopted", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_init.sql", "0002_boom.sql"]); + const recordPath = migrationRecordPath(dir); + // Shaped exactly like a real execFileSync failure: wrangler's ANSI-wrapped + // "duplicate column name" text on stderr (captured against wrangler 4.108). + const stderr = Buffer.from( + "\u001b[31m✘ \u001b[41;31m[\u001b[41;97mERROR\u001b[41;31m]\u001b[0m \u001b[1mduplicate column name: artifacts_purged_at: SQLITE_ERROR\u001b[0m\n", + ); + const run = async (args) => { + if (args.includes("--file=migrations/0002_boom.sql")) { + const err = new Error("Command failed"); + err.stderr = stderr; + err.stdout = Buffer.from(""); + err.status = 1; + throw err; + } + }; + await assert.rejects( + () => applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run, log: () => {} }), + (err) => { + assert.ok(err instanceof MigrationError); + assert.equal(err.file, "0002_boom.sql"); + assert.equal(err.sqliteMessage, "duplicate column name: artifacts_purged_at: SQLITE_ERROR"); + assert.equal(err.recordPath, recordPath); + return true; + }, + ); + // Not swallowed: 0002 must NOT be recorded as applied/adopted. + assert.deepEqual(readAppliedMigrations(recordPath), ["0001_init.sql"]); + }); +}); + +test("applyMigrations: a genuinely DIFFERENT SQL error is reported as itself, not misread as a duplicate-column adoption case", async () => { + await withTmpDir(async (dir) => { + const migrationsDir = makeMigrationsDir(dir, ["0001_typo.sql"]); + const recordPath = migrationRecordPath(dir); + const run = async () => { + const err = new Error("Command failed"); + err.stderr = Buffer.from("\u001b[31m✘ \u001b[41;31m[\u001b[41;97mERROR\u001b[41;31m]\u001b[0m \u001b[1mno such table: nope: SQLITE_ERROR\u001b[0m\n"); + err.stdout = Buffer.from(""); + throw err; + }; + await assert.rejects( + () => applyMigrations({ migrationsDir, recordPath, dbName: "handsontable-demos", run, log: () => {} }), + (err) => { + assert.equal(err.sqliteMessage, "no such table: nope: SQLITE_ERROR"); + return true; + }, + ); + }); +}); + +test("formatMigrationError: one clean line naming the file, the SQLite message, the record path, and --reset-local-db — never a raw stack trace", () => { + const err = new MigrationError({ + file: "0003_cost_ledger.sql", + sqliteMessage: "duplicate column name: artifacts_purged_at: SQLITE_ERROR", + recordPath: "/x/workers/api/.wrangler/state/dev-migrations-applied.json", + action: "applying", + }); + const formatted = formatMigrationError(err); + assert.match(formatted, /^error:/); + assert.match(formatted, /0003_cost_ledger\.sql/); + assert.match(formatted, /duplicate column name: artifacts_purged_at: SQLITE_ERROR/); + assert.match(formatted, /dev-migrations-applied\.json/); + assert.match(formatted, /--reset-local-db/); + assert.ok(!formatted.includes("\n at "), "must not include a stack-trace-shaped line"); +}); + +test("resetLocalD1: deletes local D1 state and the applied-migrations record, and logs what it deleted", () => { + withTmpDir((dir) => { + const apiDir = path.join(dir, "workers", "api"); + const stateDir = path.join(apiDir, ".wrangler", "state", "v3", "d1"); + const recordPath = migrationRecordPath(apiDir); + mkdirSync(stateDir, { recursive: true }); + writeFileSync(path.join(stateDir, "some.sqlite"), "fake"); + mkdirSync(path.dirname(recordPath), { recursive: true }); + writeFileSync(recordPath, "[]\n"); + + const lines = []; + resetLocalD1(apiDir, undefined, (l) => lines.push(l)); + + assert.equal(existsSync(stateDir), false); + assert.equal(existsSync(recordPath), false); + assert.equal(lines.length, 1); + assert.match(lines[0], /--reset-local-db/); + assert.match(lines[0], /deleted/); + + // Second call, nothing left: says so, doesn't throw. + const lines2 = []; + resetLocalD1(apiDir, undefined, (l) => lines2.push(l)); + assert.match(lines2[0], /nothing to delete/); + }); +}); + +// --------------------------------------------------------------------------- +// Docker +// --------------------------------------------------------------------------- + +test("isDockerAvailable: true when `docker info` succeeds", () => { + assert.equal(isDockerAvailable(() => {}), true); +}); + +test("isDockerAvailable: false when `docker info` throws", () => { + assert.equal( + isDockerAvailable(() => { + throw new Error("Cannot connect to the Docker daemon"); + }), + false, + ); +}); + +test("DOCKER_NOT_RUNNING_MESSAGE: names the fix (start Docker), not just the symptom", () => { + assert.match(DOCKER_NOT_RUNNING_MESSAGE, /docker info/); + assert.match(DOCKER_NOT_RUNNING_MESSAGE, /Start Docker/); +}); + +// `--wait` makes `docker compose ... up -d --wait minio clickhouse` +// block-and-fail on a real condition (a named service's healthcheck never +// goes green). That throw must not happen before this call's own teardown +// step is pushed onto `teardownSteps`, or it propagates straight past the +// try/catch around the readiness wait further down to `main().catch`, +// which only logs and `process.exit(1)`s — no cleanup, no `docker compose +// down`, orphaning whichever of minio/clickhouse did start under `up -d`. +// `bringUpO11yCompose` (extracted to dev-lib.mjs so this is unit-testable +// with a stub, matching `resetO11yLocalState`'s own pattern) needs a +// try/catch around the up call, not a bare +// `execFileSyncImpl("docker", [...up...])`. +test("bringUpO11yCompose: tears down (no -v, data kept) when the compose up itself throws, then rethrows", () => { + const calls = []; + const execFileSyncImpl = (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + if (args[3] === "up") throw new Error("container minio did not become healthy"); + return ""; + }; + assert.throws( + () => bringUpO11yCompose({ composeFile: "/x/compose.yml", composeEnv: { COMPOSE_PROJECT_NAME: "o11y-test" }, execFileSyncImpl }), + /container minio did not become healthy/, + "must rethrow the original up failure, not swallow it", + ); + assert.equal(calls.length, 2, "must call docker exactly twice: the failed up, then a down"); + assert.deepEqual(calls[0].args, ["compose", "-f", "/x/compose.yml", "up", "-d", "--wait", "minio", "clickhouse"]); + assert.deepEqual(calls[1].args, composeDownArgs("/x/compose.yml"), "the teardown must be composeDownArgs' own no -v shape"); + assert.doesNotMatch(calls[1].args.join(" "), /-v/, "must never pass -v here — only --fresh's own explicit wipe does that"); +}); + +test("bringUpO11yCompose: a successful up calls docker exactly once, no teardown", () => { + const calls = []; + const execFileSyncImpl = (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + return ""; + }; + bringUpO11yCompose({ composeFile: "/x/compose.yml", composeEnv: {}, execFileSyncImpl }); + assert.equal(calls.length, 1, "must not attempt a teardown when up itself succeeds"); +}); + +test("bringUpO11yCompose: still rethrows the original up error even if the teardown attempt ALSO fails", () => { + const execFileSyncImpl = (cmd, args) => { + if (args[3] === "up") throw new Error("up failed"); + throw new Error("down also failed"); + }; + assert.throws(() => bringUpO11yCompose({ composeFile: "/x/compose.yml", composeEnv: {}, execFileSyncImpl }), /up failed/); +}); + +// `--reset-local-db` must run after the Docker-availability check: `dev.mjs +// --tier=2 --reset-local-db` with Docker not running must not delete +// workers/api's local D1 state before exiting on the Docker-not-running +// error. `resetLocalD1`'s call site in dev.mjs (unlike its unit-tested +// form above) is hardcoded to the real `workers/api` dir, so this test +// seeds and inspects that real (gitignored, disposable) +// `.wrangler/state/v3/d1` directory directly rather than a temp one. +test("CLI: `dev.mjs --tier=2 --reset-local-db` does NOT wipe local D1 state when Docker is not running (Docker check runs first)", () => { + const stubBinDir = path.join(HERE, "fixtures", "stub-bin"); + const devScript = path.join(RUNNER_ROOT, "scripts", "dev.mjs"); + const apiDir = path.join(RUNNER_ROOT, "workers", "api"); + const d1StateDir = path.join(apiDir, ".wrangler", "state", "v3", "d1"); + const markerPath = path.join(d1StateDir, "b9-test-marker.txt"); + // SAFETY: `d1StateDir` is REAL local wrangler/D1 state for this worktree + // (this worktree may genuinely have some already, e.g. from an earlier + // `--tier=2`/`--tier=full` run or migrations applied by another test) — + // this must never destroy it. Record whether it pre-existed; the + // `finally` below removes only what THIS test itself added (the marker + // file, or — only if the whole directory did not exist before — the + // directory this test's own `mkdirSync` created). + const preExisted = existsSync(d1StateDir); + mkdirSync(d1StateDir, { recursive: true }); + writeFileSync(markerPath, "b9 marker\n"); + try { + const result = spawnSync(process.execPath, [devScript, "--tier=2", "--reset-local-db"], { + cwd: RUNNER_ROOT, + encoding: "utf8", + timeout: 10000, + env: { + ...process.env, + PATH: `${stubBinDir}:${process.env.PATH}`, + STUB_DOCKER_MODE: "fail", + }, + }); + const output = `${result.stdout}${result.stderr}`; + assert.notEqual(result.status, 0); + assert.match(output, /Start Docker/, "must fail on the Docker check"); + assert.doesNotMatch(output, /reset-local-db: deleted/, "must never actually run the reset once Docker is confirmed unavailable"); + assert.equal(existsSync(markerPath), true, "local D1 state must survive a run that fails the Docker check"); + } finally { + // Only remove what this test added — never the real pre-existing state. + if (preExisted) { + rmSync(markerPath, { force: true }); + } else { + rmSync(d1StateDir, { recursive: true, force: true }); + } + } +}); + +test("CLI: `dev.mjs --tier=2` fails fast with the Docker message when `docker info` fails, before spawning anything else", () => { + const stubBinDir = path.join(HERE, "fixtures", "stub-bin"); + const devScript = path.join(RUNNER_ROOT, "scripts", "dev.mjs"); + const result = spawnSync(process.execPath, [devScript, "--tier=2"], { + cwd: RUNNER_ROOT, + encoding: "utf8", + env: { + ...process.env, + PATH: `${stubBinDir}:${process.env.PATH}`, + STUB_DOCKER_MODE: "fail", + }, + }); + assert.notEqual(result.status, 0); + const output = `${result.stdout}${result.stderr}`; + assert.match(output, /docker info/); + assert.match(output, /Start Docker/); +}); + +// --------------------------------------------------------------------------- +// Container base-image pre-pull (dev-prepull task) +// --------------------------------------------------------------------------- + +test("parseDockerfileBaseImages: single-stage FROM", () => { + const dockerfile = `FROM docker.io/cloudflare/sandbox:0.12.3\nWORKDIR /app\n`; + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["docker.io/cloudflare/sandbox:0.12.3"]); +}); + +test("parseDockerfileBaseImages: multi-stage build — a later FROM referencing an earlier stage's alias is excluded", () => { + const dockerfile = [ + "FROM golang:1.20 AS build", + "RUN go build ./...", + "FROM build AS test", + "RUN go test ./...", + "FROM alpine:3.19", + "COPY --from=test /bin/app /app", + ].join("\n"); + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["golang:1.20", "alpine:3.19"]); +}); + +test("parseDockerfileBaseImages: matches containers/o11y/Dockerfile's real shape — two real images, no stage-name leakage", () => { + const dockerfile = ["FROM grafana/loki:3.3.2 AS loki", "FROM grafana/grafana:11.4.0", "COPY --from=loki /usr/bin/loki /usr/bin/loki"].join("\n"); + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["grafana/loki:3.3.2", "grafana/grafana:11.4.0"]); +}); + +test("parseDockerfileBaseImages: FROM scratch is excluded (never pulled)", () => { + const dockerfile = ["FROM golang:1.20 AS build", "RUN go build -o /app", "FROM scratch", "COPY --from=build /app /app"].join("\n"); + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["golang:1.20"]); +}); + +test("parseDockerfileBaseImages: ARG-based FROM resolves against the ARG's own default", () => { + const dockerfile = ["ARG BASE_IMAGE=alpine:3.19", "FROM ${BASE_IMAGE}", "RUN echo hi"].join("\n"); + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["alpine:3.19"]); +}); + +test("parseDockerfileBaseImages: dedupes an image reused across stages", () => { + const dockerfile = ["FROM node:20 AS a", "FROM node:20 AS b", "FROM node:20"].join("\n"); + assert.deepEqual(parseDockerfileBaseImages(dockerfile), ["node:20"]); +}); + +test("containerWranglerConfigsForTier: tier=1 needs none, tier=2 needs only the API worker, tier=full needs API + o11y", () => { + const root = "/runner"; + assert.deepEqual(containerWranglerConfigsForTier("1", root), []); + assert.deepEqual(containerWranglerConfigsForTier("2", root), [path.join(root, "workers", "api", "wrangler.jsonc")]); + assert.deepEqual(containerWranglerConfigsForTier("full", root), [ + path.join(root, "workers", "api", "wrangler.jsonc"), + path.join(root, "workers", "o11y", "wrangler.jsonc"), + ]); +}); + +test("readContainerDockerfilePaths: reads containers[].image from the real workers/api/wrangler.jsonc", () => { + const paths = readContainerDockerfilePaths(path.join(RUNNER_ROOT, "workers", "api", "wrangler.jsonc")); + assert.equal(paths.length, 2); + assert.ok(paths.some((p) => p.endsWith(path.join("containers", "live", "Dockerfile")))); + assert.ok(paths.some((p) => p.endsWith(path.join("containers", "builder", "Dockerfile")))); + for (const p of paths) assert.equal(existsSync(p), true); +}); + +test("collectTierBaseImages: tier=2 against the real repo resolves the shared sandbox base image once", () => { + const refs = collectTierBaseImages("2", RUNNER_ROOT); + assert.deepEqual(refs, ["docker.io/cloudflare/sandbox:0.12.3"]); +}); + +test("collectTierBaseImages: tier=full also pulls in the o11y worker's two real base images", () => { + const refs = collectTierBaseImages("full", RUNNER_ROOT); + assert.deepEqual(refs, ["docker.io/cloudflare/sandbox:0.12.3", "grafana/loki:3.3.2", "grafana/grafana:11.4.0"]); +}); + +test("shouldCheckContainerImages: true for tier 2/full unless --skip-image-check; always false for tier 1", () => { + assert.equal(shouldCheckContainerImages("2", false), true); + assert.equal(shouldCheckContainerImages("full", false), true); + assert.equal(shouldCheckContainerImages("2", true), false); + assert.equal(shouldCheckContainerImages("full", true), false); + assert.equal(shouldCheckContainerImages("1", false), false); + assert.equal(shouldCheckContainerImages("1", true), false); +}); + +test("parseArgs: --skip-image-check is only valid with --tier=2 or --tier=full", () => { + assert.equal(parseArgs(["--tier=2", "--skip-image-check"]).errors.length, 0); + assert.equal(parseArgs(["--tier=2", "--skip-image-check"]).skipImageCheck, true); + assert.equal(parseArgs(["--tier=full", "--skip-image-check"]).errors.length, 0); + assert.equal(parseArgs(["--tier=1", "--skip-image-check"]).errors.length, 1); + assert.equal(parseArgs(["--help", "--skip-image-check"]).errors.length, 0); +}); + +test("parseArgs: --skip-image-check defaults to false", () => { + assert.equal(parseArgs(["--tier=2"]).skipImageCheck, false); +}); + +test("isImagePresent: true when `docker image inspect` succeeds", () => { + assert.equal( + isImagePresent("alpine:3.19", () => {}), + true, + ); +}); + +test("isImagePresent: false when `docker image inspect` throws (image missing locally)", () => { + assert.equal( + isImagePresent("alpine:3.19", () => { + throw new Error("No such image"); + }), + false, + ); +}); + +test("ensureContainerImagesPresent: a PRESENT image is never pulled", async () => { + const calls = []; + const result = await ensureContainerImagesPresent({ + refs: ["alpine:3.19"], + execFileSyncImpl: (cmd, args) => { + calls.push(args); + if (args[0] === "image" && args[1] === "inspect") return "ok"; + throw new Error(`unexpected call: docker ${args.join(" ")}`); + }, + sleep: () => Promise.resolve(), + }); + assert.equal(result.ok, true); + assert.equal( + calls.some((a) => a[0] === "pull"), + false, + "a present image must never trigger docker pull", + ); +}); + +test("ensureContainerImagesPresent: a MISSING image is pulled exactly once (succeeds first try)", async () => { + const calls = []; + const result = await ensureContainerImagesPresent({ + refs: ["alpine:3.19"], + execFileSyncImpl: (cmd, args) => { + calls.push(args); + if (args[0] === "image" && args[1] === "inspect") throw new Error("No such image"); + if (args[0] === "pull") return "ok"; + throw new Error(`unexpected call: docker ${args.join(" ")}`); + }, + sleep: () => Promise.resolve(), + }); + assert.equal(result.ok, true); + const pullCalls = calls.filter((a) => a[0] === "pull"); + assert.equal(pullCalls.length, 1); + assert.deepEqual(pullCalls[0], ["pull", "alpine:3.19"]); +}); + +test("pullImageWithRetry: retries up to maxAttempts with backoff, then reports the last error line", async () => { + let attempts = 0; + const sleeps = []; + const result = await pullImageWithRetry({ + ref: "alpine:3.19", + execFileSyncImpl: () => { + attempts += 1; + const err = new Error("pull failed"); + err.stderr = Buffer.from(`Error response from daemon: Get "https://registry-1.docker.io/v2/": net/http: TLS handshake timeout\n`); + throw err; + }, + maxAttempts: 3, + backoffMs: 10, + sleep: (ms) => { + sleeps.push(ms); + return Promise.resolve(); + }, + }); + assert.equal(attempts, 3); + assert.equal(sleeps.length, 2); // no sleep after the last attempt + assert.equal(result.ok, false); + assert.match(result.lastErrorLine, /TLS handshake timeout/); +}); + +test("pullImageWithRetry: a real `docker pull` writes progress to stdout and the actual failure to stderr — the stderr line must win, not stdout's later one", async () => { + const result = await pullImageWithRetry({ + ref: "docker.io/cloudflare/sandbox:0.12.3", + execFileSyncImpl: () => { + const err = new Error("pull failed"); + // Matches real `docker pull` output shape: per-layer progress on + // stdout keeps writing lines AFTER stderr's own last write (the + // process failing mid-pull, not at the very start) — a naive + // "concat stdout after stderr, take the last line" extraction would + // report the harmless stdout progress line instead of a spurious error. + err.stdout = Buffer.from("0.12.3: Pulling from cloudflare/sandbox\nabc123: Downloading [==> ] 12MB/48MB\n"); + err.stderr = Buffer.from("error pulling image configuration: download failed after attempts=6: context deadline exceeded\n"); + throw err; + }, + maxAttempts: 1, + sleep: () => Promise.resolve(), + }); + assert.equal(result.ok, false); + assert.match(result.lastErrorLine, /context deadline exceeded/); + assert.doesNotMatch(result.lastErrorLine, /Downloading/); +}); + +test("ensureContainerImagesPresent: a pull that fails every attempt stops before checking any later ref", async () => { + const calls = []; + const result = await ensureContainerImagesPresent({ + refs: ["alpine:3.19", "busybox:1.36"], + execFileSyncImpl: (cmd, args) => { + calls.push(args); + if (args[0] === "image" && args[1] === "inspect") throw new Error("No such image"); + if (args[0] === "pull") { + const err = new Error("pull failed"); + err.stderr = Buffer.from("Error response from daemon: some network error\n"); + throw err; + } + throw new Error(`unexpected call: docker ${args.join(" ")}`); + }, + maxAttempts: 3, + backoffMs: 5, + sleep: () => Promise.resolve(), + }); + assert.equal(result.ok, false); + assert.equal(result.ref, "alpine:3.19"); + assert.match(result.lastErrorLine, /some network error/); + assert.ok( + calls.every((a) => a[a.length - 1] !== "busybox:1.36"), + "the second ref must never be checked once the first one exhausts its retries", + ); +}); + +test("formatImagePullFailure: names the image, the last error line, the retry command, and the escape hatch", () => { + const message = formatImagePullFailure({ ref: "docker.io/cloudflare/sandbox:0.12.3", lastErrorLine: "TLS handshake timeout" }); + assert.match(message, /docker\.io\/cloudflare\/sandbox:0\.12\.3/); + assert.match(message, /TLS handshake timeout/); + assert.match(message, /docker pull docker\.io\/cloudflare\/sandbox:0\.12\.3/); + assert.match(message, /--skip-image-check/); +}); + +test("CLI: `dev.mjs --tier=2` stops before spawning any worker when a required base image fails to pull after every retry", () => { + const stubBinDir = path.join(HERE, "fixtures", "stub-bin"); + const devScript = path.join(RUNNER_ROOT, "scripts", "dev.mjs"); + const result = spawnSync(process.execPath, [devScript, "--tier=2"], { + cwd: RUNNER_ROOT, + encoding: "utf8", + timeout: 30000, + env: { + ...process.env, + PATH: `${stubBinDir}:${process.env.PATH}`, + STUB_DOCKER_MODE: "ok", + STUB_DOCKER_IMAGE_PRESENT: "0", + STUB_DOCKER_PULL_MODE: "fail", + }, + }); + assert.notEqual(result.status, 0); + const output = `${result.stdout}${result.stderr}`; + assert.match(output, /could not pull required container base image/); + assert.match(output, /docker pull docker\.io\/cloudflare\/sandbox:0\.12\.3/); + assert.match(output, /attempt 3\/3/, "the bounded retry must actually run through the real CLI, not just report ok:false"); + assert.doesNotMatch(output, /spawning:/, "no worker should ever be spawned once the image pull gate fails"); + // The stub docker has no "ps" handler (the very next docker call after + // the image gate, listing containers) — its catch-all reply is "stub + // docker: unsupported subcommand ps". Its ABSENCE here is what actually + // proves this run stopped at the image gate and never reached that next + // step, not merely that it exited non-zero for some other reason. + assert.doesNotMatch(output, /unsupported subcommand/, "the run must stop at the image gate, never reaching the next docker call (docker ps)"); +}); + +test("CLI: `dev.mjs --tier=2 --skip-image-check` never calls `docker image inspect`/`pull` even when they'd fail", () => { + const stubBinDir = path.join(HERE, "fixtures", "stub-bin"); + const devScript = path.join(RUNNER_ROOT, "scripts", "dev.mjs"); + // `docker info` (the tier's own Docker-availability check, ahead of the + // image gate this test targets) succeeds via STUB_DOCKER_MODE=ok. The + // stub doesn't implement `docker ps` (the leftover-container baseline + // that runs right after the image gate), so this run dies there — fine, + // and fast: everything this test needs to observe (the skip line, and + // the absence of any image inspect/pull attempt) has already happened by + // then, and it proves nothing past the gate got anywhere near a real + // `wrangler`/pnpm build. + const result = spawnSync(process.execPath, [devScript, "--tier=2", "--skip-image-check"], { + cwd: RUNNER_ROOT, + encoding: "utf8", + timeout: 10000, + env: { + ...process.env, + PATH: `${stubBinDir}:${process.env.PATH}`, + STUB_DOCKER_MODE: "ok", + STUB_DOCKER_IMAGE_PRESENT: "0", + STUB_DOCKER_PULL_MODE: "fail", + }, + }); + const output = `${result.stdout}${result.stderr}`; + assert.doesNotMatch(output, /could not pull required container base image/); + assert.match(output, /--skip-image-check: skipping/); + // Proves this run DID proceed past the (skipped) gate, all the way to + // the next docker call the stub doesn't implement (`docker ps`) — the + // control for the test above: same env (a pull would fail if attempted), + // but with --skip-image-check the run gets past the gate instead of + // stopping at it. + assert.match(output, /unsupported subcommand/, "the run must proceed past the (skipped) gate to the next docker call"); +}); + +// --------------------------------------------------------------------------- +// runtime staleness +// --------------------------------------------------------------------------- + +test("isRuntimeDistStale: true when dist/ is missing", () => { + withTmpDir((dir) => { + mkdirSync(path.join(dir, "src")); + writeFileSync(path.join(dir, "src", "index.ts"), "export {}\n"); + assert.equal(isRuntimeDistStale(dir), true); + }); +}); + +test("isRuntimeDistStale: false when dist/ is newer than every src file", async () => { + await withTmpDir(async (dir) => { + mkdirSync(path.join(dir, "src")); + mkdirSync(path.join(dir, "dist")); + writeFileSync(path.join(dir, "src", "index.ts"), "export {}\n"); + await new Promise((r) => setTimeout(r, 20)); + writeFileSync(path.join(dir, "dist", "index.js"), "export {};\n"); + assert.equal(isRuntimeDistStale(dir), false); + }); +}); + +test("isRuntimeDistStale: true when a src file was edited after the last dist build", async () => { + await withTmpDir(async (dir) => { + mkdirSync(path.join(dir, "src")); + mkdirSync(path.join(dir, "dist")); + writeFileSync(path.join(dir, "dist", "index.js"), "export {};\n"); + await new Promise((r) => setTimeout(r, 20)); + writeFileSync(path.join(dir, "src", "index.ts"), "export {}\n"); + assert.equal(isRuntimeDistStale(dir), true); + }); +}); + +// --------------------------------------------------------------------------- +// pnpm install staleness ("stale dependencies after a pull" dev-stack note) +// --------------------------------------------------------------------------- + +/** Sets an exact, controllable mtime — successive `writeFileSync` calls can + * land on the same filesystem-clock tick, which would make a real ordering + * bug read as a pass here just as easily as a real fix. */ +function touch(filePath, mtimeMs) { + const seconds = mtimeMs / 1000; + utimesSync(filePath, seconds, seconds); +} + +test("isPnpmInstallNeeded: false when there is no lockfile at all (nothing to detect drift against)", () => { + withTmpDir((dir) => { + assert.equal(isPnpmInstallNeeded(dir), false); + }); +}); + +test("isPnpmInstallNeeded: true when node_modules/.modules.yaml is missing outright (never installed)", () => { + withTmpDir((dir) => { + writeFileSync(path.join(dir, "pnpm-lock.yaml"), "lockfileVersion: '9.0'\n"); + assert.equal(isPnpmInstallNeeded(dir), true); + }); +}); + +test("isPnpmInstallNeeded: false when node_modules/.modules.yaml is newer than the lockfile (installed after the last lockfile change)", () => { + withTmpDir((dir) => { + const lockfilePath = path.join(dir, "pnpm-lock.yaml"); + writeFileSync(lockfilePath, "lockfileVersion: '9.0'\n"); + mkdirSync(path.join(dir, "node_modules")); + const modulesYamlPath = path.join(dir, "node_modules", ".modules.yaml"); + writeFileSync(modulesYamlPath, "hoistedDependencies: {}\n"); + touch(lockfilePath, 1_000_000); + touch(modulesYamlPath, 2_000_000); + assert.equal(isPnpmInstallNeeded(dir), false); + }); +}); + +// This is the exact "stale dependencies after a pull" shape: a pull changed +// the lockfile (new dependency), and node_modules was never reinstalled +// against it — the one case this check exists to catch. +test("isPnpmInstallNeeded: true when the lockfile is newer than node_modules/.modules.yaml (a pull changed dependencies, never reinstalled)", () => { + withTmpDir((dir) => { + const lockfilePath = path.join(dir, "pnpm-lock.yaml"); + writeFileSync(lockfilePath, "lockfileVersion: '9.0'\n"); + mkdirSync(path.join(dir, "node_modules")); + const modulesYamlPath = path.join(dir, "node_modules", ".modules.yaml"); + writeFileSync(modulesYamlPath, "hoistedDependencies: {}\n"); + touch(modulesYamlPath, 1_000_000); + touch(lockfilePath, 2_000_000); + assert.equal(isPnpmInstallNeeded(dir), true); + }); +}); + +test("PNPM_INSTALL_NEEDED_MESSAGE names the exact fix command", () => { + assert.match(PNPM_INSTALL_NEEDED_MESSAGE, /pnpm install --frozen-lockfile/); +}); + +// --------------------------------------------------------------------------- +// Surfacing a wrangler build failure instead of waiting out the full +// readiness timeout ("stale dependencies after a pull" dev-stack note) +// --------------------------------------------------------------------------- + +test("wranglerBuildErrorLine: matches a real esbuild-shaped build failure", () => { + assert.equal( + wranglerBuildErrorLine('✘ [ERROR] Could not resolve "@jridgewell/trace-mapping"'), + 'Could not resolve "@jridgewell/trace-mapping"', + ); +}); + +test("wranglerBuildErrorLine: strips ANSI color codes before matching (wrangler 4.108's real output shape)", () => { + assert.equal( + wranglerBuildErrorLine("\x1b[31m✘ [ERROR]\x1b[0m Could not resolve \"@jridgewell/trace-mapping\""), + 'Could not resolve "@jridgewell/trace-mapping"', + ); +}); + +test("wranglerBuildErrorLine: null for an ordinary log line", () => { + assert.equal(wranglerBuildErrorLine("[o11y] Ready on http://localhost:4200"), null); +}); + +// Two real examples that must not be treated as a build failure: wrangler's +// runtime uncaught-exception logging reuses the identical `✘ [ERROR]` +// prefix for a request handler throwing at runtime — the worker came up +// fine and is already serving traffic — which is the opposite of "never +// came up". +test("wranglerBuildErrorLine: null for wrangler's own runtime uncaught-exception logging, not a build failure", () => { + assert.equal( + wranglerBuildErrorLine("✘ [ERROR] Uncaught Error: No such image available named cloudflare-dev/sandbox:f01d8965"), + null, + ); + assert.equal( + wranglerBuildErrorLine("✘ [ERROR] Uncaught Error: ReadableStream received over RPC disconnected prematurely."), + null, + ); +}); + +// --------------------------------------------------------------------------- +// waitForServer +// --------------------------------------------------------------------------- + +test("waitForServer: resolves once fetchImpl stops rejecting", async () => { + let calls = 0; + const fetchImpl = async () => { + calls += 1; + if (calls < 3) throw new Error("connection refused"); + }; + await waitForServer("http://localhost:1", 10_000, "test worker", { + fetchImpl, + sleepImpl: async () => {}, // instant — this test proves the retry, not real timing + }); + assert.equal(calls, 3); +}); + +test("waitForServer: times out with the generic message when nothing signals an early failure", async () => { + await assert.rejects( + waitForServer("http://localhost:1", 20, "test worker", { + fetchImpl: async () => { + throw new Error("connection refused"); + }, + sleepImpl: async () => {}, // instant — the deadline is real time (Date.now()), not sleep count + }), + /test worker on http:\/\/localhost:1 never came up within 20ms/, + ); +}); + +// Proves the actual race dev.mjs depends on: an early failure must win EVEN +// WHEN timeoutMs is huge and no time has elapsed yet — this is what makes it +// "immediately" rather than "eventually, once the timeout would have fired +// anyway". `sleepImpl` throws if called at all: a `waitForServer` that only +// checked `getEarlyFailure` AFTER sleeping (instead of on every failed +// attempt, before sleeping again) would call `sleepImpl` at least once +// before ever reporting the build error, and this assertion would catch +// that revert. +test("waitForServer: an early failure throws immediately, without ever sleeping, however large timeoutMs is", async () => { + await assert.rejects( + waitForServer("http://localhost:1", 120_000, "o11y worker", { + fetchImpl: async () => { + throw new Error("connection refused"); + }, + getEarlyFailure: () => 'Could not resolve "@jridgewell/trace-mapping"', + sleepImpl: () => { + throw new Error("must not sleep once an early failure is reported"); + }, + }), + /o11y worker on http:\/\/localhost:1 failed to build: Could not resolve "@jridgewell\/trace-mapping"/, + ); +}); + +// --------------------------------------------------------------------------- +// spawn plan +// --------------------------------------------------------------------------- + +test("buildPlan: tier=1 spawns only the authoring app", () => { + const ports = resolvePorts("1", {}); + assert.deepEqual( + buildPlan("1", ports).map((p) => p.name), + ["app"], + ); +}); + +test("buildPlan: tier=2 spawns the authoring app and the api worker, in that order", () => { + const ports = resolvePorts("2", {}); + assert.deepEqual( + buildPlan("2", ports).map((p) => p.name), + ["app", "api"], + ); +}); + +test("buildPlan: tier=full spawns app, api, o11y, and the local Slack capture server", () => { + const ports = resolvePorts("full", {}); + assert.deepEqual( + buildPlan("full", ports).map((p) => p.name), + ["app", "api", "o11y", "slack"], + ); +}); + +test("buildPlan: api and o11y each get their own, distinct --port and --inspector-port flags", () => { + const ports = resolvePorts("full", {}); + const plan = buildPlan("full", ports); + const api = plan.find((p) => p.name === "api"); + const o11y = plan.find((p) => p.name === "o11y"); + const flagValue = (args, flag) => args[args.indexOf(flag) + 1]; + assert.equal(flagValue(api.args, "--port"), String(ports.API_DEV_PORT)); + assert.equal(flagValue(api.args, "--inspector-port"), String(ports.API_DEV_INSPECTOR_PORT)); + assert.equal(flagValue(o11y.args, "--port"), String(ports.O11Y_DEV_PORT)); + assert.equal(flagValue(o11y.args, "--inspector-port"), String(ports.O11Y_DEV_INSPECTOR_PORT)); + assert.notEqual(flagValue(api.args, "--inspector-port"), flagValue(o11y.args, "--inspector-port")); +}); + +test("buildPlan: o11y's spawn injects O11Y_SESSION_SECRET via --var, not a fixed/predictable value", () => { + const ports = resolvePorts("full", {}); + const planA = buildPlan("full", ports); + const planB = buildPlan("full", ports); + const secretArg = (plan) => { + const o11y = plan.find((p) => p.name === "o11y"); + const idx = o11y.args.indexOf("--var"); + for (let i = idx; i < o11y.args.length; i += 2) { + if (o11y.args[i] === "--var" && o11y.args[i + 1].startsWith("O11Y_SESSION_SECRET:")) return o11y.args[i + 1]; + } + return undefined; + }; + const a = secretArg(planA); + const b = secretArg(planB); + assert.ok(a && a.startsWith("O11Y_SESSION_SECRET:")); + assert.notEqual(a, b, "each build gets a fresh ephemeral secret unless one is pinned via opts.sessionSecret"); +}); + +test("buildPlan: tier=1's app process gets no VITE_DEV_USER/VITE_API_BASE (no API worker running to point at)", () => { + const ports = resolvePorts("1", {}); + const app = buildPlan("1", ports).find((p) => p.name === "app"); + assert.equal(app.env.VITE_DEV_USER, undefined); + assert.equal(app.env.VITE_API_BASE, undefined); +}); + +test("buildPlan: tier=2/full inject VITE_DEV_USER/VITE_API_BASE as env (never written to a file) for the app process", () => { + for (const tier of ["2", "full"]) { + const ports = resolvePorts(tier, {}); + const app = buildPlan(tier, ports).find((p) => p.name === "app"); + assert.equal(app.env.VITE_DEV_USER, "dev@handsontable.com"); + assert.equal(app.env.VITE_API_BASE, `http://localhost:${ports.AUTHORING_DEV_PORT}`); + } +}); + +test("buildPlan: tier=full additionally injects VITE_TELEMETRY_LOCAL=1 for the app process", () => { + const ports = resolvePorts("full", {}); + const app = buildPlan("full", ports).find((p) => p.name === "app"); + assert.equal(app.env.VITE_TELEMETRY_LOCAL, "1"); + const tier2App = buildPlan("2", resolvePorts("2", {})).find((p) => p.name === "app"); + assert.equal(tier2App.env.VITE_TELEMETRY_LOCAL, undefined); +}); + +test("buildPlan: tier=full injects VITE_GRAFANA_URL for the app process, at the o11y worker's own origin (not AUTHORING/O11Y_GRAFANA_PORT)", () => { + const ports = resolvePorts("full", { O11Y_DEV_PORT: "6223", AUTHORING_DEV_PORT: "6220" }); + const app = buildPlan("full", ports).find((p) => p.name === "app"); + assert.equal(app.env.VITE_GRAFANA_URL, "http://localhost:6223/grafana/"); +}); + +test("buildPlan: only tier=full injects VITE_GRAFANA_URL — tier=1 and tier=2 leave it unset so a build without dev.mjs falls back to the app's own default", () => { + for (const tier of ["1", "2"]) { + const ports = resolvePorts(tier, {}); + const app = buildPlan(tier, ports).find((p) => p.name === "app"); + assert.equal(app.env.VITE_GRAFANA_URL, undefined, `tier=${tier} must not set VITE_GRAFANA_URL`); + } +}); + +test("buildPlan: never spawns wrangler via npx (spawns node_modules/.bin/wrangler directly)", () => { + const ports = resolvePorts("full", {}); + for (const proc of buildPlan("full", ports)) { + assert.notEqual(proc.bin, "npx"); + assert.doesNotMatch(proc.bin, /^npx\b/); + } +}); + +// --------------------------------------------------------------------------- +// container reporting, never stopping: Ctrl-C does not make wrangler's own +// Sandbox-container orchestration tear itself down synchronously, and +// several worktrees running `wrangler dev` on this same machine at once is +// the normal case — a "new since my own snapshot" + name-match container +// can just as easily be another worktree's session as this run's own, so +// this module must never `docker stop` one on a guess. +// --------------------------------------------------------------------------- + +test("possiblyLeftoverContainers: only a container absent from `before` AND matching this run's own worker names counts", () => { + const before = new Set(["existing-1"]); + const after = [ + { id: "existing-1", name: "workerd-handsontable-demos-api-Sandbox-xyz-proxy" }, // pre-existing — not a candidate + { id: "new-1", name: "workerd-handsontable-demos-api-Sandbox-abc-proxy" }, // new + matches — a candidate + { id: "new-2", name: "workerd-handsontable-demos-o11y-GrafanaBox-def-proxy" }, // new + matches — a candidate + { id: "new-3", name: "some-unrelated-container" }, // new but does not match — never a candidate + ]; + const candidates = possiblyLeftoverContainers(before, after); + assert.deepEqual( + candidates.map((c) => c.id).sort(), + ["new-1", "new-2"], + ); +}); + +test("possiblyLeftoverContainers: empty when nothing new appeared", () => { + const before = new Set(["a", "b"]); + const after = [ + { id: "a", name: "workerd-handsontable-demos-api-Sandbox-1-proxy" }, + { id: "b", name: "workerd-handsontable-demos-api-Sandbox-2-proxy" }, + ]; + assert.deepEqual(possiblyLeftoverContainers(before, after), []); +}); + +test("possiblyLeftoverContainers: never flags an unrelated container even if it's new (backend-postgres, mongodb, another worktree's own service)", () => { + const before = new Set(); + const after = [ + { id: "x", name: "backend-postgres-1" }, + { id: "y", name: "myhandsontable-mongodb" }, + ]; + assert.deepEqual(possiblyLeftoverContainers(before, after), []); +}); + +test("reportLeftoverContainers (the required stubbed-docker test): a foreign container that appears new during the session, matching this run's own worker-name pattern, is REPORTED but never stopped", () => { + // Simulates a false positive: worktree B + // starts its own `wrangler dev`/Tier-2 session partway through worktree + // A's (this run's) session. B's `workerd-handsontable-demos-api-Sandbox-*` + // container is "new since A's snapshot" and matches the name pattern — + // indistinguishable, by this signal alone, from a container A actually + // started itself. + const before = new Set(["existing-1"]); + const dockerCalls = []; + const execFileSyncImpl = (cmd, args) => { + dockerCalls.push([cmd, ...args]); + if (cmd !== "docker") throw new Error(`unexpected command: ${cmd}`); + if (args[0] === "stop") { + // The regression this test guards against: naive code + // ran `docker stop` on a container it could not prove was its own. + throw new Error("docker stop must NEVER be called by reportLeftoverContainers — NB2 regression"); + } + if (args[0] === "ps") { + return [ + "existing-1\tworkerd-handsontable-demos-api-Sandbox-preexisting-proxy", + "foreign-1\tworkerd-handsontable-demos-api-Sandbox-foreign-worktree-proxy", + ].join("\n"); + } + throw new Error(`unexpected docker subcommand: ${args.join(" ")}`); + }; + const logLines = []; + const candidates = reportLeftoverContainers(before, execFileSyncImpl, (msg) => logLines.push(msg)); + + assert.deepEqual(candidates.map((c) => c.id), ["foreign-1"], "the foreign container is still correctly IDENTIFIED as a candidate"); + assert.ok( + !dockerCalls.some(([, sub]) => sub === "stop"), + "docker stop must never be invoked, even for a container that looks exactly like this run's own", + ); + assert.equal(logLines.length, 1, "exactly one informational log line, no automatic action"); + assert.match(logLines[0], /NOT stopping/); + assert.match(logLines[0], /docker stop foreign-1/, "the manual cleanup command is printed for a human to run"); +}); + +test("SHUTDOWN_SIGNALS: includes SIGHUP alongside SIGINT/SIGTERM, so closing the terminal a detached session was started from still triggers cleanup", () => { + assert.deepEqual([...SHUTDOWN_SIGNALS].sort(), ["SIGHUP", "SIGINT", "SIGTERM"]); +}); + +// --------------------------------------------------------------------------- +// o11yLocalPublicOrigin +// --------------------------------------------------------------------------- + +test("o11yLocalPublicOrigin: tracks O11Y_DEV_PORT, not AUTHORING_DEV_PORT — Grafana is served from the o11y worker's own origin", () => { + assert.equal(o11yLocalPublicOrigin({ O11Y_DEV_PORT: 4200, AUTHORING_DEV_PORT: 5173 }), "http://localhost:4200"); + assert.equal(o11yLocalPublicOrigin({ O11Y_DEV_PORT: 6223, AUTHORING_DEV_PORT: 6220 }), "http://localhost:6223"); +}); + +test("buildPlan: tier=full's o11y spawn injects O11Y_LOCAL_PUBLIC_ORIGIN matching the resolved O11Y_DEV_PORT", () => { + const ports = resolvePorts("full", { O11Y_DEV_PORT: "6223" }); + const o11y = buildPlan("full", ports).find((p) => p.name === "o11y"); + assert.ok(o11y.args.includes("O11Y_LOCAL_PUBLIC_ORIGIN:http://localhost:6223")); +}); + +// --------------------------------------------------------------------------- +// --fresh (dev-persist task): compose.yml's minio/clickhouse now use named +// volumes; --fresh wipes them + workers/o11y/.wrangler/state together. +// --------------------------------------------------------------------------- + +test("parseArgs: --fresh is only valid with --tier=full", () => { + assert.equal(parseArgs(["--tier=full", "--fresh"]).errors.length, 0); + assert.equal(parseArgs(["--tier=full", "--fresh"]).fresh, true); + assert.equal(parseArgs(["--tier=1", "--fresh"]).errors.length, 1); + assert.equal(parseArgs(["--tier=2", "--fresh"]).errors.length, 1); + // Allowed with --help and no --tier (mirrors --replay/--reset-local-db). + assert.equal(parseArgs(["--help", "--fresh"]).errors.length, 0); +}); + +test("parseArgs: --fresh defaults to false", () => { + assert.equal(parseArgs(["--tier=full"]).fresh, false); +}); + +test("composeDownArgs: no -v by default (the Ctrl-C/kept-data path); -v only when fresh", () => { + const plain = composeDownArgs("/x/compose.yml"); + assert.deepEqual(plain, ["compose", "-f", "/x/compose.yml", "down"]); + assert.ok(!plain.includes("-v")); + + const fresh = composeDownArgs("/x/compose.yml", { fresh: true }); + assert.deepEqual(fresh, ["compose", "-f", "/x/compose.yml", "down", "-v"]); +}); + +test("o11yDevDataModeLine: exact startup mode line for both cases", () => { + assert.equal(o11yDevDataModeLine(false), "o11y local data: kept (MinIO/ClickHouse volumes + o11y worker state)"); + assert.equal(o11yDevDataModeLine(true), "o11y local data: fresh"); +}); + +test("resetO11yLocalState: runs `docker compose down -v` scoped to the given project, and removes only <o11yDir>/.wrangler/state", () => { + withTmpDir((dir) => { + const o11yDir = path.join(dir, "workers", "o11y"); + const otherDir = path.join(dir, "workers", "api"); // must never be touched + mkdirSync(path.join(o11yDir, ".wrangler", "state", "v3", "do"), { recursive: true }); + writeFileSync(path.join(o11yDir, ".wrangler", "state", "v3", "do", "marker.txt"), "x"); + // A file directly under o11yDir (a sibling of .wrangler/, not under it) + // — this is what actually catches a rm-path widened to o11yDir itself + // (or to `dir`): the `.wrangler/state` assertion below stays trivially + // true either way (a deleted parent takes every child path down with + // it), this one does not. + writeFileSync(path.join(o11yDir, ".dev.vars"), "O11Y_ENV=local\n"); + mkdirSync(path.join(otherDir, ".wrangler", "state"), { recursive: true }); + writeFileSync(path.join(otherDir, ".wrangler", "state", "keep-me.txt"), "x"); + + const calls = []; + const execFileSyncImpl = (cmd, args, opts) => calls.push({ cmd, args, opts }); + const composeFile = "/x/compose.yml"; + const composeEnv = { COMPOSE_PROJECT_NAME: "o11y-q1-test" }; + + const result = resetO11yLocalState({ o11yDir, composeFile, composeEnv, execFileSyncImpl }); + + assert.equal(calls.length, 1, "exactly one docker invocation"); + assert.equal(calls[0].cmd, "docker"); + assert.ok(calls[0].args.includes("-v"), "down -v (the whole point of --fresh)"); + assert.deepEqual(calls[0].args, ["compose", "-f", composeFile, "down", "-v"]); + assert.equal(calls[0].opts.env.COMPOSE_PROJECT_NAME, "o11y-q1-test", "scoped to the right project only"); + + assert.equal(result.composeDownRan, true); + assert.equal(result.stateDirRemoved, true); + assert.equal(existsSync(path.join(o11yDir, ".wrangler", "state")), false, "o11y worker state dir removed"); + assert.equal(existsSync(path.join(o11yDir, ".dev.vars")), true, "rm scoped to .wrangler/state, not all of o11yDir"); + assert.equal(existsSync(path.join(otherDir, ".wrangler", "state", "keep-me.txt")), true, "workers/api's own state untouched"); + }); +}); + +test("resetO11yLocalState: without composeFile/composeEnv (o11y:dev's own --fresh), no docker call is made at all", () => { + withTmpDir((dir) => { + const o11yDir = path.join(dir, "workers", "o11y"); + mkdirSync(path.join(o11yDir, ".wrangler", "state"), { recursive: true }); + const execFileSyncImpl = () => { + throw new Error("must not be called — o11y:dev never runs docker compose"); + }; + const result = resetO11yLocalState({ o11yDir, execFileSyncImpl }); + assert.equal(result.composeDownRan, false); + assert.equal(result.stateDirRemoved, true); + assert.equal(existsSync(path.join(o11yDir, ".wrangler", "state")), false); + }); +}); + +test("resetO11yLocalState: logs 'nothing to delete' when there is no o11y worker state at all (never throws)", () => { + withTmpDir((dir) => { + const o11yDir = path.join(dir, "workers", "o11y"); + const lines = []; + const result = resetO11yLocalState({ o11yDir, log: (l) => lines.push(l) }); + assert.equal(result.stateDirRemoved, false); + assert.ok(lines.some((l) => l.includes("nothing to delete"))); + }); +}); + +// --------------------------------------------------------------------------- +// per-worktree default COMPOSE_PROJECT_NAME: a fixed literal "o11y-dev" +// would make every worktree's `--tier=full` resolve to the same compose +// project, so one worktree's `--fresh` (or even a plain Ctrl-C) could +// wipe/stop another worktree's stack. See `defaultComposeProjectName`'s +// own doc comment in dev-lib.mjs. +// --------------------------------------------------------------------------- + +test("defaultComposeProjectName: two different worktree roots give two different names", () => { + const a = defaultComposeProjectName("/Users/dev/Code/examples/runner"); + const b = defaultComposeProjectName("/Users/dev/Code/examples-wt/Q1-persist/runner"); + assert.notEqual(a, b, "two distinct worktrees must never resolve to the same compose project"); + assert.match(a, /^o11y-dev-[0-9a-f]+$/, "must still look like a compose project name (lowercase, hyphen, hex)"); + assert.match(b, /^o11y-dev-[0-9a-f]+$/); +}); + +test("defaultComposeProjectName: the same root gives the same (stable) name every time", () => { + const root = "/Users/dev/Code/examples-wt/Y1-final/runner"; + assert.equal(defaultComposeProjectName(root), defaultComposeProjectName(root), "must be stable across calls/restarts, not randomly generated"); +}); + +// `defaultComposeProjectName` must derive from `runnerRoot`, not a fixed +// `"o11y-dev"` literal, or two different worktree paths would produce the +// same project name — the collision this exists to prevent. + +test("resolveComposeProjectName: an explicit COMPOSE_PROJECT_NAME env override always wins over the derived default", () => { + assert.equal( + resolveComposeProjectName({ COMPOSE_PROJECT_NAME: "my-shared-project" }, "/Users/dev/Code/examples/runner"), + "my-shared-project", + ); +}); + +test("resolveComposeProjectName: falls back to defaultComposeProjectName(runnerRoot) when unset", () => { + const root = "/Users/dev/Code/examples/runner"; + assert.equal(resolveComposeProjectName({}, root), defaultComposeProjectName(root)); + // An empty string is "unset" too (matches every other env-override check in + // this module, e.g. resolvePorts' own `raw !== undefined && raw !== ""`). + assert.equal(resolveComposeProjectName({ COMPOSE_PROJECT_NAME: "" }, root), defaultComposeProjectName(root)); +}); + +// Revert evidence: reverting `resolveComposeProjectName` to always return +// `defaultComposeProjectName(runnerRoot)` (dropping the `env.COMPOSE_PROJECT_NAME +// ||` short-circuit) makes the override test above fail (`"my-shared-project"` +// vs a derived `o11y-dev-<hash>`). + +test("dev.mjs's own --tier=full compose section resolves COMPOSE_PROJECT_NAME via resolveComposeProjectName, not a hardcoded literal (drift guard)", () => { + const devSrc = readFileSync(path.join(RUNNER_ROOT, "scripts", "dev.mjs"), "utf8"); + assert.match( + devSrc, + /const composeProjectName = resolveComposeProjectName\(process\.env\)/, + "dev.mjs must derive composeProjectName through the shared helper (Z-D-H2) so --fresh's own down -v (which reads composeEnv.COMPOSE_PROJECT_NAME straight from this variable) targets THIS worktree's derived project, not a value shared across worktrees", + ); + assert.doesNotMatch( + devSrc, + /process\.env\.COMPOSE_PROJECT_NAME \|\| "o11y-dev"/, + "must not reintroduce the old fixed-literal fallback that collided across worktrees", + ); +}); + +// Behavioural companion to the source-grep drift guard above: reproduces +// dev.mjs's own `--tier=full --fresh` wiring end to end (resolve the +// project name the same way dev.mjs does -> build composeEnv from it -> +// call resetO11yLocalState with it) and asserts the actual docker call +// `resetO11yLocalState` makes carries this worktree's derived project +// name, not a value shared across worktrees. This also catches a +// caller-side mistake (e.g. building `composeEnv` from a different/stale +// variable) that a pure source-text match cannot see. +test("--tier=full's own wiring: --fresh's docker compose down -v carries THIS worktree's derived COMPOSE_PROJECT_NAME (behavioural)", () => { + withTmpDir((dir) => { + const o11yDir = path.join(dir, "workers", "o11y"); + const root = path.join(dir, "some-worktree", "runner"); + // No override — exactly dev.mjs's own `resolveComposeProjectName(process.env)` + // call when COMPOSE_PROJECT_NAME is unset. + const composeProjectName = resolveComposeProjectName({}, root); + const composeFile = path.join(dir, "compose.yml"); + const composeEnv = { COMPOSE_PROJECT_NAME: composeProjectName }; + const calls = []; + const execFileSyncImpl = (cmd, args, opts) => calls.push({ cmd, args, opts }); + + resetO11yLocalState({ o11yDir, composeFile, composeEnv, execFileSyncImpl }); + + assert.equal(calls.length, 1); + assert.deepEqual(calls[0].args, composeDownArgs(composeFile, { fresh: true }), "must be the real --fresh down -v shape"); + assert.equal( + calls[0].opts.env.COMPOSE_PROJECT_NAME, + defaultComposeProjectName(root), + "the docker call actually made must carry this worktree's derived name, not a shared/stale one", + ); + }); +}); + +test("stop-roundtrip.mjs's own dev-stack collision guard imports its default from defaultComposeProjectName, not a second hardcoded literal (drift guard)", () => { + const src = readFileSync(path.join(RUNNER_ROOT, "containers", "o11y", "local", "stop-roundtrip.mjs"), "utf8"); + assert.match( + src, + /import\s*\{\s*defaultComposeProjectName\s*\}\s*from\s*"\.\.\/\.\.\/\.\.\/scripts\/dev-lib\.mjs"/, + "stop-roundtrip.mjs must import the ONE shared helper rather than keeping its own copy of the dev-stack default literal, so the two can never drift apart again", + ); + assert.match(src, /const DEV_STACK_DEFAULT_PROJECT = defaultComposeProjectName\(\)/); +}); + +// `stop-roundtrip.mjs`'s `DEV_STACK_DEFAULT_PROJECT` must not be the fixed +// literal `"o11y-dev"`, or a worktree-derived dev.mjs default would +// silently never match this script's guard, so a `stop-roundtrip.mjs` run +// under this worktree's real dev-stack project name would no longer be +// refused. + +test("run-and-deploy.md documents the per-worktree derivation and what happens to an existing single-worktree user's old volumes (they are NOT renamed — orphaned, not migrated)", () => { + const doc = readFileSync(path.join(RUNNER_ROOT, "docs", "run-and-deploy.md"), "utf8"); + assert.match(doc, /derived PER WORKTREE/i); + assert.match(doc, /defaultComposeProjectName/); + assert.match(doc, /does not rename them/i, "must correctly say docker does NOT rename the old volumes (they are orphaned; the new project starts empty)"); + assert.match(doc, /orphaned/i); + assert.match(doc, /divergence warning/i, "must call out that the existing-ledger case prints detectO11yStateDivergence's warning on the first run under the new project"); +}); + +test("o11yLedgerCommittedKeyCount: counts only 'done:' keys in the InboxWriter DO's real SQLite storage, across multiple .sqlite files", async () => { + await withTmpDir(async (dir) => { + const o11yDir = path.join(dir, "workers", "o11y"); + const inboxDir = path.join(o11yDir, ".wrangler", "state", "v3", "do", "handsontable-demos-o11y-InboxWriter"); + mkdirSync(inboxDir, { recursive: true }); + + function makeKvSqlite(fileName, rows) { + const db = new DatabaseSync(path.join(inboxDir, fileName)); + db.exec("CREATE TABLE _cf_KV (key TEXT PRIMARY KEY, value BLOB) WITHOUT ROWID"); + for (const key of rows) db.prepare("INSERT INTO _cf_KV (key, value) VALUES (?, ?)").run(key, Buffer.from("1")); + db.close(); + } + makeKvSqlite("aaa.sqlite", ["done:inbox/tenant/2026-09-24/one", "done:inbox/tenant/2026-09-24/two", "hash:20260924:abc", "wake:xyz"]); + makeKvSqlite("bbb.sqlite", ["done:inbox/tenant/2026-09-24/three"]); + // metadata.sqlite (real wrangler layout) never has a _cf_KV table — must + // be skipped, not counted as an error. + const metaDb = new DatabaseSync(path.join(inboxDir, "metadata.sqlite")); + metaDb.exec("CREATE TABLE something_else (id INTEGER)"); + metaDb.close(); + + const count = await o11yLedgerCommittedKeyCount(o11yDir); + assert.equal(count, 3); + }); +}); + +test("o11yLedgerCommittedKeyCount: 0 (never throws) when there's no o11y worker state yet", async () => { + await withTmpDir(async (dir) => { + const count = await o11yLedgerCommittedKeyCount(path.join(dir, "workers", "o11y")); + assert.equal(count, 0); + }); +}); + +test("findComposeVolume: null when docker finds nothing for that project+key; the resolved name otherwise", () => { + const found = findComposeVolume({ + composeProjectName: "o11y-q1", + volumeKey: "minio-data", + execFileSyncImpl: () => "o11y-q1_minio-data\n", + }); + assert.equal(found, "o11y-q1_minio-data"); + + const missing = findComposeVolume({ + composeProjectName: "o11y-q1", + volumeKey: "minio-data", + execFileSyncImpl: () => "", + }); + assert.equal(missing, null); +}); + +test("detectO11yStateDivergence: never touches docker when the ledger has zero committed keys (cheap path first)", async () => { + const result = await detectO11yStateDivergence({ + composeProjectName: "o11y-q1", + o11yDir: "/does/not/matter", + execFileSyncImpl: () => { + throw new Error("must not be called — nothing to warn about"); + }, + countCommittedLedgerKeys: async () => 0, + }); + assert.deepEqual(result, { divergent: false, committedCount: 0 }); +}); + +test("detectO11yStateDivergence: divergent when the ledger has committed keys but the MinIO volume is gone (the warning fires)", async () => { + const result = await detectO11yStateDivergence({ + composeProjectName: "o11y-q1", + o11yDir: "/does/not/matter", + execFileSyncImpl: () => "", // docker volume ls -q finds nothing + countCommittedLedgerKeys: async () => 7, + }); + assert.equal(result.divergent, true); + assert.equal(result.committedCount, 7); + assert.match(formatO11yDivergenceWarning(result.committedCount), /--fresh/); + assert.match(formatO11yDivergenceWarning(result.committedCount), /7/); +}); + +test("detectO11yStateDivergence: NOT divergent when committed keys exist but the MinIO volume also exists (normal case, not just fewer bytes)", async () => { + const result = await detectO11yStateDivergence({ + composeProjectName: "o11y-q1", + o11yDir: "/does/not/matter", + execFileSyncImpl: () => "o11y-q1_minio-data\n", + countCommittedLedgerKeys: async () => 7, + }); + assert.equal(result.divergent, false); +}); + +// --------------------------------------------------------------------------- +// drift: every env var / flag the script reads is documented +// --------------------------------------------------------------------------- + +function envVarNames(src) { + const names = new Set(); + const re = /process\.env\.([A-Z][A-Z0-9_]*)/g; + let m; + while ((m = re.exec(src))) names.add(m[1]); + return names; +} + +function extractSection(doc, heading) { + const start = doc.indexOf(heading); + assert.notEqual(start, -1, `heading not found: ${heading}`); + const rest = doc.slice(start + heading.length); + const next = rest.search(/\n## /); + return next === -1 ? rest : rest.slice(0, next); +} + +test("drift: every env var read by dev.mjs/dev-lib.mjs/o11y-dev.mjs is documented in run-and-deploy.md's Run locally section", () => { + const devLibSrc = readFileSync(path.join(RUNNER_ROOT, "scripts", "dev-lib.mjs"), "utf8"); + const devSrc = readFileSync(path.join(RUNNER_ROOT, "scripts", "dev.mjs"), "utf8"); + const o11yDevSrc = readFileSync(path.join(RUNNER_ROOT, "scripts", "o11y-dev.mjs"), "utf8"); + const doc = readFileSync(path.join(RUNNER_ROOT, "docs", "run-and-deploy.md"), "utf8"); + const section = extractSection(doc, "## Run locally"); + + const names = new Set([ + ...envVarNames(devLibSrc), + ...envVarNames(devSrc), + ...envVarNames(o11yDevSrc), + ...Object.keys(PORT_DEFAULTS), + ]); + assert.ok(names.size >= 10, `expected a real set of env var names, got ${names.size}`); + + const missing = [...names].filter((name) => !section.includes(`\`${name}\``)); + assert.deepEqual(missing, [], `env var(s) not documented (as a backtick-wrapped name) in run-and-deploy.md's Run locally section: ${missing.join(", ")}`); +}); + +test("drift: --tier, --replay, --fresh, and --help are documented in run-and-deploy.md's Run locally section", () => { + const doc = readFileSync(path.join(RUNNER_ROOT, "docs", "run-and-deploy.md"), "utf8"); + const section = extractSection(doc, "## Run locally"); + for (const flag of ["--tier", "--replay", "--fresh", "--help"]) { + assert.match(section, new RegExp(flag.replace("-", "\\-")), `${flag} not documented in the Run locally section`); + } +}); + +test("drift: --fresh is documented in dev.mjs --help (HELP_TEXT)", () => { + assert.match(HELP_TEXT, /--fresh/); +}); + +test("drift: pnpm dev / dev:live / dev:full / o11y:dev are all documented in run-and-deploy.md's Run locally section", () => { + const doc = readFileSync(path.join(RUNNER_ROOT, "docs", "run-and-deploy.md"), "utf8"); + const section = extractSection(doc, "## Run locally"); + for (const cmd of ["pnpm dev", "pnpm dev:live", "pnpm dev:full", "pnpm o11y:dev"]) { + assert.ok(section.includes(cmd), `${cmd} not mentioned in the Run locally section`); + } +}); diff --git a/runner/pipeline/example-analytics-ingest.test.mjs b/runner/pipeline/example-analytics-ingest.test.mjs new file mode 100644 index 0000000000..5c25e62d82 --- /dev/null +++ b/runner/pipeline/example-analytics-ingest.test.mjs @@ -0,0 +1,146 @@ +// ADR-0042 — proves the full round trip for `example.*`, not just the +// server-side half `o11y-normalise.test.mjs`'s "Faro example.open" case +// covers. +// +// That existing case feeds `processFaroBody` a fixture whose Faro item +// already carries the raw `hot.metric_kind`/`hot.ref`/`hot.area` keys — it +// never proves the browser actually sends them. The browser's own +// `beforeSend` hook runs the same `scrubTelemetry` allowlist +// (`attrs.ts#ALLOWED_ATTRIBUTE_KEYS`) before the request ever leaves the +// tab, so that allowlist must have an entry for +// `hot.metric_kind`/`hot.ref`/`hot.area` — a browser build sending the +// plain `HotAttrs` bag without it would have every one of +// `kind`/`ref`/`area` silently stripped client-side, long before +// `processFaroBody`'s own internal (re-run) scrub or `readAeOnlyAttrs`'s +// pre-scrub read ever gets a chance. `o11y-normalise.test.mjs`'s +// fixture-only test cannot see that, because it starts downstream of the +// browser. +// +// This file: (1) runs the exact item shape `apps/authoring/src/telemetry/ +// faro.ts#attrsToContext` + Faro's own `pushEvent` would produce through +// `scrubTelemetry` — simulating the browser's `beforeSend` — and asserts the +// three ADR-0042 keys survive; (2) feeds the resulting wire body through the +// real o11y ingest path and asserts zero inbox items and exactly one +// Analytics Engine point with blob17/18/19 filled. The inbox is Loki's only +// feed (§8), so "never reaches the inbox" is the same claim as "never +// reaches Loki." +// Run: node --experimental-strip-types --test pipeline/example-analytics-ingest.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; + +register("./fixtures/o11y-worker-hooks.mjs", import.meta.url); + +const { processFaroBody } = await import("../workers/o11y/src/normalise/faro.ts"); +const { scrubTelemetry, AE_COLUMNS } = await import("../packages/runtime/dist/telemetry/index.js"); + +const ENV = { O11Y_ENV: "production" }; +const SERVICE = { name: "demos-authoring", version: "deadbeef1234", environment: "production" }; + +/** Mirrors `apps/authoring/src/telemetry/faro.ts#attrsToContext`'s mapping + * for the fields this test exercises — a `HotAttrs`-shaped call becomes + * this raw wire `context`, dotted-key-remapped, before `beforeSend` runs. + * If that module's `DOTTED_ATTR_KEY` table changes, this literal has to + * change with it (faro.ts itself cannot be imported under + * `--experimental-strip-types`: it pulls in `@grafana/faro-web-sdk`, a real + * browser package this harness does not resolve — the same constraint + * `sentry.ts`/`demoEventReport.ts` document for their own files). */ +function rawExampleOpenItem() { + return { + type: "event", + payload: { + name: "example.open", + attributes: { + "hot.metric_kind": "docs", + "hot.ref": "guides/accessibility/accessibility/accessibility.md", + "hot.area": "Accessibility", + "hot.framework": "typescript", + "hot.ht_major": "18", + "hot.bucket": "18.1", + "hot.reason": "entry", + }, + }, + meta: { app: { name: "demos-authoring", version: "deadbeef1234" } }, + }; +} + +test("browser scrub (beforeSend) keeps hot.metric_kind/hot.ref/hot.area — this task's own ALLOWED_ATTRIBUTE_KEYS addition", () => { + const scrubbed = scrubTelemetry(rawExampleOpenItem()); + assert.ok(scrubbed, "the item must survive scrubbing at all (not a console-dropped kind)"); + const attrs = scrubbed.payload.attributes; + assert.equal(attrs["hot.metric_kind"], "docs"); + assert.equal(attrs["hot.ref"], "guides/accessibility/accessibility/accessibility.md"); + assert.equal(attrs["hot.area"], "Accessibility"); + // Not asserted here: `hot.bucket`/`hot.reason` survival is a separate + // `ATTR_HOT_BUCKET`/`ATTR_HOT_REASON` addition to this same AE-only + // category (`attrs.ts#AE_ONLY_ATTRIBUTE_KEYS`) — a separate concern owns + // proving those two, this file owns `kind`/`ref`/`area`. + // + // Sanity: an attribute genuinely outside every allowlist category is still + // dropped — this test is not accidentally passing because the allowlist + // has become a no-op. + const withForbidden = rawExampleOpenItem(); + withForbidden.payload.attributes["url.full"] = "https://example.com/secret?token=abc"; + const scrubbedForbidden = scrubTelemetry(withForbidden); + assert.equal(scrubbedForbidden.payload.attributes["url.full"], undefined); +}); + +test("example.open: end to end from a scrubbed browser payload to one AE point, zero inbox items", async () => { + const scrubbed = scrubTelemetry(rawExampleOpenItem()); + // The real Faro transport body shape: one shared `meta` plus + // separate typed arrays, `events` here. + const wireBody = { meta: scrubbed.meta, events: [{ name: scrubbed.payload.name, attributes: scrubbed.payload.attributes }] }; + + const [item] = await processFaroBody(wireBody, ENV, SERVICE, Date.now()); + + // An example.* event + // now gets a hash-only ingestItem (no `record`) so a redelivered batch + // can't double-count this AE point — but it must still never reach the + // inbox/Loki (§6 unchanged): `record` stays absent. + assert.ok(item.ingestItem, "example.* still needs a hash to dedupe on (A-I4 remainder)"); + assert.equal(item.ingestItem.record, undefined, "example.* is never stored (§6) — never reaches the inbox, so never Loki"); + assert.equal(item.invalid, undefined); + assert.equal(item.aePoints.length, 1); + const point = item.aePoints[0]; + assert.equal(point.indexes[0], "example.open"); + + const blobAt = (column) => { + const slot = AE_COLUMNS[column]; + const n = Number(/^blob(\d+)$/.exec(slot)[1]); + return point.blobs[n - 1]; + }; + assert.equal(blobAt("kind"), "docs", "blob17"); + assert.equal(blobAt("ref"), "guides/accessibility/accessibility/accessibility.md", "blob18"); + assert.equal(blobAt("area"), "Accessibility", "blob19"); + assert.equal(blobAt("framework"), "typescript"); + assert.equal(blobAt("ht_major"), "18"); + // `bucket` (blob16) / `reason` (blob9) are T07's own AE-only keys — see the + // comment on the scrub test above for why they are not asserted here. +}); + +test("example.engaged: same taxonomy channel, no reason blob", async () => { + const item = { + type: "event", + payload: { + name: "example.engaged", + attributes: { + "hot.metric_kind": "starter", + "hot.ref": "react", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.bucket": "18.1", + }, + }, + meta: { app: { name: "demos-authoring", version: "deadbeef1234" } }, + }; + const scrubbed = scrubTelemetry(item); + const wireBody = { meta: scrubbed.meta, events: [{ name: scrubbed.payload.name, attributes: scrubbed.payload.attributes }] }; + const [result] = await processFaroBody(wireBody, ENV, SERVICE, Date.now()); + // Hash-only ingestItem, still never stored — see the + // "example.open" test above for the full reasoning. + assert.ok(result.ingestItem); + assert.equal(result.ingestItem.record, undefined); + assert.equal(result.aePoints.length, 1); + assert.equal(result.aePoints[0].indexes[0], "example.engaged"); +}); diff --git a/runner/pipeline/example-analytics-taxonomy.test.mjs b/runner/pipeline/example-analytics-taxonomy.test.mjs new file mode 100644 index 0000000000..c4f3b361e4 --- /dev/null +++ b/runner/pipeline/example-analytics-taxonomy.test.mjs @@ -0,0 +1,181 @@ +// ADR-0042 — pins `apps/authoring/src/exampleAnalytics.ts`'s pure taxonomy +// logic: which `loadWorkspace` lineage maps to which `kind`, and what `ref`/ +// `area`/`framework` a resolved example carries — read from the loaded +// docs-example entry, never from the URL (the task's own Traps). +// +// Run: node --experimental-strip-types --test pipeline/example-analytics-taxonomy.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { + browserCountsSave, + consumeForkMarker, + exampleActionAttrs, + exampleOpenAttrs, + exampleOpenKey, + exampleTaxonomy, + kindOfLineage, +} from "../apps/authoring/src/exampleAnalytics.ts"; + +// ---- kindOfLineage ------------------------------------------------------------- + +test("kindOfLineage: every loadWorkspace lineage prefix", () => { + assert.equal(kindOfLineage("catalog:react"), "starter"); + assert.equal(kindOfLineage("docs:18.1:guides/accessibility/accessibility/react/example1.tsx"), "docs"); + assert.equal(kindOfLineage("import:jsfiddle"), "import"); + assert.equal(kindOfLineage("payload:theme-builder"), "payload"); + // A saved demo's id carries no colon at all (DEV-2859's own redaction rule). + assert.equal(kindOfLineage("dem0-abc123"), "saved"); + assert.equal(kindOfLineage(""), "saved"); +}); + +// ---- exampleTaxonomy: docs ----------------------------------------------------- + +test("exampleTaxonomy(docs): ref is the guide, not docsPath or the URL", () => { + const taxonomy = exampleTaxonomy({ + lineage: "docs:18.1:guides/accessibility/accessibility/react/example1.tsx", + framework: "react", // the app's own catalog framework — must be overridden below + htMajor: "18", + bucket: "18.1", + docs: { + guide: "guides/accessibility/accessibility/accessibility.md", + area: "Accessibility", + framework: "reactts", + }, + }); + assert.deepEqual(taxonomy, { + kind: "docs", + ref: "guides/accessibility/accessibility/accessibility.md", + area: "Accessibility", + framework: "reactts", // read from the loaded entry, not the app's own `framework` + ht_major: "18", + bucket: "18.1", + }); +}); + +test("exampleTaxonomy(docs): falls back to the lineage suffix when no entry is given", () => { + const taxonomy = exampleTaxonomy({ + lineage: "docs:18.1:guides/x/x/react/example1.tsx", + framework: "react", + htMajor: "18", + }); + assert.equal(taxonomy.ref, "18.1:guides/x/x/react/example1.tsx"); + assert.equal(taxonomy.area, undefined); + assert.equal(taxonomy.framework, "react"); +}); + +// ---- exampleTaxonomy: starter/saved/import/payload ----------------------------- + +test("exampleTaxonomy(starter): ref is the framework id, no area", () => { + const taxonomy = exampleTaxonomy({ lineage: "catalog:vue3", framework: "vue3", htMajor: "18", bucket: "18.1" }); + assert.deepEqual(taxonomy, { kind: "starter", ref: "vue3", framework: "vue3", ht_major: "18", bucket: "18.1" }); +}); + +test("exampleTaxonomy(saved): ref is the demo id", () => { + const taxonomy = exampleTaxonomy({ lineage: "dem0-xyz", framework: "react", htMajor: "next" }); + assert.equal(taxonomy.kind, "saved"); + assert.equal(taxonomy.ref, "dem0-xyz"); + assert.equal(taxonomy.area, undefined); + assert.equal(taxonomy.bucket, undefined); +}); + +test("exampleTaxonomy(import/payload): ref is the lineage's own suffix", () => { + assert.equal( + exampleTaxonomy({ lineage: "import:jsfiddle", framework: "react", htMajor: "18" }).ref, + "jsfiddle", + ); + assert.equal( + exampleTaxonomy({ lineage: "payload:theme-builder", framework: "react", htMajor: "18" }).ref, + "theme-builder", + ); +}); + +// ---- exampleOpenAttrs / exampleActionAttrs ------------------------------------- + +test("exampleOpenAttrs: carries reason, exampleActionAttrs does not", () => { + const taxonomy = exampleTaxonomy({ + lineage: "docs:18.1:guides/x/x/react/example1.tsx", + framework: "react", + htMajor: "18", + bucket: "18.1", + docs: { guide: "guides/x/x/x.md", area: "Columns", framework: "react" }, + }); + const open = exampleOpenAttrs(taxonomy, "deep-link"); + assert.equal(open.reason, "deep-link"); + assert.equal(open.kind, "docs"); + assert.equal(open.ref, "guides/x/x/x.md"); + assert.equal(open.area, "Columns"); + assert.equal(open.bucket, "18.1"); + + const action = exampleActionAttrs(taxonomy); + assert.equal("reason" in action, false, "example.engaged/forked/saved/shared/downloaded carry no reason"); + assert.equal(action.kind, "docs"); +}); + +test("exampleActionAttrs: omits area/bucket when the taxonomy has none, never sends an empty string", () => { + const taxonomy = exampleTaxonomy({ lineage: "catalog:react", framework: "react", htMajor: "18" }); + const attrs = exampleActionAttrs(taxonomy); + assert.equal("area" in attrs, false); + assert.equal("bucket" in attrs, false); +}); + +// ---- exampleOpenKey (dedup) ----------------------------------------------------- + +test("exampleOpenKey: same lineage + same version is the same key; a version change is a different key", () => { + const a = exampleOpenKey("docs:18.1:guides/x/x/react/example1.tsx", "18.1.2"); + const b = exampleOpenKey("docs:18.1:guides/x/x/react/example1.tsx", "18.1.2"); + const c = exampleOpenKey("docs:18.1:guides/x/x/react/example1.tsx", "18.1.3"); + assert.equal(a, b); + assert.notEqual(a, c); +}); + +// ---- consumeForkMarker: entry=fork ------------------------------------------ +// +// onFork navigates with a full `location.href` reload (App.tsx's own +// established pattern for every route change, never client-side routing), +// which destroys every in-memory flag, so the one-shot signal must survive +// in the URL itself, stripped on read. Never localStorage/sessionStorage +// (the contract keeps this path off browser storage). + +test("consumeForkMarker: detects the marker and strips it down to an empty search", () => { + const { isFork, search } = consumeForkMarker("?fork=1"); + assert.equal(isFork, true); + assert.equal(search, ""); +}); + +test("consumeForkMarker: strips only the marker, keeps other params", () => { + const { isFork, search } = consumeForkMarker("?v=18.0.0&fork=1"); + assert.equal(isFork, true); + assert.equal(search, "?v=18.0.0"); +}); + +test("consumeForkMarker: no marker present -> isFork false, search returned unchanged", () => { + const { isFork, search } = consumeForkMarker("?v=18.0.0"); + assert.equal(isFork, false); + assert.equal(search, "?v=18.0.0"); +}); + +test("consumeForkMarker: empty search -> isFork false, still an empty search", () => { + const { isFork, search } = consumeForkMarker(""); + assert.equal(isFork, false); + assert.equal(search, ""); +}); + +test("consumeForkMarker: one-shot -- reading the stripped search a second time no longer counts as fork", () => { + const first = consumeForkMarker("?fork=1"); + const second = consumeForkMarker(first.search); + assert.equal(first.isFork, true); + assert.equal(second.isFork, false, "a manual reload of the same (already-stripped) URL must not re-count as a fork"); +}); + +// ---- browserCountsSave --------------------------------------------------------- + +test("browserCountsSave: a Save response carrying the API's exampleSaved marker is never counted by the browser", () => { + assert.equal(browserCountsSave({ ok: true, htVersion: "18.0.0", exampleSaved: true }), false); + assert.equal(browserCountsSave({ ok: true, htVersion: "18.0.0", exampleSaved: false }), false); +}); + +test("browserCountsSave: a Save response without the marker (an API that does not count saves) is counted once by the browser", () => { + assert.equal(browserCountsSave({ ok: true, htVersion: "18.0.0" }), true); + assert.equal(browserCountsSave(null), true); +}); diff --git a/runner/pipeline/example-daily-rollup.test.mjs b/runner/pipeline/example-daily-rollup.test.mjs new file mode 100644 index 0000000000..30a00ca770 --- /dev/null +++ b/runner/pipeline/example-daily-rollup.test.mjs @@ -0,0 +1,322 @@ +// ADR-0042 §5 — the nightly `example_daily` rollup (`workers/api/src/reconcile.ts`). +// +// `pivotExampleDaily` and `previousUtcDay` are pure — tested directly, no I/O. +// +// `writeExampleDaily` is tested against a real SQLite database created from +// the real migration files (`workers/api/migrations/0008_example_daily.sql`, +// then `0009_example_daily_downloaded.sql`), via Node's built-in `node:sqlite` +// — not a hand-rolled regex fake of `env.DB`, so "running the rollup twice +// for one day yields identical rows" is a claim about the actual +// `PRIMARY KEY (day, kind, ref, framework, ht_major)` constraint. The +// dedicated "0009" test section applies 0008 alone, writes a row, then +// applies 0009 — proving the ADD COLUMN is additive against data that +// predates it, the real production ordering. +// +// `queryExampleEventTotals`'s live AE/ClickHouse HTTP read is not +// exercised here (no Analytics Engine credentials, no live ClickHouse in +// this run). Its production pre-flight config guard (a missing +// AE_SQL_TOKEN/CF_ACCOUNT_ID throws before any `fetch` happens) is +// exercised below, since it needs no credential or network access at all. +// Run: node --experimental-strip-types --test pipeline/example-daily-rollup.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { DatabaseSync } from "node:sqlite"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { register } from "node:module"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); +const { pivotExampleDaily, previousUtcDay, writeExampleDaily, queryExampleEventTotals, rollupExampleDaily } = await import( + "../workers/api/src/reconcile.ts" +); +const { captures } = await import("./fixtures/sentry-cloudflare-stub.mjs"); + +const MIGRATION = readFileSync( + fileURLToPath(new URL("../workers/api/migrations/0008_example_daily.sql", import.meta.url)), + "utf8", +); +const MIGRATION_0009 = readFileSync( + fileURLToPath(new URL("../workers/api/migrations/0009_example_daily_downloaded.sql", import.meta.url)), + "utf8", +); + +/** A `node:sqlite`-backed fake of the two `env.DB` methods `writeExampleDaily` + * uses (`prepare().bind()` returning something `batch` can run) — enough + * surface for this file, not a general D1 fake. */ +function fakeD1(db) { + return { + prepare(sql) { + return { + bind(...args) { + return { + async run() { + db.prepare(sql).run(...args); + return { success: true }; + }, + }; + }, + }; + }, + async batch(statements) { + const results = []; + for (const stmt of statements) results.push(await stmt.run()); + return results; + }, + }; +} + +// Every test in this file runs against 0008 THEN 0009 applied in sequence — +// the real migration order production runs, not a single hand-merged schema +// — so a bug in 0009's ADD COLUMN (wrong type, wrong default, wrong table) +// would show up here exactly as it would against a real D1. +function freshDb() { + const db = new DatabaseSync(":memory:"); + db.exec(MIGRATION); + db.exec(MIGRATION_0009); + return db; +} + +/** Plain objects — `node:sqlite`'s `.all()` returns null-prototype rows, + * which `assert.deepEqual` treats as unequal to a literal object even when + * every field matches. */ +function allRows(db) { + return db + .prepare("SELECT * FROM example_daily ORDER BY kind, ref, framework, ht_major") + .all() + .map((row) => ({ ...row })); +} + +// ---- pivotExampleDaily (pure) --------------------------------------------------- + +test("pivotExampleDaily: one row per (kind, ref, area, framework, ht_major), one column per metric", () => { + const rows = pivotExampleDaily("2026-09-22", [ + { metric: "example.open", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", total: 12 }, + { metric: "example.engaged", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", total: 5 }, + { metric: "example.saved", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", total: 1 }, + // A distinct taxonomy tuple (different framework) must not merge with the one above. + { metric: "example.open", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "vue3", ht_major: "18", total: 3 }, + ]); + assert.equal(rows.length, 2); + const react = rows.find((r) => r.framework === "react"); + assert.deepEqual(react, { + day: "2026-09-22", + kind: "docs", + ref: "guides/x/x.md", + area: "Columns", + framework: "react", + ht_major: "18", + opens: 12, + engaged: 5, + forked: 0, + saved: 1, + shared: 0, + downloaded: 0, + }); + const vue = rows.find((r) => r.framework === "vue3"); + assert.equal(vue.opens, 3); + assert.equal(vue.engaged, 0); +}); + +test("pivotExampleDaily: rounds a fractional (sampled) AE total to an integer count", () => { + const [row] = pivotExampleDaily("2026-09-22", [ + { metric: "example.open", kind: "starter", ref: "react", area: "", framework: "react", ht_major: "18", total: 7.6 }, + ]); + assert.equal(row.opens, 8); +}); + +test("pivotExampleDaily: an unrecognised index1 value is ignored, not thrown on", () => { + const rows = pivotExampleDaily("2026-09-22", [ + { metric: "o11y.ingest", kind: "docs", ref: "x", area: "", framework: "react", ht_major: "18", total: 99 }, + ]); + assert.equal(rows.length, 0); +}); + +// ADR-0042 §2 names `example.downloaded` as one of the six `example.*` +// metrics; this pins that it is now pivoted into its own `downloaded` +// column (0009_example_daily_downloaded.sql), not silently dropped the way +// an unrecognised metric is above. +test("pivotExampleDaily: example.downloaded is pivoted into its own `downloaded` column", () => { + const [row] = pivotExampleDaily("2026-09-22", [ + { metric: "example.downloaded", kind: "starter", ref: "react", area: "", framework: "react", ht_major: "18", total: 6 }, + ]); + assert.equal(row.downloaded, 6); + assert.equal(row.opens, 0); +}); + +// ---- previousUtcDay (pure) ------------------------------------------------------- + +test("previousUtcDay: the day before `now`, UTC, half-open [start, end)", () => { + const { day, start, end } = previousUtcDay(new Date("2026-09-23T11:38:00Z")); + assert.equal(day, "2026-09-22"); + assert.equal(start, "2026-09-22 00:00:00"); + assert.equal(end, "2026-09-23 00:00:00"); +}); + +test("previousUtcDay: a `now` right at UTC midnight still resolves the FULL prior day, not the current one", () => { + const { day, start, end } = previousUtcDay(new Date("2026-09-23T00:00:00Z")); + assert.equal(day, "2026-09-22"); + assert.equal(start, "2026-09-22 00:00:00"); + assert.equal(end, "2026-09-23 00:00:00"); +}); + +// ---- writeExampleDaily against a real SQLite DB, the real migration ------------- + +test("writeExampleDaily: writes rows honouring the real PRIMARY KEY", async () => { + const db = freshDb(); + const env = { DB: fakeD1(db) }; + await writeExampleDaily(env, "2026-09-22", [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 10, engaged: 3, forked: 0, saved: 1, shared: 0, downloaded: 0 }, + { day: "2026-09-22", kind: "starter", ref: "vue3", area: "", framework: "vue3", ht_major: "18", opens: 4, engaged: 0, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + ]); + const rows = allRows(db); + assert.equal(rows.length, 2); + assert.equal(rows.find((r) => r.kind === "docs").opens, 10); + assert.equal(rows.find((r) => r.kind === "starter").opens, 4); +}); + +test("writeExampleDaily: running it TWICE for the same day yields identical rows (idempotency)", async () => { + const db = freshDb(); + const env = { DB: fakeD1(db) }; + const rows = [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 10, engaged: 3, forked: 0, saved: 1, shared: 0, downloaded: 0 }, + ]; + await writeExampleDaily(env, "2026-09-22", rows); + await writeExampleDaily(env, "2026-09-22", rows); + assert.deepEqual(allRows(db), [{ ...rows[0] }]); +}); + +test("writeExampleDaily: `downloaded` round-trips through the real column (0009), and stays identical on a re-run", async () => { + const db = freshDb(); + const env = { DB: fakeD1(db) }; + const rows = [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 10, engaged: 3, forked: 0, saved: 1, shared: 0, downloaded: 7 }, + ]; + await writeExampleDaily(env, "2026-09-22", rows); + assert.equal(allRows(db)[0].downloaded, 7); + await writeExampleDaily(env, "2026-09-22", rows); + assert.deepEqual(allRows(db), [{ ...rows[0] }]); +}); + +test("writeExampleDaily: a group that disappears on a re-run is REMOVED, not left stale (why a bare INSERT OR REPLACE is not enough)", async () => { + const db = freshDb(); + const env = { DB: fakeD1(db) }; + await writeExampleDaily(env, "2026-09-22", [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 10, engaged: 0, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + { day: "2026-09-22", kind: "starter", ref: "vue3", area: "", framework: "vue3", ht_major: "18", opens: 4, engaged: 0, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + ]); + // Re-run: the vue3 starter had zero example.* events this time, so the + // pivot never produces a row for it at all. + await writeExampleDaily(env, "2026-09-22", [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 11, engaged: 1, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + ]); + const rows = allRows(db); + assert.equal(rows.length, 1, "the vue3 row from the first run must be gone"); + assert.equal(rows[0].kind, "docs"); + assert.equal(rows[0].opens, 11); +}); + +// ---- a misconfigured production AE SQL read must not silently wipe the day - + +test("queryExampleEventTotals: production with no AE_SQL_TOKEN/CF_ACCOUNT_ID THROWS, never returns []", async () => { + const env = { PREVIEW_HOST: "demos.handsontable.com" }; // production, both secrets absent + await assert.rejects( + () => queryExampleEventTotals(env, "2026-09-22 00:00:00", "2026-09-23 00:00:00"), + /AE_SQL_TOKEN|CF_ACCOUNT_ID/, + ); +}); + +test("queryExampleEventTotals: production, a 200 response with no data array THROWS, never degrades to []", async () => { + // A response shape change or a truncated body must not read as "zero + // events today" either — same rule as the missing-credential case + // above, one step further down the same function. + const realFetch = globalThis.fetch; + globalThis.fetch = async () => new Response(JSON.stringify({ meta: [], rows: 0 }), { status: 200 }); + try { + const env = { PREVIEW_HOST: "demos.handsontable.com", AE_SQL_TOKEN: "tok", CF_ACCOUNT_ID: "acct" }; + await assert.rejects( + () => queryExampleEventTotals(env, "2026-09-22 00:00:00", "2026-09-23 00:00:00"), + /no "data" array/, + ); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("rollupExampleDaily: a misconfigured production read is refused loudly and never deletes the day's rows", async () => { + const db = freshDb(); + // `rollupExampleDaily` computes its own `previousUtcDay()` internally, from + // the real clock — seed the row under THAT day, not a hardcoded literal, + // or this test would prove nothing on any date but the one it was written + // on (the DELETE would target a different day than the seeded row, so + // "the row survives" would pass whether or not the fix is present). + const { day } = previousUtcDay(); + await writeExampleDaily({ DB: fakeD1(db) }, day, [ + { day, kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 10, engaged: 3, forked: 0, saved: 1, shared: 0, downloaded: 0 }, + ]); + let batchCalls = 0; + const spyD1 = { ...fakeD1(db), batch: (...args) => { batchCalls += 1; return fakeD1(db).batch(...args); } }; + const capturesBefore = captures.length; + + const env = { PREVIEW_HOST: "demos.handsontable.com", DB: spyD1 }; // production, no AE_SQL_TOKEN/CF_ACCOUNT_ID + const result = await rollupExampleDaily(env); + + assert.equal(batchCalls, 0, "writeExampleDaily's DELETE must never run when the read was refused"); + assert.equal(result.rows, 0); + assert.equal(allRows(db).length, 1, "the prior run's row for the day must survive untouched"); + assert.equal(allRows(db)[0].opens, 10); + + const newCaptures = captures.slice(capturesBefore); + assert.equal(newCaptures.length, 1, "the refusal must be reported loudly (Sentry)"); + assert.equal(newCaptures[0].kind, "exception"); + assert.deepEqual(newCaptures[0].context, { tags: { context: "example-daily-rollup" } }); +}); + +test("writeExampleDaily: never touches a DIFFERENT day's rows", async () => { + const db = freshDb(); + const env = { DB: fakeD1(db) }; + await writeExampleDaily(env, "2026-09-21", [ + { day: "2026-09-21", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 5, engaged: 0, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + ]); + await writeExampleDaily(env, "2026-09-22", [ + { day: "2026-09-22", kind: "docs", ref: "guides/x/x.md", area: "Columns", framework: "react", ht_major: "18", opens: 9, engaged: 0, forked: 0, saved: 0, shared: 0, downloaded: 0 }, + ]); + const rows = allRows(db); + assert.equal(rows.length, 2); + assert.equal(rows.find((r) => r.day === "2026-09-21").opens, 5); + assert.equal(rows.find((r) => r.day === "2026-09-22").opens, 9); +}); + +// ---- 0009_example_daily_downloaded.sql: additive, safe on production data ---- + +test("0009: applying it AFTER 0008 against a row already written adds `downloaded` defaulted to 0, every other column untouched", () => { + // Deliberately does NOT go through `freshDb()` (which already applies both + // migrations) — this test's whole point is the ORDER production runs in: + // 0008 ships first (already applied against real data), a row is written + // under the five-counter schema, and ONLY THEN does 0009 land. This is the + // "safe on production data" claim from the migration file's own header, + // proven against a real SQLite schema change, not asserted in prose. + const db = new DatabaseSync(":memory:"); + db.exec(MIGRATION); // 0008 only + db.exec( + `INSERT INTO example_daily (day, kind, ref, area, framework, ht_major, opens, engaged, forked, saved, shared) + VALUES ('2026-09-22', 'docs', 'guides/x/x.md', 'Columns', 'react', '18', 10, 3, 0, 1, 0)`, + ); + // Pre-migration sanity: the column genuinely does not exist yet. + assert.throws(() => db.prepare("SELECT downloaded FROM example_daily").get(), /no such column/); + + db.exec(MIGRATION_0009); // 0009, applied after real data already exists + + const row = { ...db.prepare("SELECT * FROM example_daily").get() }; + assert.equal(row.downloaded, 0, "a pre-existing row backfills to downloaded = 0, never null or an error"); + assert.equal(row.opens, 10, "every pre-existing column is untouched by the ADD COLUMN"); + assert.equal(row.engaged, 3); + assert.equal(row.forked, 0); + assert.equal(row.saved, 1); + assert.equal(row.shared, 0); + + // A fresh write after 0009 lands can now populate a real downloaded count + // on the SAME row, exactly like any other counter. + db.exec("UPDATE example_daily SET downloaded = 4 WHERE day = '2026-09-22'"); + assert.equal(db.prepare("SELECT downloaded FROM example_daily").get().downloaded, 4); +}); diff --git a/runner/pipeline/example-saved-point.test.mjs b/runner/pipeline/example-saved-point.test.mjs new file mode 100644 index 0000000000..7c351f44e3 --- /dev/null +++ b/runner/pipeline/example-saved-point.test.mjs @@ -0,0 +1,185 @@ +// `example.saved` (contract §5, ADR-0042 §2) is written by the API worker when +// an editor Save's rebuild succeeds, with the values the browser's saved-demo +// taxonomy produces. Driven through the real router with the shared fakes. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test, { after } from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { AUTHOR, demoRow, makeEnv } from "./fixtures/worker-harness.mjs"; +import { exampleActionAttrs, exampleTaxonomy } from "../apps/authoring/src/exampleAnalytics.ts"; +import { toAePoint } from "../packages/runtime/dist/telemetry/index.js"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const REAL_FETCH = globalThis.fetch; +globalThis.fetch = async (input, init) => { + const url = typeof input === "string" ? input : input.url; + if (url.startsWith("https://login.invalid") && init?.headers?.Authorization === "Bearer test-token") { + return Response.json({ email: AUTHOR, sub: "u1" }); + } + throw new Error(`unexpected network fetch in example-saved-point.test.mjs: ${url}`); +}; +after(() => { + globalThis.fetch = REAL_FETCH; +}); + +const DEMO_ID = "abc123"; +const FILES = { + "/package.json": JSON.stringify({ name: "demo", dependencies: { handsontable: "16.0.2" } }), + "/index.js": "console.log(1)", +}; + +/** Points land in memory through the production `bindingSink`; `waitUntil` + * promises are kept so a test can wait for work scheduled past the response. */ +function setup(rows = [demoRow({ id: DEMO_ID, framework: "react", ht_version: "16.0.2" })]) { + const { env } = makeEnv(rows); + const points = []; + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + env.SERVICE_VERSION = "api-sha"; + const pending = []; + const ctx = { waitUntil: (p) => pending.push(Promise.resolve(p)), passThroughOnException() {} }; + const saved = async () => { + await Promise.all(pending); + return points.filter((p) => p.indexes[0] === "example.saved"); + }; + return { env, ctx, saved, pending, points }; +} + +const patch = (body, { auth = true } = {}) => + new Request(`https://demos.handsontable.com/api/demos/${DEMO_ID}`, { + method: "PATCH", + headers: { "Content-Type": "application/json", ...(auth ? { Authorization: "Bearer test-token" } : {}) }, + body: JSON.stringify(body), + }); + +test("an editor Save writes exactly one example.saved point carrying the browser's saved-demo taxonomy", async () => { + const { env, ctx, saved } = setup(); + const res = await worker.fetch(patch({ files: FILES, htVersion: "16.0.2", exampleHtMajor: "16" }), env, ctx); + assert.equal(res.status, 200); + const points = await saved(); + assert.equal(points.length, 1, `expected exactly 1 example.saved point, got ${points.length}`); + + // The row the browser path produced for this save: `App.tsx` opens a saved + // demo with its id as the lineage, and its facade attrs become the AE point. + const browser = toAePoint( + "example.saved", + { count: 1 }, + { + service_name: "demos-authoring", + service_version: "authoring-sha", + environment: "production", + ...exampleActionAttrs(exampleTaxonomy({ lineage: DEMO_ID, framework: "react", htMajor: "16" })), + }, + ); + const [point] = points; + assert.deepEqual(point.indexes, browser.indexes); + assert.deepEqual(point.blobs.slice(2), browser.blobs.slice(2), "blob3..blob20 match the browser's row"); + assert.deepEqual(point.doubles, browser.doubles); + assert.equal(point.blobs[0], "demos-api"); + assert.equal(point.blobs[1], "api-sha"); + assert.equal(point.blobs[16], "saved"); + assert.equal(point.blobs[17], DEMO_ID); +}); + +test("the point's ht_major is the major the editor opened the demo at, not the version the Save pins", async () => { + const { env, ctx, saved } = setup(); + const res = await worker.fetch(patch({ files: FILES, htVersion: "16.0.2", exampleHtMajor: "15" }), env, ctx); + assert.equal(res.status, 200); + const points = await saved(); + assert.equal(points.length, 1); + assert.equal(points[0].blobs[6], "15"); +}); + +test("an unauthenticated Save writes no example.saved point", async () => { + const { env, ctx, saved } = setup(); + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor: "16" }, { auth: false }), env, ctx); + assert.equal(res.status, 401); + assert.equal((await saved()).length, 0); +}); + +test("a Save of someone else's demo writes no example.saved point", async () => { + const { env, ctx, saved } = setup([demoRow({ id: DEMO_ID, created_by: "other@handsontable.com" })]); + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor: "16" }), env, ctx); + assert.equal(res.status, 403); + assert.equal((await saved()).length, 0); +}); + +test("a Save refused while a build is running writes no example.saved point", async () => { + const { env, ctx, saved } = setup([ + demoRow({ id: DEMO_ID, build_status: "building", updated_at: new Date().toISOString() }), + ]); + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor: "16" }), env, ctx); + assert.equal(res.status, 409); + assert.equal((await saved()).length, 0); +}); + +test("a Save whose rebuild fails writes no example.saved point", async () => { + const { env, ctx, saved } = setup(); + env.ARTIFACTS.put = async () => { + throw new Error("R2 unavailable"); + }; + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor: "16" }), env, ctx); + assert.ok(res.status >= 500, `expected a 5xx, got ${res.status}`); + assert.equal((await saved()).length, 0); +}); + +test("a metadata-only PATCH (the Edit info dialog) writes no example.saved point", async () => { + const { env, ctx, saved } = setup(); + const res = await worker.fetch(patch({ title: "Renamed", exampleHtMajor: "16" }), env, ctx); + assert.equal(res.status, 200); + assert.equal((await saved()).length, 0); +}); + +test("a Save without a valid exampleHtMajor (closed telemetry gate, a caller that omits it) writes no point", async () => { + for (const exampleHtMajor of [undefined, "", "20", 16, "latest"]) { + const { env, ctx, saved } = setup(); + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor }), env, ctx); + assert.equal(res.status, 200, `status for ${JSON.stringify(exampleHtMajor)}`); + assert.equal((await saved()).length, 0, `points for ${JSON.stringify(exampleHtMajor)}`); + } +}); + +test("every rebuild response carries the exampleSaved marker, true only when the point is written", async () => { + for (const [exampleHtMajor, expected] of [["16", true], [undefined, false], ["20", false]]) { + const { env, ctx } = setup(); + const res = await worker.fetch(patch({ files: FILES, exampleHtMajor }), env, ctx); + assert.equal(res.status, 200); + const body = await res.json(); + assert.equal(body.exampleSaved, expected, `exampleSaved for ${JSON.stringify(exampleHtMajor)}`); + } +}); + +test("a metadata-only PATCH carries no exampleSaved marker", async () => { + const { env, ctx } = setup(); + const res = await worker.fetch(patch({ title: "Renamed", exampleHtMajor: "16" }), env, ctx); + assert.equal(res.status, 200); + assert.equal("exampleSaved" in (await res.json()), false); +}); + +test("the rebuild and its point are handed to waitUntil, so a client disconnect cannot cancel them", async () => { + // The oracle is the D1 write and the point completing through `waitUntil` + // alone: the rebuild is held until the handler has registered its work. + const { env, ctx, pending, points } = setup(); + let release; + const gate = new Promise((r) => { release = r; }); + const put = env.ARTIFACTS.put.bind(env.ARTIFACTS); + env.ARTIFACTS.put = async (...args) => { await gate; return put(...args); }; + const writes = []; + const prepare = env.DB.prepare.bind(env.DB); + env.DB.prepare = (sql) => { if (/UPDATE demos SET ht_version=/.test(sql)) writes.push(sql); return prepare(sql); }; + + const response = worker.fetch(patch({ files: FILES, exampleHtMajor: "16" }), env, ctx); + await new Promise((r) => setTimeout(r, 50)); + assert.ok(pending.length >= 1, "the save must be registered with waitUntil before it settles"); + release(); + await Promise.all(pending); + assert.equal(writes.length, 1, "the waitUntil promise covers the D1 update"); + assert.equal(points.filter((p) => p.indexes[0] === "example.saved").length, 1, "and the point"); + assert.equal((await response).status, 200); +}); diff --git a/runner/pipeline/faro-config.test.mjs b/runner/pipeline/faro-config.test.mjs new file mode 100644 index 0000000000..e1f7133c4b --- /dev/null +++ b/runner/pipeline/faro-config.test.mjs @@ -0,0 +1,120 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { resolveTelemetryEnabled, telemetryEnvironment } from "../apps/authoring/src/telemetry/gate.ts"; + +// Contract §10 "Local telemetry gate in the browser" + ADR §E.4's last +// bullet. `telemetry/gate.ts` is import-free for the same reason as +// `reportingGate.ts` — this file imports it directly under +// `--experimental-strip-types`. +// +// The actual Faro wiring (`telemetry/faro.ts`) cannot be tested here: it +// pulls in `@grafana/faro-web-sdk` and reads `import.meta.env`, so +// `node --test` cannot import it. This file pins the decision `faro.ts` +// delegates to `gate.ts`, and the settings it takes from `faroConfig.ts`. + +test("production leg: reuses resolveReporting's decision verbatim, regardless of the local flag/host", () => { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: true, localFlag: undefined, hostname: undefined }), + true, + ); + // Even a WRONG local flag/host does not close a production-open gate — the + // production leg is unconditional once `productionReportingEnabled` is true. + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: true, localFlag: "0", hostname: "evil.test" }), + true, + ); +}); + +test("production closed + no local flag: closed", () => { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: false, localFlag: undefined, hostname: "localhost" }), + false, + ); +}); + +test("local leg: opens on localhost with the exact flag '1'", () => { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: false, localFlag: "1", hostname: "localhost" }), + true, + ); +}); + +test("local leg: opens on 127.0.0.1 too", () => { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: false, localFlag: "1", hostname: "127.0.0.1" }), + true, + ); +}); + +test("local leg: any other flag value stays closed (only the literal '1' opens it)", () => { + for (const localFlag of [undefined, "", "true", "0", "yes"]) { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: false, localFlag, hostname: "localhost" }), + false, + `localFlag=${JSON.stringify(localFlag)}`, + ); + } +}); + +test("local leg: any other host stays closed, even with the flag set", () => { + for (const hostname of [undefined, "demos.handsontable.com", "example.com", "0.0.0.0"]) { + assert.equal( + resolveTelemetryEnabled({ productionReportingEnabled: false, localFlag: "1", hostname }), + false, + `hostname=${JSON.stringify(hostname)}`, + ); + } +}); + +// The contract's explicit negative: neither `import.meta.env.DEV` nor +// `navigator.webdriver` are inputs to this function at all — Playwright serves +// a production `vite preview` build under automation (contract §10), so a +// DEV/webdriver check would make `e2e/telemetry-faro.spec.ts` unable to ever +// see Faro fire against its own built dist. This is a structural guard: the +// function's parameter list itself has no such field, so there is nothing a +// future edit could "helpfully" wire up without changing the signature this +// test imports. +test("the gate takes no DEV/webdriver input — only productionReportingEnabled, localFlag, hostname", () => { + assert.deepEqual(resolveTelemetryEnabled.length, 1); // one destructured object param +}); + +test("telemetryEnvironment: production when the production leg opened the gate", () => { + assert.equal(telemetryEnvironment(true), "production"); +}); + +test("telemetryEnvironment: local otherwise", () => { + assert.equal(telemetryEnvironment(false), "local"); +}); + +// ---- batching and delivery (`telemetry/faroConfig.ts`) --------------------------- + +const { FARO_BATCHING, FARO_RETRY, FARO_BUFFER_SIZE } = await import("../apps/authoring/src/telemetry/faroConfig.ts"); +// The 429's Retry-After is the limiter window (`o11y-routes.test.mjs` pins the two equal). +const wranglerText = readFileSync(fileURLToPath(new URL("../workers/o11y/wrangler.jsonc", import.meta.url)), "utf8"); +const RATE_LIMIT_PERIOD_SECONDS = Number(/"ratelimits"[\s\S]*?"period":\s*(\d+)/.exec(wranglerText)?.[1]); + +test("batching: one flush per 5 s, 50 items per batch", () => { + assert.deepEqual(FARO_BATCHING, { enabled: true, sendTimeout: 5_000, itemLimit: 50 }); +}); + +test("retry: a 429's Retry-After (the limiter window) plus Faro's 20 % jitter fits under maxBackoffMs", () => { + assert.equal(FARO_RETRY.maxBackoffMs, 75_000); + assert.equal(RATE_LIMIT_PERIOD_SECONDS, 60); + // Faro drops a batch whose Retry-After exceeds maxBackoffMs, and caps the jittered wait at it. + assert.ok(FARO_RETRY.maxBackoffMs >= RATE_LIMIT_PERIOD_SECONDS * 1000 * 1.2); + assert.ok(FARO_RETRY.maxAttempts >= 2, "a 429'd batch gets at least one retry"); +}); + +test("the delivery queue stays bounded", () => { + assert.equal(FARO_BUFFER_SIZE, 30); +}); + +test("faro.ts hands these settings to initializeFaro and its FetchTransport", () => { + const source = readFileSync(fileURLToPath(new URL("../apps/authoring/src/telemetry/faro.ts", import.meta.url)), "utf8"); + assert.match(source, /batching:\s*\{\s*\.\.\.FARO_BATCHING\s*\}/); + assert.match(source, /new FetchTransport\(\{[^}]*bufferSize:\s*FARO_BUFFER_SIZE[^}]*retry:\s*\{\s*\.\.\.FARO_RETRY\s*\}/); + // With `transports` given, a top-level `url` would make Faro log a config error. + assert.doesNotMatch(source, /initializeFaro\(\{\s*url:/); +}); diff --git a/runner/pipeline/fixtures/cloudflare-containers-stub.mjs b/runner/pipeline/fixtures/cloudflare-containers-stub.mjs new file mode 100644 index 0000000000..2120576fa7 --- /dev/null +++ b/runner/pipeline/fixtures/cloudflare-containers-stub.mjs @@ -0,0 +1,196 @@ +// Structural stand-in for `@cloudflare/containers` under plain `node --test` +// (the real package imports `cloudflare:workers` at load time, which only +// exists inside workerd — the same reason worker-hooks.mjs stubs +// `@cloudflare/sandbox`). Provides just the `Container` surface +// `workers/o11y/src/box.ts` actually uses: a constructor storing +// `ctx`/`env`, `getState()`, `start()`, `containerFetch()`, and the +// lifecycle hooks, all routed through the mutable `hooks` registry below so +// a test can install its own spy/behaviour per call without a mocking +// library. Reset `hooks` to `defaultHooks()` in a `beforeEach`/`afterEach` — +// this module is shared (imported once) across every test in a file. + +export function defaultHooks() { + return { + // (self, startOptions, waitOptions) -> void + async start(self, _startOptions, _waitOptions) { + self._state = { status: "running", lastChange: Date.now() }; + }, + // (self, port|undefined, cancellationOptions|undefined, startOptions|undefined) -> void + async startAndWaitForPorts(self) { + self._state = { status: "healthy", lastChange: Date.now() }; + }, + // (self, requestOrUrl, portOrInit, portParam) -> Response + async containerFetch(_self, _requestOrUrl, _portOrInit, _portParam) { + return new Response("stub: containerFetch not configured for this call", { status: 500 }); + }, + // (self, signal) -> void + // + // The real `Container.prototype.stop` only + // signals the process (SIGTERM) and awaits `syncPendingStoppedEvents` — + // it never sets `status: "stopping"` (that status exists in the + // library's types and its `ContainerState.setStopping()` method exists, + // but nothing in the package ever calls it). `getState()` keeps + // reporting whatever it reported before `stop()` was called + // (`"running"`/`"healthy"`) until the container process actually exits + // and the real `onStop` path (or, here, a test explicitly setting + // `self._state = { status: "stopped", ... }`) reflects that. This used + // to fabricate `"stopping"`, which is exactly why `box.ts`'s own + // a `state.status === "stopping"` branch would have a test that + // passed against a state the real library never produces. + async stop(_self, _signal) {}, + }; +} + +export const hooks = defaultHooks(); + +/** `@cloudflare/containers` `dist/lib/helpers.js#parseTimeExpression`: + * seconds, from a number or an `"<n>s|m|h"` string. */ +function parseTimeExpression(expr) { + if (typeof expr === "number") return expr; + const match = /^(\d+)([smh])$/.exec(expr); + if (!match) throw new Error(`invalid time expression ${expr}`); + const value = parseInt(match[1], 10); + return match[2] === "s" ? value : match[2] === "m" ? value * 60 : value * 3600; +} + +export class Container { + constructor(ctx, env, options) { + this.ctx = ctx; + this.env = env; + this.options = options; + // See box.ts's `containerFetch` override: a real + // `DurableObjectState.container` (public, unlike the base library's own + // private `this.container` field) is what `box.ts` now ALSO checks + // before letting the base class's own auto-start path run, because a + // persisted `getState()` status can lag the real container by a few + // minutes after a host loss (its own doc comment). Every test in this + // repo simulates container lifecycle purely by assigning `_state` + // (directly, or via a `hooks.start`/`hooks.stop` override) — the setter + // below keeps `ctx.container.running` in lockstep with whatever `_state` + // a test sets, so every EXISTING test (where the two never actually + // diverge) keeps passing unchanged. A test that wants to model the + // desync itself (a stale "healthy" `_state` after the real process + // already exited) sets `box.ctx.container.running = false` AFTER + // setting `_state`, deliberately breaking the lockstep for that one + // assertion. + if (!this.ctx.container) this.ctx.container = { running: false }; + this._state = { status: "stopped", lastChange: Date.now() }; + } + + get _state() { + return this.__state; + } + + set _state(value) { + this.__state = value; + this.ctx.container.running = value?.status === "running" || value?.status === "healthy"; + } + + async getState() { + return { ...this._state }; + } + + async start(startOptions, waitOptions) { + return hooks.start(this, startOptions, waitOptions); + } + + async startAndWaitForPorts(portsOrArgs, cancellationOptions, startOptions) { + return hooks.startAndWaitForPorts(this, portsOrArgs, cancellationOptions, startOptions); + } + + // ---- in-flight accounting and the idle clock ------------------------------ + // + // Mirrors `@cloudflare/containers@0.3.7` `dist/lib/container.js`, because + // this is what decides whether the `sleepAfter` idle stop can ever fire: + // - `containerFetch` does `inflightRequests++` (:887) before proxying; + // - a response WITH a body is returned as `new Response(readable, res)` + // after `res.body.pipeTo(writable).finally(() => decrementInflight())` + // through an `IdentityTransformStream` (:955-960), so the count drops + // only once the CALLER consumes or cancels that body; + // - a body-less response, or a throw, decrements at once (:962, :966); + // - `isActivityExpired()` (:1687-1692) returns false and renews the clock + // while `inflightRequests > 0`; the base `alarm()` loop calls it and + // `onActivityExpired()` → `stop()` only when it returns true (:1566). + // Before this, `renewActivityTimeout()` was a no-op and nothing counted, + // so a caller that never released a response body (box.ts `isReady()` + // did exactly that on every probe) looked idle here and pinned the real + // container awake until the 4-hour cap. Deviations: a hook that THROWS + // still throws (the real library turns it into a 500 response), so + // existing tests that inject a throw keep their meaning; WebSocket + // responses are not modelled (nothing here proxies one). + inflightRequests = 0; + sleepAfterMs = 0; + + async containerFetch(requestOrUrl, portOrInit, portParam) { + this.inflightRequests++; + let res; + try { + this.renewActivityTimeout(); + res = await hooks.containerFetch(this, requestOrUrl, portOrInit, portParam); + } catch (e) { + this.decrementInflight(); + throw e; + } + if (res.body !== null) { + const { readable, writable } = new TransformStream(); + res.body + .pipeTo(writable) + .finally(() => this.decrementInflight()) + // The library leaves this rejection (a cancelled body) unhandled; + // under node it would crash the test process instead. + .catch(() => {}); + return new Response(readable, res); + } + this.decrementInflight(); + return res; + } + + decrementInflight() { + this.inflightRequests = Math.max(0, this.inflightRequests - 1); + if (this.inflightRequests === 0) this.renewActivityTimeout(); + } + + renewActivityTimeout() { + this.sleepAfterMs = Date.now() + parseTimeExpression(this.sleepAfter ?? "10m") * 1000; + } + + isActivityExpired() { + if (this.inflightRequests > 0) { + this.renewActivityTimeout(); + return false; + } + return this.sleepAfterMs <= Date.now(); + } + + async stop(signal) { + return hooks.stop(this, signal); + } + + async destroy() { + this._state = { status: "stopped_with_code", exitCode: 137, lastChange: Date.now() }; + } + + onStart() {} + onStop(_params) {} + onActivityExpired() { + return this.stop(); + } + onError(error) { + throw error; + } + + /** The real `Container.schedule()` persists to SQLite and is later + * invoked by the base class's own `alarm()` loop — machinery this stub + * does not reimplement (see `box.ts`'s own tests, `o11y-wake.test.mjs`, + * which monkey-patch `instance.schedule` per test instead, to observe + * what gets scheduled without needing real timing). This default just + * records the call and never auto-invokes it — a harmless no-op for + * every test that calls `wake()`/`#doWake` (which schedules the 4-hour + * hard cap) without caring about scheduling at all. */ + async schedule(when, callback, payload) { + this._scheduled ??= []; + const entry = { taskId: `stub-${this._scheduled.length}`, when, callback, payload }; + this._scheduled.push(entry); + return entry; + } +} diff --git a/runner/pipeline/fixtures/cost-ledger-fake.mjs b/runner/pipeline/fixtures/cost-ledger-fake.mjs new file mode 100644 index 0000000000..ad5b61d40f --- /dev/null +++ b/runner/pipeline/fixtures/cost-ledger-fake.mjs @@ -0,0 +1,109 @@ +// A minimal in-memory `env.DB` fake covering exactly the `cost_ledger` and +// `runner_settings` SQL shapes `budget.ts`/`settings.ts`/`reconcile.ts` issue +// — not a general SQL engine, the same "cover exactly what the routes touch" +// scope `worker-harness.mjs#fakeD1` documents for the demos/tokens tables. +// Used by `o11y-cost.test.mjs`/`o11y-alerts.test.mjs` (the o11y spend cap +// rule reads through `budget.ts#computeO11ySpend`). + +/** @returns {{ DB: object, _ledger: Map, _settings: Map }} */ +export function fakeCostD1(seedLedgerRows = []) { + // key: `${day}|${sku}|${source}` + const ledger = new Map(); + for (const row of seedLedgerRows) { + ledger.set(`${row.day}|${row.sku}|${row.source}`, { ...row, updated_at: row.updated_at ?? Date.now() }); + } + const settings = new Map(); // key -> { value, updated_at, updated_by } + + function upsertEstimateRow(day, sku, units, usd, updatedAt) { + const key = `${day}|${sku}|estimate`; + const existing = ledger.get(key); + if (existing) { + ledger.set(key, { day, sku, source: "estimate", units: existing.units + units, usd: existing.usd + usd, updated_at: updatedAt }); + } else { + ledger.set(key, { day, sku, source: "estimate", units, usd, updated_at: updatedAt }); + } + } + + function setBillingRow(day, sku, units, usd, updatedAt) { + ledger.set(`${day}|${sku}|billing`, { day, sku, source: "billing", units, usd, updated_at: updatedAt }); + } + + const DB = { + prepare(sql) { + let binds = []; + const stmt = { + bind(...args) { + binds = args; + return stmt; + }, + async run() { + if (/INSERT INTO cost_ledger/.test(sql) && /'estimate'/.test(sql)) { + const [day, sku, units, usd, updatedAt] = binds; + upsertEstimateRow(day, sku, units, usd, updatedAt); + return { success: true }; + } + if (/INSERT INTO cost_ledger/.test(sql) && /'billing'/.test(sql)) { + const [day, sku, units, usd, updatedAt] = binds; + setBillingRow(day, sku, units, usd, updatedAt); + return { success: true }; + } + if (/DELETE FROM cost_ledger/.test(sql)) { + const [cutoff] = binds; + for (const [key, row] of [...ledger]) if (row.day < cutoff) ledger.delete(key); + return { success: true }; + } + if (/INSERT INTO runner_settings/.test(sql)) { + const [key, value, updatedAt, updatedBy] = binds; + settings.set(key, { value, updated_at: updatedAt, updated_by: updatedBy }); + return { success: true }; + } + if (/DELETE FROM runner_settings/.test(sql)) { + const [key] = binds; + settings.delete(key); + return { success: true }; + } + throw new Error(`fakeCostD1: unhandled run() SQL: ${sql}`); + }, + async all() { + if (/FROM cost_ledger WHERE day >= /.test(sql)) { + const [since] = binds; + const rows = [...ledger.values()].filter((r) => r.day >= since).sort((a, b) => (a.day < b.day ? 1 : -1)); + return { results: rows }; + } + if (/FROM cost_ledger/.test(sql) && /GROUP BY day, sku/.test(sql)) { + const [dayLike] = binds; + const prefix = dayLike.replace(/%$/, ""); + const skuMatch = /sku IN \(([^)]+)\)/.exec(sql); + const allowedSkus = skuMatch ? skuMatch[1].split(",").map((s) => s.trim().replace(/'/g, "")) : null; + const byDaySku = new Map(); + for (const row of ledger.values()) { + if (!row.day.startsWith(prefix)) continue; + if (allowedSkus && !allowedSkus.includes(row.sku)) continue; + const key = `${row.day}|${row.sku}`; + const existing = byDaySku.get(key); + // COALESCE(billing, estimate, 0), same precedence as the real query. + if (!existing || row.source === "billing") byDaySku.set(key, row); + else if (existing.source !== "billing") byDaySku.set(key, row); + } + return { results: [...byDaySku.values()].map((r) => ({ usd: r.usd, reconciled: r.source === "billing" ? 1 : 0 })) }; + } + throw new Error(`fakeCostD1: unhandled all() SQL: ${sql}`); + }, + async first() { + if (/FROM runner_settings WHERE key = /.test(sql)) { + const [key] = binds; + const row = settings.get(key); + return row ? { value: row.value, updated_at: row.updated_at, updated_by: row.updated_by } : null; + } + throw new Error(`fakeCostD1: unhandled first() SQL: ${sql}`); + }, + }; + return stmt; + }, + batch(stmts) { + return Promise.all(stmts.map((s) => s.run())); + }, + }; + + return { DB, _ledger: ledger, _settings: settings }; +} diff --git a/runner/pipeline/fixtures/fake-ae-query.mjs b/runner/pipeline/fixtures/fake-ae-query.mjs new file mode 100644 index 0000000000..2d6fe6a2f9 --- /dev/null +++ b/runner/pipeline/fixtures/fake-ae-query.mjs @@ -0,0 +1,102 @@ +// A tiny fake Analytics Engine query engine for `pipeline/o11y-alerts.test.mjs`: +// recognises exactly the SQL shapes `workers/o11y/src/alerts/rules.ts`'s +// shared helpers generate — grouped `sum(_sample_interval * <count col>)` +// counts and a `quantileExactWeighted` read — and answers them from a +// plain JS array of seeded rows, so each rule's threshold/comparison logic +// is testable without a live ClickHouse/AE endpoint. Column slots are +// resolved generically via `AE_COLUMNS`, never a hand-numbered +// `blob8`/`double1` literal, so this fixture stays correct if the contract +// ever renumbers a slot. +// +// Deliberately narrow: throws on any SQL shape it does not recognise, +// rather than silently answering `[]`. + +import { AE_COLUMNS } from "@handsontable/demo-runtime/telemetry"; + +const SLOT_TO_NAME = Object.fromEntries(Object.entries(AE_COLUMNS).map(([k, v]) => [v, k])); + +/** + * @param {Array<Record<string, unknown> & { metric: string; ageMs?: number }>} rows + * Each row is a logical record — `metric`, plus whichever contract column + * names (`outcome`, `tier`, `surface`, `demo_id`, `ht_major`, + * `duration_ms`, `count`) the rule under test filters/groups/sums on. + * `ageMs` (default 0 = "now") is how old the row is, for window filtering. + */ +export function makeFakeAeQuery(rows) { + const calls = []; + + async function queryFn(_env, sql) { + calls.push(sql); + + const metricMatch = /index1 = '([^']*)'/.exec(sql); + const metric = metricMatch?.[1]; + let candidates = rows.filter((r) => r.metric === metric); + + // Window bounds: `timestamp >= now() - INTERVAL 'S' SECOND` (always + // present) and, for a day-over-day comparison, an upper bound too: + // `AND timestamp < now() - INTERVAL 'E' SECOND`. + const startMatch = /timestamp >= now\(\) - INTERVAL '(\d+)' SECOND/.exec(sql); + const windowStartS = startMatch ? Number(startMatch[1]) : Infinity; + const endMatch = /timestamp < now\(\) - INTERVAL '(\d+)' SECOND/.exec(sql); + const windowEndS = endMatch ? Number(endMatch[1]) : -Infinity; + candidates = candidates.filter((r) => { + const ageS = (r.ageMs ?? 0) / 1000; + return ageS <= windowStartS && ageS > windowEndS; + }); + + // Extra equality filters: `AND blobN = 'value'` / `AND doubleN = 'value'`. + const filterRe = /AND (blob\d+|double\d+) = '([^']*)'/g; + let fm; + while ((fm = filterRe.exec(sql))) { + const logical = SLOT_TO_NAME[fm[1]]; + if (!logical) throw new Error(`fake-ae-query: unknown slot in filter: ${fm[1]}`); + const value = fm[2]; + candidates = candidates.filter((r) => String(r[logical] ?? "") === value); + } + + // Set-exclusion filters: `AND blobN NOT IN ('a', 'b', ...)` — minor + // triage item 6 (`fiveXxRateRule`'s route-class exclusion). + const notInRe = /AND (blob\d+|double\d+) NOT IN \(([^)]*)\)/g; + let nim; + while ((nim = notInRe.exec(sql))) { + const logical = SLOT_TO_NAME[nim[1]]; + if (!logical) throw new Error(`fake-ae-query: unknown slot in NOT IN filter: ${nim[1]}`); + const excluded = new Set([...nim[2].matchAll(/'([^']*)'/g)].map((m) => m[1])); + candidates = candidates.filter((r) => !excluded.has(String(r[logical] ?? ""))); + } + + // `quantileExactWeighted(q)(<col>, toUInt32(_sample_interval)) AS p` + const qm = /quantileExactWeighted\(([\d.]+)\)\((double\d+),/.exec(sql); + if (qm) { + const valueLogical = SLOT_TO_NAME[qm[2]]; + if (!valueLogical) throw new Error(`fake-ae-query: unknown slot in quantile: ${qm[2]}`); + const values = candidates + .map((r) => r[valueLogical]) + .filter((v) => typeof v === "number") + .sort((a, b) => a - b); + if (values.length === 0) return []; + const q = Number(qm[1]); + const idx = Math.min(values.length - 1, Math.max(0, Math.ceil(q * values.length) - 1)); + return [{ p: values[idx] }]; + } + + // `SELECT <col> AS <alias>, sum(_sample_interval * <countCol>) AS c ... GROUP BY <col>` + const groupMatch = /SELECT (blob\d+) AS (\w+), sum\(_sample_interval \* (double\d+)\) AS c/.exec(sql); + if (groupMatch) { + const groupLogical = SLOT_TO_NAME[groupMatch[1]]; + const countLogical = SLOT_TO_NAME[groupMatch[3]]; + if (!groupLogical || !countLogical) throw new Error(`fake-ae-query: unknown slot in SELECT: ${sql}`); + const totals = new Map(); + for (const r of candidates) { + const key = String(r[groupLogical] ?? ""); + totals.set(key, (totals.get(key) ?? 0) + Number(r[countLogical] ?? 1)); + } + const alias = groupMatch[2]; + return [...totals.entries()].map(([k, c]) => ({ [alias]: k, c })); + } + + throw new Error(`fake-ae-query: unrecognised SQL shape: ${sql}`); + } + + return { queryFn, calls }; +} diff --git a/runner/pipeline/fixtures/faro/example-open.json b/runner/pipeline/fixtures/faro/example-open.json new file mode 100644 index 0000000000..e95acb687f --- /dev/null +++ b/runner/pipeline/fixtures/faro/example-open.json @@ -0,0 +1,21 @@ +{ + "meta": { + "app": { "name": "demos-authoring", "version": "deadbeef1234" } + }, + "events": [ + { + "name": "example.open", + "timestamp": "2026-01-01T00:00:00.000Z", + "attributes": { + "hot.surface": "authoring", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.bucket": "18.1", + "hot.metric_kind": "docs", + "hot.ref": "/guide/getting-started", + "hot.area": "getting-started", + "hot.reason": "entry" + } + } + ] +} diff --git a/runner/pipeline/fixtures/faro/exception-code-frame.json b/runner/pipeline/fixtures/faro/exception-code-frame.json new file mode 100644 index 0000000000..677a09e600 --- /dev/null +++ b/runner/pipeline/fixtures/faro/exception-code-frame.json @@ -0,0 +1,26 @@ +{ + "meta": { + "app": { "name": "demos-authoring", "version": "deadbeef1234" }, + "browser": { "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36" } + }, + "exceptions": [ + { + "type": "ReferenceError", + "value": "x is not defined\n\n 1 | function f() {\n> 2 | return x + 1;\n | ^\n 3 | }", + "timestamp": "2026-01-01T00:00:00.000Z", + "stacktrace": { + "frames": [ + { "filename": "https://8787-abc123-tok3n.demos.handsontable.com/assets/index-abc123.js?t=1700000000" }, + { "filename": "https://demos.handsontable.com/assets/main-def456.js" } + ] + }, + "context": { + "hot.surface": "authoring", + "hot.tier": "1", + "hot.framework": "react", + "hot.ht_major": "18", + "handled": "false" + } + } + ] +} diff --git a/runner/pipeline/fixtures/faro/log.json b/runner/pipeline/fixtures/faro/log.json new file mode 100644 index 0000000000..0ccf4bb79b --- /dev/null +++ b/runner/pipeline/fixtures/faro/log.json @@ -0,0 +1,15 @@ +{ + "meta": { + "app": { "name": "demos-authoring", "version": "deadbeef1234" } + }, + "logs": [ + { + "message": "share link copied", + "timestamp": "2026-01-01T00:00:00.000Z", + "context": { + "hot.surface": "share", + "hot.demo_id": "r-react-18-0-0" + } + } + ] +} diff --git a/runner/pipeline/fixtures/faro/measurement.json b/runner/pipeline/fixtures/faro/measurement.json new file mode 100644 index 0000000000..791fc6bec4 --- /dev/null +++ b/runner/pipeline/fixtures/faro/measurement.json @@ -0,0 +1,20 @@ +{ + "meta": { + "app": { "name": "demos-authoring", "version": "deadbeef1234" } + }, + "measurements": [ + { + "type": "preview.ready_ms", + "values": { "duration_ms": 842 }, + "timestamp": "2026-01-01T00:00:00.000Z", + "context": { + "hot.surface": "authoring", + "hot.tier": "1", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.outcome": "ready", + "hot.bucket": "18.1" + } + } + ] +} diff --git a/runner/pipeline/fixtures/faro/web-vitals.json b/runner/pipeline/fixtures/faro/web-vitals.json new file mode 100644 index 0000000000..338c56122f --- /dev/null +++ b/runner/pipeline/fixtures/faro/web-vitals.json @@ -0,0 +1,19 @@ +{ + "meta": { + "app": { "name": "demos-authoring", "version": "deadbeef1234" } + }, + "measurements": [ + { + "type": "web-vitals", + "values": { "lcp": 1234.5, "inp": 45, "cls": 0.02, "fcp": 900 }, + "timestamp": "2026-01-01T00:00:00.000Z", + "context": { + "hot.surface": "share", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.device": "mobile", + "hot.demo_id": "r-react-18-0-0" + } + } + ] +} diff --git a/runner/pipeline/fixtures/o11y-cloudflare-workers-stub.mjs b/runner/pipeline/fixtures/o11y-cloudflare-workers-stub.mjs new file mode 100644 index 0000000000..4e63139a62 --- /dev/null +++ b/runner/pipeline/fixtures/o11y-cloudflare-workers-stub.mjs @@ -0,0 +1,23 @@ +// Structural stand-in for `cloudflare:workers` (only exists inside +// workerd), used by `o11y-worker-hooks.mjs` and `worker-hooks.mjs`. +// `InboxWriter` (writer.ts) extends `DurableObject<Env>` and reads only +// `this.ctx`/`this.env` — the real base class's constructor signature is +// `(ctx, env)`, mirrored exactly. `WorkerEntrypoint` covers +// `workers/o11y/src/heartbeat.ts` (`O11yHeartbeat`) and +// `workers/api/src/o11y-usage.ts` (`O11yUsage`) the same way, since both +// read only `this.ctx`/`this.env` too. Shared across both hook files +// rather than a second stub. + +export class DurableObject { + constructor(ctx, env) { + this.ctx = ctx; + this.env = env; + } +} + +export class WorkerEntrypoint { + constructor(ctx, env) { + this.ctx = ctx; + this.env = env; + } +} diff --git a/runner/pipeline/fixtures/o11y-harness.mjs b/runner/pipeline/fixtures/o11y-harness.mjs new file mode 100644 index 0000000000..df8f25357a --- /dev/null +++ b/runner/pipeline/fixtures/o11y-harness.mjs @@ -0,0 +1,191 @@ +// Shared in-memory env for the o11y worker's tests — the same pattern +// `worker-harness.mjs` uses for `workers/api`'s D1/KV/R2 fakes (TESTING.md: +// "in-memory fakes for worker bindings"). Imports nothing from +// `workers/o11y/src/`, so it is safe to import before `o11y-worker-hooks.mjs` +// is registered (mirrors `worker-harness.mjs`'s own header note). + +export const ctx = { + waitUntil(promise) { + // Tests await route handlers directly and then drain this queue, so a + // `writePoint`'s `ctx.waitUntil` write lands before assertions run. + this._pending.push(Promise.resolve(promise).catch(() => {})); + }, + _pending: [], + async drain() { + await Promise.all(this._pending); + this._pending.length = 0; + }, +}; + +/** A real `DurableObjectStorage`-shaped `Map` fake — the overloaded `get` + * branches on `Array.isArray` at runtime (plain JS, no TS overload + * wrangling needed here). `transaction()` calls the closure with itself, so + * nested `txn.get`/`.put`/`.delete`/`.list` calls hit the same backing + * `Map` atomically-in-spirit (single-threaded Node, no real concurrency to + * guard against). */ +// The real SQLite-backed DO storage API caps `get`/`put`/`delete` at 128 +// keys/pairs per call (see `workers/o11y/src/inbox/storage.ts`'s +// `DO_STORAGE_MAX_KEYS_PER_CALL` doc comment for the exact Cloudflare docs +// quote and URL). Every multi-key call below throws past the real limit, +// the same as `inbox/storage.ts#memoryStorage()`, so a caller that forgets +// to chunk is caught here rather than in production. +const DO_STORAGE_MAX_KEYS_PER_CALL = 128; + +export function makeDurableObjectStorage(seed = new Map()) { + const data = seed; + let alarm = null; + const storage = { + async get(keyOrKeys) { + if (Array.isArray(keyOrKeys)) { + if (keyOrKeys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`makeDurableObjectStorage().get: ${keyOrKeys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL}`); + } + const out = new Map(); + for (const k of keyOrKeys) if (data.has(k)) out.set(k, data.get(k)); + return out; + } + return data.get(keyOrKeys); + }, + async put(entries) { + const keys = Object.keys(entries); + if (keys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`makeDurableObjectStorage().put: ${keys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL}`); + } + for (const [k, v] of Object.entries(entries)) data.set(k, v); + }, + async delete(keys) { + if (keys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`makeDurableObjectStorage().delete: ${keys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL}`); + } + let n = 0; + for (const k of keys) if (data.delete(k)) n++; + return n; + }, + // A real `DurableObjectStorage.list` + // also accepts `start`/`end`/`limit` (ledger.ts's bounded-range prune + // sweeps use exactly these, never an unbounded `prefix`-only scan — see + // `storage.ts`'s `ListOptions` doc comment); this fake must not silently + // ignore all three: a call with `start`/`end` but no `prefix` fell + // through the `!options?.prefix` check and returned the WHOLE storage + // Map, so a prune call's `storage.delete([...matches])` deleted + // EVERYTHING in the DO, not just the intended stale range (caught by + // `o11y-cap-wake.test.mjs`'s backlog test going from a real `written` + // key to `undefined`). Always sorts ascending, matching the real + // binding's default (`reverse` is never requested by this codebase). + async list(options) { + const matches = []; + for (const [k, v] of data) { + if (options?.prefix && !k.startsWith(options.prefix)) continue; + if (options?.start !== undefined && k < options.start) continue; + if (options?.end !== undefined && k >= options.end) continue; + matches.push([k, v]); + } + matches.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)); + const limited = options?.limit !== undefined ? matches.slice(0, options.limit) : matches; + return new Map(limited); + }, + async transaction(closure) { + return closure(storage); + }, + async getAlarm() { + return alarm; + }, + async setAlarm(t) { + alarm = t instanceof Date ? t.getTime() : t; + }, + async deleteAlarm() { + alarm = null; + }, + _data: data, + }; + return storage; +} + +export function makeR2Bucket() { + const objects = new Map(); + return { + objects, + async put(key, value) { + objects.set(key, value instanceof Uint8Array ? value : new Uint8Array(value)); + }, + async get(key) { + const v = objects.get(key); + if (!v) return null; + return { body: v, async arrayBuffer() { return v.buffer; } }; + }, + }; +} + +export function makeAnalyticsEngine() { + const points = []; + return { + points, + writeDataPoint(point) { + points.push(point); + }, + }; +} + +const SECRET = "test-export-secret"; +const SENTRY_SECRET = "test-sentry-secret"; + +/** + * `InboxWriterClass` is the real `InboxWriter` (`workers/o11y/src/inbox/writer.ts`), + * dynamically imported by the caller **after** `o11y-worker-hooks.mjs` is + * registered (this module must not import it itself — see the header). One + * shared `DurableObjectStorage` fake backs the constructed instance, so a + * caller that wants to simulate a restart just constructs a second + * `InboxWriterClass` instance over the same `doStorage`. + */ +export function makeEnv(InboxWriterClass, overrides = {}) { + const doStorage = overrides.doStorage ?? makeDurableObjectStorage(); + const r2 = overrides.r2 ?? makeR2Bucket(); + const ae = overrides.ae ?? makeAnalyticsEngine(); + + const env = { + O11Y_ENV: "production", + // Replaces ACCESS_TEAM_DOMAIN/ACCESS_AUD (Cloudflare Access). + LOGIN_BROKER_URL: "https://mcp-auth-proxy.example.test", + O11Y_SESSION_SECRET: "test-session-secret-at-least-32-bytes-long", + GITHUB_OIDC_REPOSITORY: "handsontable/examples", + GITHUB_OIDC_WORKFLOW_REF: "handsontable/examples/.github/workflows/master.yml@refs/heads/master", + O11Y_EXPORT_SECRET: SECRET, + SENTRY_HOOK_SECRET: SENTRY_SECRET, + AE_SQL_TOKEN: "test-ae-token", + O11Y_INBOX: r2, + O11Y_LOKI_STATE: makeR2Bucket(), + O11Y_MAPS: makeR2Bucket(), + RUNNER_EVENTS: ae, + API: { fetch: async () => new Response(null, { status: 204 }) }, + RATE_LIMITER: { limit: async () => ({ success: true }) }, + ...overrides.env, + }; + + const doState = { storage: doStorage }; + const inboxWriterInstance = new InboxWriterClass(doState, env); + + env.INBOX_WRITER = { + jurisdiction() { + return this; + }, + idFromName(name) { + return { toString: () => name, name }; + }, + get() { + return inboxWriterInstance; + }, + }; + env.GRAFANA_BOX = { + jurisdiction() { + return this; + }, + idFromName(name) { + return { toString: () => name, name }; + }, + get() { + throw new Error("GrafanaBox not constructed in this harness (T01's class)"); + }, + }; + + return { env, doStorage, r2, ae, inboxWriterInstance }; +} diff --git a/runner/pipeline/fixtures/o11y-inbox-helpers.mjs b/runner/pipeline/fixtures/o11y-inbox-helpers.mjs new file mode 100644 index 0000000000..7be2cf8a45 --- /dev/null +++ b/runner/pipeline/fixtures/o11y-inbox-helpers.mjs @@ -0,0 +1,23 @@ +// Test-only helper for `pipeline/o11y-inbox.test.mjs`: production reads +// pending rows only through the bounded `pack.ts#collectRowBatch`. + +const ROW_PREFIX = "row:"; + +function rowNumber(key) { + return Number(key.slice(ROW_PREFIX.length)); +} + +/** All pending rows, grouped by tenant, in row-insertion order (numeric sort + * in memory — unbounded, so this is for tests only, over a small, known row + * count). */ +export async function pendingRowsByTenant(storage) { + const rows = await storage.list({ prefix: ROW_PREFIX }); + const sorted = [...rows.entries()].sort(([a], [b]) => rowNumber(a) - rowNumber(b)); + const byTenant = new Map(); + for (const entry of sorted) { + const list = byTenant.get(entry[1].tenant) ?? []; + list.push(entry); + byTenant.set(entry[1].tenant, list); + } + return byTenant; +} diff --git a/runner/pipeline/fixtures/o11y-symbolicate-drain-child.mjs b/runner/pipeline/fixtures/o11y-symbolicate-drain-child.mjs new file mode 100644 index 0000000000..bdb8bfc22b --- /dev/null +++ b/runner/pipeline/fixtures/o11y-symbolicate-drain-child.mjs @@ -0,0 +1,66 @@ +// Child process for `pipeline/o11y-symbolicate-drain.test.mjs`. +// +// The parent spawns this with `--disallow-code-generation-from-strings`, +// the V8 policy workerd applies to every Worker (`eval` / `new Function` +// throw `EvalError: Code generation from strings disallowed for this +// context`). A `node --test` file cannot set that flag for itself, and +// without it Node happily runs a map library that generates code — a +// symbolication test could pass in Node while every lookup in the real +// Worker throws. +// +// Runs the real `drain.ts#drainBatch` with the real +// `symbolicate.ts#symbolicateResourceLogs` over one gzipped inbox object +// the parent wrote, and prints one JSON line to stdout: whether code +// generation really was blocked in this process, the batch outcome, the +// decoded Loki push bodies, and every skip report `onSkip` received. +// +// argv: <workdir> (holds `inbox.ndjson.gz`, `inbox-key.txt` and `maps/<key>`) + +import { register } from "node:module"; +import { readFileSync, existsSync } from "node:fs"; +import path from "node:path"; + +register("./o11y-worker-hooks.mjs", import.meta.url); + +const workdir = process.argv[2]; +if (!workdir) throw new Error("usage: o11y-symbolicate-drain-child.mjs <workdir>"); + +let codegenBlocked = false; +try { + new Function("return 1"); +} catch (err) { + codegenBlocked = err instanceof EvalError; +} + +const { drainBatch } = await import("../../workers/o11y/src/drain/drain.ts"); +const { symbolicateResourceLogs } = await import("../../workers/o11y/src/drain/symbolicate.ts"); + +const inboxKey = readFileSync(path.join(workdir, "inbox-key.txt"), "utf8").trim(); +const inboxObject = new Uint8Array(readFileSync(path.join(workdir, "inbox.ndjson.gz"))); + +async function gunzip(bytes) { + const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream("gzip")); + return new Response(stream).text(); +} + +const pushes = []; +const skips = []; +const mapReads = []; +const result = await drainBatch([inboxKey], new Set(), { + fetchObject: async (key) => (key === inboxKey ? inboxObject : null), + pushToLoki: async (tenant, gzippedBody) => { + pushes.push({ tenant, body: JSON.parse(await gunzip(gzippedBody)) }); + return { status: 204 }; + }, + symbolicate: (records) => + symbolicateResourceLogs(records, { + getMap: async (key) => { + mapReads.push(key); + const file = path.join(workdir, "maps", key); + return existsSync(file) ? readFileSync(file, "utf8") : null; + }, + onSkip: (reported, suppressed, overCap) => skips.push({ reported, suppressed, overCap }), + }), +}); + +process.stdout.write(`${JSON.stringify({ codegenBlocked, result, pushes, skips, mapReads })}\n`); diff --git a/runner/pipeline/fixtures/o11y-worker-hooks.mjs b/runner/pipeline/fixtures/o11y-worker-hooks.mjs new file mode 100644 index 0000000000..00d64ff9d8 --- /dev/null +++ b/runner/pipeline/fixtures/o11y-worker-hooks.mjs @@ -0,0 +1,57 @@ +// Module hooks that make the real o11y worker (workers/o11y/src/index.ts, +// re-exporting `InboxWriter` and `GrafanaBox`) loadable under plain +// `node --experimental-strip-types --test`, registered via +// `module.register()` before the worker is imported. One shared file, not +// two, since both specs import through `index.ts`. +// +// Three obstacles: the worker's modules import each other by `.js` +// specifier (the Workers bundler's shape) but the files on disk are `.ts` +// — map the extension for relative imports inside `workers/o11y/src/`. +// `cloudflare:workers` (`InboxWriter`'s DurableObject base) only exists +// inside workerd, and needs only an inert stub +// (`o11y-cloudflare-workers-stub.mjs`) since `InboxWriter` touches nothing +// beyond `this.ctx`/`this.env`. `@cloudflare/containers` (`GrafanaBox`'s +// base) also only exists inside workerd; `GrafanaBox`'s own +// container-lifecycle specs drive a real instance, so its stub +// (`cloudflare-containers-stub.mjs`) is a fuller structural double. + +const CLOUDFLARE_WORKERS_STUB = new URL("./o11y-cloudflare-workers-stub.mjs", import.meta.url).href; +const CLOUDFLARE_CONTAINERS_STUB = new URL("./cloudflare-containers-stub.mjs", import.meta.url).href; + +// `jose`, `source-map-js` (a `workers/o11y` devDependency, borrowed the same +// way for `pipeline/o11y-symbolicate.test.mjs`, which needs to build a real +// source map with `SourceMapGenerator` to test against — `drain/symbolicate.ts` +// itself resolves maps with `@jridgewell/trace-mapping`) and +// `@handsontable/demo-runtime` (any subpath) are +// `workers/o11y`'s dependencies, not the pipeline's — a plain node resolve +// only succeeds when the importing file lives under `workers/o11y/`, which +// every gate/normalise module does. A test file under `pipeline/` that also +// wants to sign a test JWT, build a source map, or read +// `decodeNdjson`/`toAePoint` directly to assert on inbox output, has no such +// ancestor `node_modules` entry; borrow one by resolving as if the request +// came from inside `workers/o11y/src/` instead. +const WORKERS_O11Y_SRC_URL = new URL("../../workers/o11y/src/index.ts", import.meta.url).href; +const BORROWED_SPECIFIERS = ["jose", "source-map-js", "@handsontable/demo-runtime"]; + +export async function resolve(specifier, context, nextResolve) { + if (specifier === "cloudflare:workers") { + return { url: CLOUDFLARE_WORKERS_STUB, shortCircuit: true }; + } + if (specifier === "@cloudflare/containers") { + return { url: CLOUDFLARE_CONTAINERS_STUB, shortCircuit: true }; + } + if ( + BORROWED_SPECIFIERS.some((s) => specifier === s || specifier.startsWith(`${s}/`)) + && !context.parentURL?.includes("/workers/o11y/") + ) { + return nextResolve(specifier, { ...context, parentURL: WORKERS_O11Y_SRC_URL }); + } + if ( + specifier.startsWith(".") + && specifier.endsWith(".js") + && context.parentURL?.includes("/workers/o11y/src/") + ) { + return nextResolve(`${specifier.slice(0, -3)}.ts`, context); + } + return nextResolve(specifier, context); +} diff --git a/runner/pipeline/fixtures/otlp/build-protobuf-fixtures.mjs b/runner/pipeline/fixtures/otlp/build-protobuf-fixtures.mjs new file mode 100644 index 0000000000..3daa6df6e2 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/build-protobuf-fixtures.mjs @@ -0,0 +1,122 @@ +#!/usr/bin/env node +// Hand-encodes the OTLP `ExportLogsServiceRequest` protobuf fixtures the +// tests replay (`pipeline/o11y-normalise.test.mjs`, +// `scripts/o11y-replay-fixtures.mjs`) — the binary-wire mirror of +// `pipeline/fixtures/otlp/json/basic.json` and `.../zero-timestamp.json`, +// same field values, so the two decoders +// (`workers/o11y/src/normalise/otlp.ts`'s `decodeOtlpJson` / +// `otlp-protobuf.ts`'s `decodeOtlpProtobuf`) can be tested against +// equivalent inputs. Uses `@bufbuild/protobuf/wire`'s `BinaryWriter` only. +// Every `repeated` field must be written as one tag+length-prefix per +// element (protobuf's actual wire rule) — wrapping a whole loop's worth of +// elements in one shared frame round-trips as garbage. +// +// Regenerate: `node --experimental-strip-types +// pipeline/fixtures/otlp/build-protobuf-fixtures.mjs`, run with a `cwd` +// inside `workers/o11y` (or `NODE_PATH` pointing at its `node_modules`) so +// `@bufbuild/protobuf` resolves. Output committed — these are fixtures, +// not build artifacts. + +import { writeFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { BinaryWriter, WireType } from "@bufbuild/protobuf/wire"; + +/** Writes an `AnyValue` message's contents (just `string_value = 1`, the + * only kind these fixtures need). */ +function anyValueString(w, value) { + w.tag(1, WireType.LengthDelimited).string(value); +} + +/** Writes a `KeyValue` message's contents: `key = 1`, `value = 2` (AnyValue). */ +function keyValueContents(w, key, value) { + w.tag(1, WireType.LengthDelimited).string(key); + w.tag(2, WireType.LengthDelimited).fork(); + anyValueString(w, value); + w.join(); +} + +/** Writes one `repeated KeyValue` element at `fieldNo` — each element of a + * repeated message field gets its **own** tag + length prefix; this is the + * bug the file header describes fixing. */ +function writeKeyValue(w, fieldNo, key, value) { + w.tag(fieldNo, WireType.LengthDelimited).fork(); + keyValueContents(w, key, value); + w.join(); +} + +function writeLogRecord(w, fieldNo, { timeUnixNano, observedTimeUnixNano, body, attrs }) { + w.tag(fieldNo, WireType.LengthDelimited).fork(); + if (timeUnixNano !== undefined) w.tag(1, WireType.Bit64).fixed64(BigInt(timeUnixNano)); + if (observedTimeUnixNano !== undefined) w.tag(11, WireType.Bit64).fixed64(BigInt(observedTimeUnixNano)); + w.tag(5, WireType.LengthDelimited).fork(); // body = 5 (AnyValue) + anyValueString(w, body); + w.join(); + for (const [k, v] of attrs) writeKeyValue(w, 6, k, v); // attributes = 6 + w.join(); +} + +function writeScopeLogs(w, fieldNo, records) { + w.tag(fieldNo, WireType.LengthDelimited).fork(); + for (const record of records) writeLogRecord(w, 2, record); // log_records = 2 + w.join(); +} + +function writeResource(w, fieldNo, attrs) { + w.tag(fieldNo, WireType.LengthDelimited).fork(); + for (const [k, v] of attrs) writeKeyValue(w, 1, k, v); // attributes = 1 + w.join(); +} + +function writeResourceLogs(w, fieldNo, { resourceAttrs, records }) { + w.tag(fieldNo, WireType.LengthDelimited).fork(); + writeResource(w, 1, resourceAttrs); // resource = 1 + writeScopeLogs(w, 2, records); // scope_logs = 2 + w.join(); +} + +function build(resourceAttrs, records) { + const w = new BinaryWriter(); + writeResourceLogs(w, 1, { resourceAttrs, records }); // ExportLogsServiceRequest.resource_logs = 1 + return w.finish(); +} + +const basic = build( + [ + ["service.name", "demos-api"], + ["service.version", "cafef00d"], + ["deployment.environment.name", "production"], + ], + [ + { + timeUnixNano: "1735689600000000000", + body: "api.request route=api/demos status=200 (protobuf)", + attrs: [ + ["hot.surface", "api"], + ["hot.outcome", "2xx"], + ["cf.ray", "8a1b2c3d4e5f6789"], + ], + }, + ], +); + +const zeroTimestamp = build( + [ + ["service.name", "demos-api"], + ["service.version", "cafef00d"], + ["deployment.environment.name", "production"], + ], + [ + { + timeUnixNano: "0", + observedTimeUnixNano: "0", + body: "a protobuf record with no real timestamp, exit criterion 3", + attrs: [["hot.surface", "api"]], + }, + ], +); + +const dir = fileURLToPath(new URL(".", import.meta.url)); +writeFileSync(`${dir}protobuf/basic.bin`, basic); +writeFileSync(`${dir}protobuf/zero-timestamp.bin`, zeroTimestamp); +console.log(`wrote ${basic.byteLength} bytes to protobuf/basic.bin`); +console.log(`wrote ${zeroTimestamp.byteLength} bytes to protobuf/zero-timestamp.bin`); diff --git a/runner/pipeline/fixtures/otlp/deploy-event.json b/runner/pipeline/fixtures/otlp/deploy-event.json new file mode 100644 index 0000000000..23c0d626a1 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/deploy-event.json @@ -0,0 +1,5 @@ +{ + "service": "handsontable-demos-api", + "sha": "abc123def456abc123def456abc123def456abc1", + "cf_version_id": "01234567-89ab-cdef-0123-456789abcdef" +} diff --git a/runner/pipeline/fixtures/otlp/json/basic.json b/runner/pipeline/fixtures/otlp/json/basic.json new file mode 100644 index 0000000000..daf8884e85 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/basic.json @@ -0,0 +1,29 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { "key": "service.name", "value": { "stringValue": "demos-api" } }, + { "key": "service.version", "value": { "stringValue": "cafef00d" } }, + { "key": "deployment.environment.name", "value": { "stringValue": "production" } } + ] + }, + "scopeLogs": [ + { + "logRecords": [ + { + "timeUnixNano": "1735689600000000000", + "severityText": "INFO", + "body": { "stringValue": "api.request route=api/demos status=200" }, + "attributes": [ + { "key": "hot.surface", "value": { "stringValue": "api" } }, + { "key": "hot.outcome", "value": { "stringValue": "2xx" } }, + { "key": "cf.ray", "value": { "stringValue": "8a1b2c3d4e5f6789" } } + ] + } + ] + } + ] + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/json/cloudflare-invocation-log.json b/runner/pipeline/fixtures/otlp/json/cloudflare-invocation-log.json new file mode 100644 index 0000000000..e4cd5c310f --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/cloudflare-invocation-log.json @@ -0,0 +1,257 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { + "key": "cloudflare.colo", + "value": { + "stringValue": "FRA" + } + }, + { + "key": "faas.invoked_region", + "value": { + "stringValue": "EEUR" + } + }, + { + "key": "cloudflare.script_name", + "value": { + "stringValue": "handsontable-demos-api" + } + }, + { + "key": "cloud.provider", + "value": { + "stringValue": "cloudflare" + } + }, + { + "key": "cloud.platform", + "value": { + "stringValue": "cloudflare.workers" + } + }, + { + "key": "faas.name", + "value": { + "stringValue": "handsontable-demos-api" + } + }, + { + "key": "faas.version", + "value": { + "stringValue": "00000000-0000-0000-0000-000000000000" + } + }, + { + "key": "telemetry.sdk.language", + "value": { + "stringValue": "js" + } + }, + { + "key": "telemetry.sdk.name", + "value": { + "stringValue": "workers-observability" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "handsontable-demos-api" + } + }, + { + "key": "cloudflare.script_version.id", + "value": { + "stringValue": "00000000-0000-0000-0000-000000000000" + } + } + ], + "droppedAttributesCount": 0 + }, + "scopeLogs": [ + { + "scope": { + "name": "workers-observability" + }, + "logRecords": [ + { + "timeUnixNano": "1790166300617000000", + "observedTimeUnixNano": "1790166300622000000", + "severityNumber": 9, + "body": { + "stringValue": "GET https://demos.handsontable.com/api/demos/r-react-18-0-0" + }, + "attributes": [ + { + "key": "cloudflare.execution_model", + "value": { + "stringValue": "stateless" + } + }, + { + "key": "cloudflare.handler_type", + "value": { + "stringValue": "fetch" + } + }, + { + "key": "faas.invocation_id", + "value": { + "stringValue": "00000000000000000000000000000000" + } + }, + { + "key": "cloudflare.ray_id", + "value": { + "stringValue": "0000000000000000" + } + }, + { + "key": "faas.trigger", + "value": { + "stringValue": "http" + } + }, + { + "key": "url.full", + "value": { + "stringValue": "https://demos.handsontable.com/api/demos/r-react-18-0-0" + } + }, + { + "key": "http.request.method", + "value": { + "stringValue": "GET" + } + }, + { + "key": "http.request.header.accept", + "value": { + "stringValue": "*/*" + } + }, + { + "key": "http.request.header.accept-encoding", + "value": { + "stringValue": "gzip, br" + } + }, + { + "key": "user_agent.original", + "value": { + "stringValue": "Mozilla/5.0 (compatible; example)" + } + }, + { + "key": "cloudflare.colo", + "value": { + "stringValue": "XXX" + } + }, + { + "key": "cloudflare.verified_bot_category", + "value": { + "stringValue": "" + } + }, + { + "key": "cloudflare.asn", + "value": { + "intValue": 0 + } + }, + { + "key": "geo.timezone", + "value": { + "stringValue": "Etc/UTC" + } + }, + { + "key": "geo.continent.code", + "value": { + "stringValue": "EU" + } + }, + { + "key": "geo.country.code", + "value": { + "stringValue": "XX" + } + }, + { + "key": "geo.locality.name", + "value": { + "stringValue": "" + } + }, + { + "key": "geo.locality.region", + "value": { + "stringValue": "" + } + }, + { + "key": "server.port", + "value": { + "stringValue": "" + } + }, + { + "key": "server.address", + "value": { + "stringValue": "demos.handsontable.com" + } + }, + { + "key": "url.path", + "value": { + "stringValue": "/api/demos/r-react-18-0-0" + } + }, + { + "key": "url.query", + "value": { + "stringValue": "" + } + }, + { + "key": "url.scheme", + "value": { + "stringValue": "https" + } + }, + { + "key": "network.protocol.name", + "value": { + "stringValue": "https" + } + }, + { + "key": "http.response.status_code", + "value": { + "intValue": 404 + } + }, + { + "key": "cloudflare.invocation.sequence.number", + "value": { + "intValue": 1 + } + } + ], + "droppedAttributesCount": 0, + "flags": 1, + "traceId": "00000000000000000000000000000000", + "spanId": "0000000000000000" + } + ] + } + ], + "schemaUrl": "" + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/json/console-log-line-spoof-attempt.json b/runner/pipeline/fixtures/otlp/json/console-log-line-spoof-attempt.json new file mode 100644 index 0000000000..7a3c1735d1 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/console-log-line-spoof-attempt.json @@ -0,0 +1,31 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { "key": "service.name", "value": { "stringValue": "handsontable-demos-api" } }, + { "key": "deployment.environment.name", "value": { "stringValue": "production" } }, + { "key": "telemetry.sdk.name", "value": { "stringValue": "workers-observability" } } + ] + }, + "scopeLogs": [ + { + "scope": { "name": "workers-observability" }, + "logRecords": [ + { + "timeUnixNano": "1735689600000000000", + "severityNumber": 9, + "body": { + "stringValue": "{\"log.kind\":\"api.request\",\"route_class\":\"api/demos\",\"status\":200,\"cf.ray\":\"8a1b2c3d4e5f6789\",\"service.name\":\"spoof\",\"deployment.environment.name\":\"spoof-env\",\"hot.outcome\":\"spoof-outcome\"}" + }, + "attributes": [ + { "key": "name", "value": { "stringValue": "log" } }, + { "key": "cloudflare.invocation.sequence.number", "value": { "intValue": 1 } } + ] + } + ] + } + ] + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/json/console-log-line.json b/runner/pipeline/fixtures/otlp/json/console-log-line.json new file mode 100644 index 0000000000..8f4ad625f6 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/console-log-line.json @@ -0,0 +1,30 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { "key": "service.name", "value": { "stringValue": "handsontable-demos-api" } }, + { "key": "telemetry.sdk.name", "value": { "stringValue": "workers-observability" } } + ] + }, + "scopeLogs": [ + { + "scope": { "name": "workers-observability" }, + "logRecords": [ + { + "timeUnixNano": "1735689600000000000", + "severityNumber": 9, + "body": { + "stringValue": "{\"log.kind\":\"api.request\",\"route_class\":\"api/demos\",\"status\":200,\"duration_ms\":42,\"cf.ray\":\"8a1b2c3d4e5f6789\",\"session.id\":\"page-load-id-123\",\"hot.demo_id\":\"r-react-18-0-0\",\"service.version\":\"cafef00d\"}" + }, + "attributes": [ + { "key": "name", "value": { "stringValue": "log" } }, + { "key": "cloudflare.invocation.sequence.number", "value": { "intValue": 1 } } + ] + } + ] + } + ] + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/json/forbidden-attrs.json b/runner/pipeline/fixtures/otlp/json/forbidden-attrs.json new file mode 100644 index 0000000000..599fb4581e --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/forbidden-attrs.json @@ -0,0 +1,35 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { "key": "service.name", "value": { "stringValue": "demos-api" } }, + { "key": "service.version", "value": { "stringValue": "cafef00d" } }, + { "key": "deployment.environment.name", "value": { "stringValue": "production" } }, + { "key": "url.full", "value": { "stringValue": "https://demos.handsontable.com/d/abc?token=secret123" } }, + { "key": "user_agent.original", "value": { "stringValue": "Mozilla/5.0 (Windows NT 10.0) Chrome/119.0" } }, + { "key": "geo.city", "value": { "stringValue": "Warsaw" } }, + { "key": "asn.number", "value": { "stringValue": "12345" } } + ] + }, + "scopeLogs": [ + { + "logRecords": [ + { + "timeUnixNano": "1735689600000000000", + "body": { + "stringValue": "stale preview request from https://8787-abc123-tok3n.demos.handsontable.com/src/main.js?t=1700000000 user-agent Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36" + }, + "attributes": [ + { "key": "url.full", "value": { "stringValue": "https://demos.handsontable.com/d/abc?token=secret123" } }, + { "key": "http.user_agent", "value": { "stringValue": "Mozilla/5.0 (Windows NT 10.0) Chrome/119.0" } }, + { "key": "geo.country", "value": { "stringValue": "PL" } }, + { "key": "hot.surface", "value": { "stringValue": "api" } } + ] + } + ] + } + ] + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/json/zero-timestamp.json b/runner/pipeline/fixtures/otlp/json/zero-timestamp.json new file mode 100644 index 0000000000..3bc99ffe66 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/json/zero-timestamp.json @@ -0,0 +1,25 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { "key": "service.name", "value": { "stringValue": "demos-api" } }, + { "key": "service.version", "value": { "stringValue": "cafef00d" } }, + { "key": "deployment.environment.name", "value": { "stringValue": "production" } } + ] + }, + "scopeLogs": [ + { + "logRecords": [ + { + "timeUnixNano": "0", + "observedTimeUnixNano": "0", + "body": { "stringValue": "a record with no real timestamp, exit criterion 3" }, + "attributes": [{ "key": "hot.surface", "value": { "stringValue": "api" } }] + } + ] + } + ] + } + ] +} diff --git a/runner/pipeline/fixtures/otlp/protobuf/basic.bin b/runner/pipeline/fixtures/otlp/protobuf/basic.bin new file mode 100644 index 0000000000..cdda63ab84 Binary files /dev/null and b/runner/pipeline/fixtures/otlp/protobuf/basic.bin differ diff --git a/runner/pipeline/fixtures/otlp/protobuf/zero-timestamp.bin b/runner/pipeline/fixtures/otlp/protobuf/zero-timestamp.bin new file mode 100644 index 0000000000..6b0fe72d6c Binary files /dev/null and b/runner/pipeline/fixtures/otlp/protobuf/zero-timestamp.bin differ diff --git a/runner/pipeline/fixtures/otlp/sentry-issue.json b/runner/pipeline/fixtures/otlp/sentry-issue.json new file mode 100644 index 0000000000..1d6ff50196 --- /dev/null +++ b/runner/pipeline/fixtures/otlp/sentry-issue.json @@ -0,0 +1,15 @@ +{ + "action": "created", + "installation": { "uuid": "11111111-2222-3333-4444-555555555555" }, + "data": { + "issue": { + "id": "987654321", + "shortId": "DEMOS-42", + "title": "TypeError: cannot read properties of undefined", + "level": "error", + "permalink": "https://handsoncode.sentry.io/issues/987654321/?query=is%3Aunresolved", + "lastRelease": { "version": "abc123def456abc123def456abc123def456abc1" } + } + }, + "actor": { "type": "application", "id": "sentry" } +} diff --git a/runner/pipeline/fixtures/stub-bin/curl b/runner/pipeline/fixtures/stub-bin/curl new file mode 100755 index 0000000000..aa90004c60 --- /dev/null +++ b/runner/pipeline/fixtures/stub-bin/curl @@ -0,0 +1,72 @@ +#!/bin/bash +# A minimal `curl` stand-in for pipeline/o11y-shutdown-snapshot.test.mjs. +# Ignores its real arguments entirely (the test only needs to control the +# RESPONSE `r2_list_prefix`/`snapshot_index_keys` see, not exercise real +# S3-sigv4 request construction — that is exercised for real by +# `containers/o11y/local/stop-roundtrip.mjs`). Behaviour is driven by two +# env vars the test sets: +# STUB_CURL_MODES comma-separated modes, one per successive invocation +# (clamped to the last entry once exhausted): +# fail -> exit 22, no output (curl -f on a +# network/5xx failure) +# empty -> a valid, empty <ListBucketResult> +# haskey -> a valid listing with one +# pre-existing uploader-named <Key> +# malformed -> HTTP 200 but not real S3 XML (a proxy +# error page) +# code200 -> bare `200` on stdout, nothing else — F2 +# fix (B-I2, second wave): `r2_put_and_verify` +# (lib.sh) calls curl with +# `-o /dev/null -w '%{http_code}'` for its +# PUT and its HEAD confirmation, so its +# whole stdout IS the status-code text, +# never S3 list XML — a REAL +# `run_stop_protocol()` marker write needs +# this mode for both of those calls, after +# whatever `empty`/`haskey` listing modes +# precede them in the same MODES list. +# STUB_CURL_COUNTER_FILE a file this script increments (one line per +# call) so the test can also assert call counts. +set -u +counter_file="${STUB_CURL_COUNTER_FILE:?STUB_CURL_COUNTER_FILE not set}" +call_index="$(wc -l < "$counter_file" 2>/dev/null || echo 0)" +printf 'x\n' >> "$counter_file" + +modes="${STUB_CURL_MODES:-empty}" +IFS=',' read -r -a mode_array <<< "$modes" +mode_count=${#mode_array[@]} +if [ "$call_index" -ge "$mode_count" ]; then + mode="${mode_array[$((mode_count - 1))]}" +else + mode="${mode_array[$call_index]}" +fi + +case "$mode" in + fail) + exit 22 + ;; + empty) + printf '<?xml version="1.0" encoding="UTF-8"?><ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><Name>loki</Name><IsTruncated>false</IsTruncated></ListBucketResult>' + exit 0 + ;; + haskey) + printf '<?xml version="1.0" encoding="UTF-8"?><ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><IsTruncated>false</IsTruncated><Contents><Key>index/index/19999/1700000000-uploaderA-abc.tsdb.gz</Key></Contents></ListBucketResult>' + exit 0 + ;; + truncated) + printf '<?xml version="1.0" encoding="UTF-8"?><ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><IsTruncated>true</IsTruncated><Contents><Key>index/index/19999/1700000000-uploaderA-abc.tsdb.gz</Key></Contents></ListBucketResult>' + exit 0 + ;; + malformed) + printf '<html><body>502 Bad Gateway</body></html>' + exit 0 + ;; + code200) + printf '200' + exit 0 + ;; + *) + echo "stub curl: unknown STUB_CURL_MODES entry '$mode'" >&2 + exit 99 + ;; +esac diff --git a/runner/pipeline/fixtures/stub-bin/docker b/runner/pipeline/fixtures/stub-bin/docker new file mode 100755 index 0000000000..0a76fe41bf --- /dev/null +++ b/runner/pipeline/fixtures/stub-bin/docker @@ -0,0 +1,35 @@ +#!/bin/bash +# A minimal `docker` stand-in for pipeline/dev-script.test.mjs's CLI-level +# tests. Understands `docker info` (dev-lib.mjs#isDockerAvailable), +# `docker image inspect <ref>` and `docker pull <ref>` (dev-lib.mjs's +# container base-image pre-pull gate — isImagePresent/pullImageWithRetry) — +# anything else exits 1 unconditionally, since no test here should ever get +# that far (each CLI test is designed to have observed everything it needs +# before reaching an unstubbed subcommand). +set -u +if [ "${1:-}" = "info" ]; then + if [ "${STUB_DOCKER_MODE:-ok}" = "fail" ]; then + echo "Cannot connect to the Docker daemon at unix:///var/run/docker.sock. Is the docker daemon running?" >&2 + exit 1 + fi + echo "stub docker info: ok" + exit 0 +fi +if [ "${1:-}" = "image" ] && [ "${2:-}" = "inspect" ]; then + if [ "${STUB_DOCKER_IMAGE_PRESENT:-1}" = "1" ]; then + echo "stub docker image inspect: present (${3:-})" + exit 0 + fi + echo "Error: No such image: ${3:-}" >&2 + exit 1 +fi +if [ "${1:-}" = "pull" ]; then + if [ "${STUB_DOCKER_PULL_MODE:-ok}" = "fail" ]; then + echo "Error response from daemon: Get \"https://registry-1.docker.io/v2/\": net/http: TLS handshake timeout" >&2 + exit 1 + fi + echo "stub docker pull: ok for ${2:-}" + exit 0 +fi +echo "stub docker: unsupported subcommand $*" >&2 +exit 1 diff --git a/runner/pipeline/fixtures/worker-harness.mjs b/runner/pipeline/fixtures/worker-harness.mjs index 3271604800..2b77733c4d 100644 --- a/runner/pipeline/fixtures/worker-harness.mjs +++ b/runner/pipeline/fixtures/worker-harness.mjs @@ -194,7 +194,10 @@ export function fakeR2(seed = {}) { async get(key) { const value = store.get(key); if (value === undefined) return null; - return { body: value, async text() { return value; } }; + // `size` mirrors a real R2Object's byte length — `share.ts#serveDemoAsset` + // reads it for the `serve.d`/`serve.embed` AE point's `bytes` column on a + // non-HTML asset, where the body is streamed rather than re-encoded. + return { body: value, size: Buffer.byteLength(String(value), "utf8"), async text() { return value; } }; }, async delete(key) { store.delete(key); diff --git a/runner/pipeline/fixtures/worker-hooks.mjs b/runner/pipeline/fixtures/worker-hooks.mjs index 43e10f3868..e6cd828ba7 100644 --- a/runner/pipeline/fixtures/worker-hooks.mjs +++ b/runner/pipeline/fixtures/worker-hooks.mjs @@ -23,9 +23,15 @@ // untestable. Additive only: every symbol used in workers/api/src passes // through or no-ops, so specs that assert nothing about Sentry are // unaffected. +// +// - `cloudflare:workers`: `index.ts` re-exports `O11yUsage` +// (`o11y-usage.ts`), a `WorkerEntrypoint` — reuses the same structural +// stub `o11y-worker-hooks.mjs` uses for `workers/o11y/src`, rather than +// a second copy. const SANDBOX_STUB = new URL("./cloudflare-sandbox-stub.mjs", import.meta.url).href; const SENTRY_STUB = new URL("./sentry-cloudflare-stub.mjs", import.meta.url).href; +const CLOUDFLARE_WORKERS_STUB = new URL("./o11y-cloudflare-workers-stub.mjs", import.meta.url).href; export async function resolve(specifier, context, nextResolve) { if (specifier === "@cloudflare/sandbox") { @@ -34,6 +40,9 @@ export async function resolve(specifier, context, nextResolve) { if (specifier === "@sentry/cloudflare") { return { url: SENTRY_STUB, shortCircuit: true }; } + if (specifier === "cloudflare:workers") { + return { url: CLOUDFLARE_WORKERS_STUB, shortCircuit: true }; + } if ( specifier.startsWith(".") && specifier.endsWith(".js") diff --git a/runner/pipeline/lite-beacon.test.mjs b/runner/pipeline/lite-beacon.test.mjs new file mode 100644 index 0000000000..61577752ab --- /dev/null +++ b/runner/pipeline/lite-beacon.test.mjs @@ -0,0 +1,901 @@ +// The lite beacon: the standalone ES5 reporter (`monitor.ts`'s +// `injectLiteReporterIntoHtml`, contract §9, ADR §C.5) and the o11y worker's +// `POST /telemetry/lite` route (`workers/o11y/src/lite.ts`). +// +// The reporter half is executed, not read (the DEV-2129 lesson every other +// ES5-reporter test in this repo follows) — a transpiler/output test that +// only inspects the string would pass over a script that cannot actually run. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { Parser } from "acorn"; +import { + LITE_CLIENT_MESSAGE_MAX, + LITE_CLIENT_STACK_MAX, + LITE_ENDPOINT, + LITE_REPORTER_MARKER, + LITE_REPORTER_MAX_BYTES, + LITE_VITALS_SAMPLE_RATE, + MONITOR_EVENT_CEILING, + injectLiteReporterIntoHtml, +} from "../packages/runtime/dist/monitor.js"; +import { isValidLitePayload, LITE_PAYLOAD_MAX_BYTES } from "../packages/runtime/dist/telemetry/index.js"; + +register("./fixtures/o11y-worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/o11y/src/index.ts"); +const { InboxWriter } = await import("../workers/o11y/src/inbox/writer.ts"); +const { hashRecord } = await import("../workers/o11y/src/normalise/hash.ts"); +const { makeEnv, ctx } = await import("./fixtures/o11y-harness.mjs"); +// Dynamic, not a static top-level import: `@handsontable/demo-runtime` only +// resolves through `o11y-worker-hooks.mjs`'s own `resolve()` hook (registered +// above via `register()`), and static imports are hoisted ahead of that +// call — the same reason every other borrowed-specifier import in this repo's +// `pipeline/*.test.mjs` files is `await import(...)` placed after `register()`, +// never a plain `import … from`. +const { AE_COLUMNS } = await import("@handsontable/demo-runtime/telemetry"); + +/** Reads a numeric metric field (`count`, etc.) out of a fake AE point via + * the real contract slot (`AE_COLUMNS.count`, e.g. `"double1"`) — the same + * helper `o11y-routes.test.mjs` uses, never a hardcoded array index. */ +function metricValue(point, name) { + const m = /^double(\d+)$/.exec(AE_COLUMNS[name]); + return point.doubles[Number(m[1]) - 1]; +} + +// ---- the reporter, built for a representative config -------------------------- + +const CONFIG = { surface: "d", demo: "r-react-18-0-0", ht: "18", fw: "react" }; + +function reporterScriptSource(config = CONFIG) { + const html = injectLiteReporterIntoHtml("<html><head></head><body></body></html>", config); + const match = /<script>([\s\S]*)<\/script>/.exec(html); + assert.ok(match, "the injector emitted no inline script"); + return match[1]; +} + +test("the injected reporter parses as ES5 (acorn ecmaVersion 5)", () => { + // Node's `new Function` accepts syntax an old runtime would reject + // (DEV-2129's own lesson, restated for this reporter) — acorn at + // `ecmaVersion: 5` is the real gate. + assert.doesNotThrow(() => Parser.parse(reporterScriptSource(), { ecmaVersion: 5 })); +}); + +test("the injected script stays inside its own size budget", () => { + for (const config of [ + CONFIG, + { surface: "embed", demo: "r-vanilla-typescript-18-1-1-a-fairly-long-id", ht: "next", fw: "vanilla-typescript" }, + ]) { + const bytes = Buffer.byteLength(reporterScriptSource(config), "utf8"); + assert.ok( + bytes <= LITE_REPORTER_MAX_BYTES, + `script for ${JSON.stringify(config)} is ${bytes} bytes, budget is ${LITE_REPORTER_MAX_BYTES}`, + ); + } +}); + +test("the marker survives injection and makes a second pass a no-op", () => { + const once = injectLiteReporterIntoHtml("<html><head></head><body></body></html>", CONFIG); + assert.ok(once.includes(LITE_REPORTER_MARKER)); + const twice = injectLiteReporterIntoHtml(once, CONFIG); + assert.equal(twice, once, "second injection returns the same string, unchanged"); +}); + +// ---- the reporter, executed ---------------------------------------------------- + +/** Stubs for every bare global the reporter references — passed as `new + * Function` parameters, exactly like `monitor-inject.test.mjs#runReporter` + * does for the framed reporter. In production every one of these resolves to + * the real global; here each is a plain object the test controls, which is + * what lets the sampling test fix `Math.random()` without touching the real, + * shared `Math` global. */ +function makeStubs(opts = {}) { + const sent = []; + const winListeners = new Map(); + const docListeners = new Map(); + const poRegistry = []; + + const window_ = { + __proto__: null, + addEventListener(type, cb) { + if (!winListeners.has(type)) winListeners.set(type, []); + winListeners.get(type).push(cb); + }, + fire(type, ev) { + for (const cb of winListeners.get(type) ?? []) cb(ev); + }, + }; + const document_ = { + visibilityState: "visible", + addEventListener(type, cb) { + if (!docListeners.has(type)) docListeners.set(type, []); + docListeners.get(type).push(cb); + }, + fire(type) { + for (const cb of docListeners.get(type) ?? []) cb(); + }, + }; + const navigator_ = { + userAgent: opts.userAgent ?? "Mozilla/5.0 (X11; Linux x86_64) Chrome/128.0 Safari/537.36", + sendBeacon(url, body) { + sent.push({ url, payload: JSON.parse(String(body)) }); + return true; + }, + }; + const navigationEntries = "navigationEntries" in opts ? opts.navigationEntries : [{ responseStart: 42 }]; + const performance_ = { + getEntriesByType(type) { + return type === "navigation" ? navigationEntries : []; + }, + }; + class FakePerformanceObserver { + constructor(cb) { + this.cb = cb; + this.type = null; + } + observe(config) { + this.type = config.type; + this.config = config; + poRegistry.push(this); + } + } + const Math_ = { random: opts.random ?? (() => 0.99) }; // unsampled by default + const Date_ = { now: opts.now ?? (() => 1700000000000) }; + + return { + sent, + window_, + document_, + navigator_, + performance_, + FakePerformanceObserver, + poRegistry, + Math_, + Date_, + /** Deliver entries to every observer registered for `type`, the way a + * real `PerformanceObserver` calls back incrementally. */ + fireEntries(type, entries) { + for (const o of poRegistry) if (o.type === type) o.cb({ getEntries: () => entries }); + }, + }; +} + +function runLite(config = CONFIG, opts = {}) { + const h = makeStubs(opts); + const source = reporterScriptSource(config); + // eslint-disable-next-line no-new-func + new Function("window", "document", "navigator", "performance", "PerformanceObserver", "Math", "Date", source)( + h.window_, + h.document_, + h.navigator_, + h.performance_, + h.FakePerformanceObserver, + h.Math_, + h.Date_, + ); + return h; +} + +test("an uncaught error produces one payload within the caps, matching the validator", () => { + const h = runLite(); + h.window_.fire("error", { error: Object.assign(new Error("boom"), { name: "TypeError" }) }); + assert.equal(h.sent.length, 1); + const [{ url, payload }] = h.sent; + assert.equal(url, LITE_ENDPOINT); + assert.equal(payload.t, "err"); + assert.equal(payload.n, "TypeError"); + assert.equal(payload.m, "boom"); + assert.equal(payload.val, null); + assert.equal(payload.s, CONFIG.surface); + assert.equal(payload.demo, CONFIG.demo); + assert.equal(payload.ht, CONFIG.ht); + assert.equal(payload.fw, CONFIG.fw); + assert.ok(["desktop", "mobile", "tablet"].includes(payload.dev)); + assert.ok(isValidLitePayload(payload), "the reporter's own payload must satisfy the ingest validator"); +}); + +// Two parallel page loads that throw in the same millisecond must not +// produce byte-identical beacons deduped as one. A per-beacon `id` in +// every `bc()` payload (`monitor.ts`'s `reporterSource`) pins that two +// beacons sent in the same fixed millisecond get different ids. +test("two beacons sent within the same millisecond get different ids", () => { + // The default `Math_` stub returns a fixed value on every call (needed so + // the vitals sampling tests can pin `Math.random()`), which would make + // every `id` in this test identical too — real randomness is what the + // reporter actually uses in production, so it is what this uniqueness + // property must be proven against. + const h = runLite(CONFIG, { now: () => 1700000000000, random: () => Math.random() }); + h.window_.fire("error", { error: new Error("first") }); + h.window_.fire("error", { error: new Error("second") }); + assert.equal(h.sent.length, 2); + assert.equal(h.sent[0].payload.ts, h.sent[1].payload.ts, "precondition: same millisecond"); + assert.notEqual(h.sent[0].payload.id, h.sent[1].payload.id, "same-millisecond beacons must get different ids"); + for (const { payload } of h.sent) { + assert.match(payload.id, /^[0-9a-z]{0,16}$/, "id must be lowercase base-36, at most 16 chars"); + assert.ok(isValidLitePayload(payload), "a payload carrying an id must still satisfy the ingest validator"); + } +}); + +test("an unhandled rejection is relayed the same way, with the reason's own name/message", () => { + const h = runLite(); + h.window_.fire("unhandledrejection", { reason: Object.assign(new RangeError("out of range"), {}) }); + assert.equal(h.sent.length, 1); + assert.equal(h.sent[0].payload.n, "RangeError"); + assert.equal(h.sent[0].payload.m, "out of range"); +}); + +test("message and stack are truncated well under the field caps before sending", () => { + const h = runLite(); + const longMessage = "x".repeat(5000); + const longStack = "at frame\n".repeat(600); + h.window_.fire("error", { error: Object.assign(new Error(longMessage), { stack: longStack }) }); + const { payload } = h.sent[0]; + // The caps are UTF-8 bytes, not `.length` (UTF-16 + // code units) — ASCII text happens to make the two numbers equal, which is + // exactly the coincidence that let a non-ASCII payload slip past a + // char count). `Buffer.byteLength` is the Node-side + // stand-in for the reporter's own `bl()`. + assert.ok(Buffer.byteLength(payload.m, "utf8") <= LITE_CLIENT_MESSAGE_MAX); + assert.ok(Buffer.byteLength(payload.st, "utf8") <= LITE_CLIENT_STACK_MAX); + assert.ok(isValidLitePayload(payload), "a maxed-out error must still fit the 2 KB payload cap"); +}); + +test("a huge (1 MB) error message is trimmed in well under 50ms, not quadratic time", () => { + // Before the fix, `bt(s,n)` re-encoded the WHOLE string on every + // `slice(0,-1)` iteration — quadratic in string length. Measured against + // the exact pre-fix function: 10k chars ~100ms, 50k chars ~2.4s, ~40s + // projected at 200k. A demo throwing `new Error(hugeString)` (a message + // embedding a data dump or a large JSON value) would then freeze the + // main thread of the `/d`/`/embed` host page synchronously, in the + // capturing `error` listener — exactly the "never harms the page it + // observes" rule this reporter exists to uphold. + const h = runLite(); + const hugeMessage = "x".repeat(1_000_000); + const start = performance.now(); + h.window_.fire("error", { error: Object.assign(new Error(hugeMessage), { name: "TypeError" }) }); + const elapsedMs = performance.now() - start; + assert.ok(elapsedMs < 50, `trimming a 1 MB message took ${elapsedMs}ms, expected well under 50ms`); + assert.equal(h.sent.length, 1); + const { payload } = h.sent[0]; + assert.ok(Buffer.byteLength(payload.m, "utf8") <= LITE_CLIENT_MESSAGE_MAX); + assert.ok(isValidLitePayload(payload)); +}); + +test("a non-ASCII error message/stack is byte-trimmed, never silently dropped for being over budget", () => { + // Before the fix, `tc()` truncated by `.length` (UTF-16 code units): a + // message of mostly multi-byte characters truncated to `LITE_CLIENT_ + // MESSAGE_MAX` *characters* could still serialize to well over 2 KB of + // UTF-8, and the o11y route's own `isValidLitePayload` would then silently + // drop the whole beacon at ingest — reported as "sent" client-side, never + // actually stored. Chinese, Cyrillic, and an emoji together exercise 2-, + // 3- and 4-byte UTF-8 sequences in one message. + const h = runLite(); + const nonAsciiMessage = "网格渲染失败: не удалось отрисовать таблицу 😵‍💫 ".repeat(20); + const nonAsciiStack = "at 渲染函数 (файл.js:1:1)\n".repeat(60); + h.window_.fire("error", { + error: Object.assign(new Error(nonAsciiMessage), { name: "渲染Error", stack: nonAsciiStack }), + }); + assert.equal(h.sent.length, 1, "a non-ASCII error must still be sent, not silently swallowed"); + const { payload } = h.sent[0]; + assert.ok(Buffer.byteLength(payload.n, "utf8") <= 100, "name byte budget"); + assert.ok(Buffer.byteLength(payload.m, "utf8") <= LITE_CLIENT_MESSAGE_MAX, "message byte budget"); + assert.ok(Buffer.byteLength(payload.st, "utf8") <= LITE_CLIENT_STACK_MAX, "stack byte budget"); + const totalBytes = Buffer.byteLength(JSON.stringify(payload), "utf8"); + assert.ok(totalBytes <= 2048, `serialized payload is ${totalBytes} bytes, over the 2 KB contract cap`); + assert.ok(isValidLitePayload(payload), "must satisfy the ingest validator's own byte check"); + // No lone surrogate or truncated multi-byte sequence — a string that fails + // to round-trip through JSON is the tell for a truncation cut mid-character. + assert.doesNotThrow(() => JSON.parse(JSON.stringify(payload))); +}); + +test("worst-case JSON-escaping content (all quotes and backslashes) still fits the 2 KB cap or is dropped, never sent oversize", () => { + // JSON.stringify expands every `"`/`\` to two output characters — the one + // inflation a per-field *byte* budget on the raw string does not see. This + // is the adversarial case `bc()`'s own final serialized-length check exists + // for, catching what per-field trimming alone cannot. + const h = runLite(); + const adversarialMessage = '"\\'.repeat(400); + h.window_.fire("error", { error: Object.assign(new Error(adversarialMessage), { stack: adversarialMessage }) }); + if (h.sent.length === 1) { + const totalBytes = Buffer.byteLength(JSON.stringify(h.sent[0].payload), "utf8"); + assert.ok(totalBytes <= 2048, `serialized payload is ${totalBytes} bytes, over the 2 KB contract cap`); + } + // Either it fit and was sent (and just got asserted above), or the final + // check refused to send it — both are correct; sending an oversize payload + // is the only wrong outcome, and that's what the assertion above would + // have caught. +}); + +test("errors stop at the monitor ceiling, never more", () => { + const h = runLite(); + for (let i = 0; i < MONITOR_EVENT_CEILING + 10; i++) { + h.window_.fire("error", { error: new Error(`err ${i}`) }); + } + assert.equal(h.sent.length, MONITOR_EVENT_CEILING); +}); + +test("a resource-load failure (no Error, foreign target) is not relayed — §9 has no network kind", () => { + const h = runLite(); + h.window_.fire("error", { target: { tagName: "IMG" } }); + assert.equal(h.sent.length, 0); +}); + +test("device class reflects the user agent", () => { + const mobile = runLite(CONFIG, { userAgent: "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X)" }); + mobile.window_.fire("error", { error: new Error("x") }); + assert.equal(mobile.sent[0].payload.dev, "mobile"); + + const tablet = runLite(CONFIG, { userAgent: "Mozilla/5.0 (iPad; CPU OS 17_0 like Mac OS X)" }); + tablet.window_.fire("error", { error: new Error("x") }); + assert.equal(tablet.sent[0].payload.dev, "tablet"); +}); + +// ---- vitals: sampling, "once per page", LCP/CLS/INP/TTFB ----------------------- + +/** Simulate one page view: sampled iff `randomValue < LITE_VITALS_SAMPLE_RATE`. + * Returns the harness so the caller can drive PerformanceObserver deliveries + * and the hide/pagehide report trigger. */ +function simulatePage(randomValue) { + return runLite(CONFIG, { random: () => randomValue }); +} + +test("vitals are sampled at the contract rate, decided once per page", () => { + // 100 simulated page views, evenly spread across [0, 1) — deterministic: + // exactly the ones under LITE_VITALS_SAMPLE_RATE (0.1) sample. + const N = 100; + let sampledPages = 0; + for (let i = 0; i < N; i++) { + const h = simulatePage(i / N); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + const vitals = h.sent.filter((s) => s.payload.t === "vital"); + if (vitals.length > 0) sampledPages++; + } + const expected = Math.round(N * LITE_VITALS_SAMPLE_RATE); + assert.equal(sampledPages, expected, `expected ~${expected}/${N} page views to sample vitals`); +}); + +test("vitals report never more than once per page, even if hidden fires twice", () => { + const h = simulatePage(0); // sampled: 0 < 0.1 + h.fireEntries("layout-shift", [{ value: 0.1, hadRecentInput: false }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + h.document_.fire("visibilitychange"); // fires again — must be a no-op + h.window_.fire("pagehide"); + const clsBeacons = h.sent.filter((s) => s.payload.t === "vital" && s.payload.n === "CLS"); + assert.equal(clsBeacons.length, 1); +}); + +test("an unsampled page view sends no vitals at all, even with a hide event", () => { + const h = simulatePage(0.99); // unsampled + h.fireEntries("layout-shift", [{ value: 0.5, hadRecentInput: false }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + assert.equal(h.sent.filter((s) => s.payload.t === "vital").length, 0); +}); + +test("LCP reports the last candidate observed, not the first", () => { + const h = simulatePage(0); + h.fireEntries("largest-contentful-paint", [{ renderTime: 500 }]); + h.fireEntries("largest-contentful-paint", [{ renderTime: 500 }, { renderTime: 1200 }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + const lcp = h.sent.find((s) => s.payload.n === "LCP"); + assert.ok(lcp, "an LCP beacon must be sent"); + assert.equal(lcp.payload.val, 1200); +}); + +test("CLS sums every layout-shift entry without recent input, across deliveries", () => { + const h = simulatePage(0); + h.fireEntries("layout-shift", [{ value: 0.05, hadRecentInput: false }, { value: 0.02, hadRecentInput: true }]); + h.fireEntries("layout-shift", [{ value: 0.03, hadRecentInput: false }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + const cls = h.sent.find((s) => s.payload.n === "CLS"); + assert.ok(cls); + assert.ok(Math.abs(cls.payload.val - 0.08) < 1e-9, `expected ~0.08, got ${cls.payload.val}`); +}); + +test("INP approximation: the longest real-interaction event duration, ignoring interactionId 0", () => { + const h = simulatePage(0); + h.fireEntries("event", [ + { interactionId: 0, duration: 900 }, // not a real interaction — ignored + { interactionId: 7, duration: 120 }, + { interactionId: 8, duration: 260 }, + ]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + const inp = h.sent.find((s) => s.payload.n === "INP"); + assert.ok(inp); + assert.equal(inp.payload.val, 260); +}); + +test("no INP is sent when nothing crosses the interaction filter", () => { + const h = simulatePage(0); + h.fireEntries("event", [{ interactionId: 0, duration: 900 }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + assert.equal(h.sent.some((s) => s.payload.n === "INP"), false); +}); + +test("TTFB comes from the navigation entry's responseStart", () => { + const h = simulatePage(0); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + const ttfb = h.sent.find((s) => s.payload.n === "TTFB"); + assert.ok(ttfb); + assert.equal(ttfb.payload.val, 42); +}); + +test("every sent vital payload satisfies the ingest validator", () => { + const h = simulatePage(0); + h.fireEntries("layout-shift", [{ value: 0.01, hadRecentInput: false }]); + h.document_.visibilityState = "hidden"; + h.document_.fire("visibilitychange"); + assert.ok(h.sent.length > 0); + for (const { payload } of h.sent) assert.ok(isValidLitePayload(payload), JSON.stringify(payload)); +}); + +// ---- the o11y worker's POST /telemetry/lite route ------------------------------- +// +// Driven through the REAL router (`workers/o11y/src/index.ts`'s default +// export), the same `o11y-routes.test.mjs` pattern — status codes and the +// stored shape are proven through the actual route, never a re-declared copy. + +function litePayload(overrides = {}) { + return { + v: 1, + t: "err", + s: "d", + demo: "abc12345", + ht: "18", + fw: "react", + n: "TypeError", + m: "grid.render is not a function", + val: null, + dev: "desktop", + ts: Date.now(), + ...overrides, + }; +} + +function freshEnv() { + return makeEnv(InboxWriter); +} + +function liteRequest(body, headers = {}) { + return new Request("https://demos.handsontable.com/telemetry/lite", { + method: "POST", + headers: { Origin: "https://demos.handsontable.com", "content-type": "application/json", ...headers }, + body: typeof body === "string" ? body : JSON.stringify(body), + }); +} + +test("POST /telemetry/lite: an accepted error beacon answers 2xx and lands in the browser tenant", async () => { + const { env, doStorage } = freshEnv(); + const res = await worker.fetch(liteRequest(litePayload()), env, ctx); + await ctx.drain(); + assert.ok(res.status >= 200 && res.status < 300, `expected 2xx, got ${res.status}`); + + const rowKeys = [...doStorage._data.keys()].filter((k) => k.startsWith("row:")); + assert.ok(rowKeys.length > 0, "a record must be pending in storage"); + const stored = rowKeys.map((k) => doStorage._data.get(k)).flatMap((row) => row.resourceLogs); + assert.equal(stored.length, 1); +}); + +// `checkBrowserGates` (`gates/browser.ts`) is the one gate function both +// `/telemetry/collect` (`index.ts#handleCollect`) and `/telemetry/lite` +// (`lite.ts`) call, keyed only by `cf-connecting-ip` — never by route. No +// route-level test proves that: a route-scoped rate limiter (e.g. one +// budget per path) would satisfy every per-route test unchanged. Driven +// through the real router on both routes, with a fake `RATE_LIMITER` that +// enforces one shared budget across whatever `key` it is called with — +// fails if either route starts keying its rate limit separately. +test("POST /telemetry/collect and POST /telemetry/lite share the same rate limiter (same key, one shared budget)", async () => { + const BUDGET = 2; + const calls = []; + // Budgeted per key (a `Map`), + // not with one process-wide counter — a global counter would still hit + // 429 on the BUDGET+1'th call even if `/telemetry/collect` and + // `/telemetry/lite` used two DIFFERENT (route-prefixed) keys, since it + // never actually checks which key is being spent. Per-key budgeting + // means the `liteRes` 429 assertion below can only pass if the two + // routes' calls landed on the SAME key's counter — the real behaviour + // this test exists to prove. + const usedByKey = new Map(); + const rateLimiter = { + async limit({ key }) { + calls.push(key); + const used = (usedByKey.get(key) ?? 0) + 1; + usedByKey.set(key, used); + return { success: used <= BUDGET }; + }, + }; + const { env } = makeEnv(InboxWriter, { env: { RATE_LIMITER: rateLimiter } }); + const ip = "203.0.113.7"; + + // Spend the whole shared budget on /telemetry/collect alone — a minimal, + // even structurally-invalid body is fine: the rate limit gate runs BEFORE + // any body is read (`gates/browser.ts#checkBrowserGates`, called first in + // both `handleCollect` and `lite.ts`). + const collectRequest = () => + new Request("https://demos.handsontable.com/telemetry/collect", { + method: "POST", + headers: { Origin: "https://demos.handsontable.com", "content-type": "application/json", "cf-connecting-ip": ip }, + body: "{}", + }); + for (let i = 0; i < BUDGET; i++) { + const res = await worker.fetch(collectRequest(), env, ctx); + await ctx.drain(); + assert.notEqual(res.status, 429, `/telemetry/collect call ${i + 1} of ${BUDGET} must still be within budget`); + } + + // The budget is now spent — a /telemetry/lite request from the SAME ip + // must be refused too, proving the two routes share the same counter, not + // two independent ones. + const liteRes = await worker.fetch(liteRequest(litePayload(), { "cf-connecting-ip": ip }), env, ctx); + await ctx.drain(); + assert.equal(liteRes.status, 429, "/telemetry/lite must be rate-limited once /telemetry/collect has spent the shared budget for this ip"); + + assert.equal(calls.length, BUDGET + 1); + assert.ok(calls.every((k) => k === ip), "both routes must call the rate limiter with the identical key"); +}); + +test("POST /telemetry/lite: the stored record's resourceLogs land under the browser tenant scope", async () => { + const { env, r2 } = freshEnv(); + await worker.fetch(liteRequest(litePayload()), env, ctx); + await ctx.drain(); + const inboxWriter = env.INBOX_WRITER.get(); + await inboxWriter.alarm(); + assert.equal(r2.objects.size, 1); + const [key] = [...r2.objects.keys()]; + assert.match(key, /^inbox\/browser\//, "the lite beacon is browser-tenant, never worker"); +}); + +test("POST /telemetry/lite: an oversize body is dropped with reason 'size'", async () => { + const { env, ae, doStorage } = freshEnv(); + const oversized = litePayload({ st: "x".repeat(3000) }); + const res = await worker.fetch(liteRequest(oversized), env, ctx); + await ctx.drain(); + assert.equal(res.status, 413); + const rowKeys = [...doStorage._data.keys()].filter((k) => k.startsWith("row:")); + assert.equal(rowKeys.length, 0, "an oversize beacon must never reach storage"); + const ingestPoints = ae.points.filter((p) => p.indexes[0] === "o11y.ingest"); + assert.ok(ingestPoints.some((p) => p.blobs[8] === "size"), "reason (blob9, index 8) must be 'size'"); +}); + +test("POST /telemetry/lite: a malformed payload is dropped with an o11y.ingest point, never stored", async () => { + const { env, ae, doStorage } = freshEnv(); + const res = await worker.fetch(liteRequest({ v: 1, t: "err" }), env, ctx); // missing required fields + await ctx.drain(); + assert.equal(res.status, 400); + const rowKeys = [...doStorage._data.keys()].filter((k) => k.startsWith("row:")); + assert.equal(rowKeys.length, 0); + const ingestPoints = ae.points.filter((p) => p.indexes[0] === "o11y.ingest"); + assert.ok(ingestPoints.some((p) => p.blobs[7] === "dropped"), "hot.outcome (blob8, index 7) must be 'dropped'"); +}); + +test("POST /telemetry/lite: not-quite-JSON is a clean 400, not a 500", async () => { + const { env } = freshEnv(); + const res = await worker.fetch(liteRequest("{not json", {}), env, ctx); + await ctx.drain(); + assert.equal(res.status, 400); +}); + +test("POST /telemetry/lite: the stored timestamp is the beacon's ts, clamped to the receive time", async () => { + const { env, r2 } = freshEnv(); + // Ten minutes in the past — outside ADR §C.2's ±5-minute clamp window. + const farPast = Date.now() - 10 * 60 * 1000; + await worker.fetch(liteRequest(litePayload({ ts: farPast })), env, ctx); + await ctx.drain(); + const inboxWriter = env.INBOX_WRITER.get(); + const before = Date.now(); + await inboxWriter.alarm(); + const [, bytes] = [...r2.objects.entries()][0]; + const text = await new Response(new Blob([bytes]).stream().pipeThrough(new DecompressionStream("gzip"))).text(); + const resourceLogs = JSON.parse(text.trim()); + const nano = BigInt(resourceLogs.scopeLogs[0].logRecords[0].timeUnixNano); + const storedMs = Number(nano / 1_000_000n); + assert.ok(Math.abs(storedMs - farPast) > 5000, "the far-past ts must not survive unclamped"); + assert.ok(storedMs >= before - 60_000, "the clamped timestamp must fall back to (near) the receive time"); +}); + +test("POST /telemetry/lite: an accepted web_vital beacon writes a web_vital AE point with the right blobs", async () => { + const { env, ae } = freshEnv(); + const vital = { + v: 1, + t: "vital", + s: "d", + demo: "abc12345", + ht: "18", + fw: "react", + n: "LCP", + val: 2500, + dev: "desktop", + ts: Date.now(), + }; + const res = await worker.fetch(liteRequest(vital), env, ctx); + await ctx.drain(); + assert.ok(res.status >= 200 && res.status < 300, `expected 2xx, got ${res.status}`); + const point = ae.points.find((p) => p.indexes[0] === "web_vital"); + assert.ok(point, "a web_vital point must be written"); + assert.equal(point.blobs[3], "d"); // hot.surface, blob4 + assert.equal(point.blobs[8], "LCP"); // reason, blob9 + assert.equal(point.blobs[11], "abc12345"); // demo_id, blob12 + assert.equal(point.doubles[2], 2500); // value, double3 +}); + +// `handleLite`'s accepted/duplicate counters (below) must stay +// unconditional, with no gate on whether the item carried a stored +// `record`. This pins that explicitly for the Observability-self +// dashboard: a first-time, AE-only web-vital beacon must still count as +// "accepted", not just its own web_vital point. +test("POST /telemetry/lite: a first-time web_vital beacon (AE-only) still writes an o11y.ingest accepted point", async () => { + const { env, ae } = freshEnv(); + const res = await worker.fetch(liteRequest(liteVitalPayload()), env, ctx); + await ctx.drain(); + assert.ok(res.status >= 200 && res.status < 300); + + const ingestAccepted = ae.points.find( + (p) => p.indexes[0] === "o11y.ingest" && p.blobs?.includes("lite") && p.blobs?.includes("accepted"), + ); + assert.ok(ingestAccepted, "an o11y.ingest accepted point must be written for a first-time AE-only vital beacon"); + assert.equal(metricValue(ingestAccepted, "count"), 1); +}); + +// ---- lite-beacon vitals ----------------------------------------------------- +// +// A Faro measurement is AE-only (`storeRecord = false`, normalise/faro.ts) +// — this route must not store a Loki record for a lite web-vital beacon +// from `/d`/`/embed`. Same pattern here: AE only, dedupe and accounting +// kept. An error beacon (`t: "err"`) is unaffected — it must still be +// stored. + +function liteVitalPayload(overrides = {}) { + return { + v: 1, + t: "vital", + s: "d", + demo: "abc12345", + ht: "18", + fw: "react", + n: "LCP", + val: 2500, + dev: "desktop", + ts: Date.now(), + ...overrides, + }; +} + +test("POST /telemetry/lite: an accepted web_vital beacon never reaches storage (AE-only) — zero pending rows, zero R2 objects", async () => { + const { env, doStorage, r2 } = freshEnv(); + const res = await worker.fetch(liteRequest(liteVitalPayload()), env, ctx); + await ctx.drain(); + assert.ok(res.status >= 200 && res.status < 300, `expected 2xx, got ${res.status}`); + + const rowKeys = [...doStorage._data.keys()].filter((k) => k.startsWith("row:")); + assert.equal(rowKeys.length, 0, "a web-vital beacon must never leave a pending row in storage"); + + const inboxWriter = env.INBOX_WRITER.get(); + await inboxWriter.alarm(); + assert.equal(r2.objects.size, 0, "a web-vital beacon must never produce a stored R2 object"); +}); + +test("POST /telemetry/lite: an error beacon is unaffected — it still lands in the browser tenant", async () => { + const { env, r2 } = freshEnv(); + await worker.fetch(liteRequest(litePayload()), env, ctx); // t: "err" (litePayload's default) + await ctx.drain(); + const inboxWriter = env.INBOX_WRITER.get(); + await inboxWriter.alarm(); + assert.equal(r2.objects.size, 1, "an error beacon must still be stored"); +}); + +test("POST /telemetry/lite: a duplicated web_vital beacon (identical payload, redelivered) does not double-count its web_vital point", async () => { + const { env, ae } = freshEnv(); + const vital = liteVitalPayload(); + const first = await worker.fetch(liteRequest(vital), env, ctx); + await ctx.drain(); + assert.ok(first.status >= 200 && first.status < 300); + + const second = await worker.fetch(liteRequest(vital), env, ctx); + await ctx.drain(); + assert.ok(second.status >= 200 && second.status < 300, "a duplicate vital beacon must still answer 2xx"); + + const vitalPoints = ae.points.filter((p) => p.indexes[0] === "web_vital"); + assert.equal(vitalPoints.length, 1, "a redelivered vital beacon must write exactly one web_vital point, not two — dedupe must still work with no stored record"); + + const ingestPoints = ae.points.filter((p) => p.indexes[0] === "o11y.ingest"); + assert.ok( + ingestPoints.some((p) => p.blobs[7] === "duplicate"), + "the second delivery's o11y.ingest point must say duplicate", + ); +}); + +// Two parallel page loads that throw in the same millisecond produce +// byte-identical beacons (message and `ts` both came from one Date.now()). +// This pins what the hash already does for beacons that really differ: +// messages that differ only in digits share one fingerprint but must never +// share a dedupe hash, even with the same `ts`. The fingerprint normalises +// digits away; the hash must not. +test("POST /telemetry/lite: 15 beacons whose messages differ only in digits are 15 accepted records with one fingerprint", async () => { + const { env, ae } = freshEnv(); + const ts = Date.now(); + for (let i = 0; i < 15; i++) { + const res = await worker.fetch( + liteRequest(litePayload({ s: "embed", n: "Error", m: `R9 embed alert ${ts + i}`, ts })), + env, + ctx, + ); + await ctx.drain(); + assert.equal(res.status, 204); + } + const ingest = ae.points.filter((p) => p.indexes[0] === "o11y.ingest" && p.blobs.includes("lite")); + assert.deepEqual( + ingest.map((p) => p.blobs[7]), + Array(15).fill("accepted"), + "every distinct beacon must be accepted, none counted as a duplicate", + ); + const errors = ae.points.filter((p) => p.indexes[0] === "error.uncaught"); + assert.equal(errors.length, 15); + const fingerprints = new Set(errors.map((p) => p.blobs.find((b) => b.startsWith("embed:")))); + assert.equal(fingerprints.size, 1, "precondition: the messages normalise to one fingerprint"); +}); + +// Two beacons byte-identical including `ts` (two parallel page loads +// throwing in the same millisecond, or several throws in one synchronous +// pass) must not collapse to one record. `id` is the only field that +// differs between them. +test("POST /telemetry/lite: two beacons identical except id are both accepted", async () => { + const { env, ae } = freshEnv(); + const ts = Date.now(); + const base = litePayload({ s: "embed", n: "Error", m: "R9 embed alert", ts }); + + const first = await worker.fetch(liteRequest({ ...base, id: "aaaaaaaa" }), env, ctx); + await ctx.drain(); + const second = await worker.fetch(liteRequest({ ...base, id: "bbbbbbbb" }), env, ctx); + await ctx.drain(); + assert.ok(first.status >= 200 && first.status < 300); + assert.ok(second.status >= 200 && second.status < 300); + + const ingest = ae.points.filter((p) => p.indexes[0] === "o11y.ingest" && p.blobs.includes("lite")); + assert.deepEqual( + ingest.map((p) => p.blobs[7]), + ["accepted", "accepted"], + "byte-identical-but-for-id beacons must both be accepted, not deduped", + ); + const errors = ae.points.filter((p) => p.indexes[0] === "error.uncaught"); + assert.equal(errors.length, 2, "each accepted beacon must write its own error.uncaught point"); +}); + +test("POST /telemetry/lite: the same beacon (same id) posted twice is 1 accepted + 1 duplicate", async () => { + const { env, ae } = freshEnv(); + const payload = litePayload({ s: "embed", n: "Error", m: "R9 embed alert", id: "cccccccc" }); + + const first = await worker.fetch(liteRequest(payload), env, ctx); + await ctx.drain(); + const second = await worker.fetch(liteRequest(payload), env, ctx); + await ctx.drain(); + assert.ok(first.status >= 200 && first.status < 300); + assert.ok(second.status >= 200 && second.status < 300, "a duplicate beacon must still answer 2xx"); + + const ingest = ae.points.filter((p) => p.indexes[0] === "o11y.ingest" && p.blobs.includes("lite")); + assert.deepEqual( + ingest.map((p) => p.blobs[7]), + ["accepted", "duplicate"], + "a redelivered beacon carrying the same id must still dedupe", + ); + const errors = ae.points.filter((p) => p.indexes[0] === "error.uncaught"); + assert.equal(errors.length, 1, "the duplicate must not write a second error.uncaught point"); +}); + +// The conditional spread in `lite.ts` (`...(body.id !== undefined ? { +// extra: { beacon_id: body.id } } : {})`) must leave the dedupe hash of an +// id-less beacon byte-for-byte unchanged — an id-less beacon only ever +// comes from an old, already-cached `/d`/`/embed` reporter that cannot be +// made to send one. Driven through the real route (`worker.fetch`), not a +// hand-built `PreHashRecord`, so this actually exercises the conditional +// spread in `lite.ts` rather than `hashRecord` in isolation. The hash +// itself is never in the HTTP response, so this reads it back from the +// `hash:<yyyymmdd>:<sha256>` key `InboxWriter`'s dedupe bucket writes +// (`inbox/dedupe.ts#bucketedHashKey`). +// +// To regenerate the pinned literal: run this exact request (fixed payload, +// fixed `ts`, no `id`) through `workers/o11y/src/lite.ts` and capture the +// hash it produces. +test("the dedupe hash of an id-less beacon matches its prior literal value exactly (guards the conditional spread)", async () => { + const { env, doStorage } = freshEnv(); + const payload = litePayload({ ts: 1700000000000 }); // litePayload()'s own defaults carry no `id` field at all + assert.equal("id" in payload, false, "precondition: the request carries no id field"); + + const res = await worker.fetch(liteRequest(payload), env, ctx); + await ctx.drain(); + assert.equal(res.status, 204); + + const hashKeys = [...doStorage._data.keys()].filter((k) => k.startsWith("hash:")); + assert.equal(hashKeys.length, 1, "exactly one dedupe hash bucket entry must be written"); + const match = /^hash:\d{8}:([0-9a-f]{64})$/.exec(hashKeys[0]); + assert.ok(match, `unexpected hash key shape: ${hashKeys[0]}`); + assert.equal( + match[1], + "9b995c41f1b322b45ea017d0321bb713e807b7c7e3e257c8620a1aaa82ee271e", + "an id-less beacon must hash exactly as it did before F32 — the conditional spread must add no key at all", + ); +}); + +// Unit-level companion to the route-level test above, on `hashRecord` +// directly: a record carrying the `extra.beacon_id` key `lite.ts` adds once +// `body.id !== undefined` must hash to something ELSE than the same record +// without it — otherwise the whole feature would be a no-op. +test("hashRecord: adding a beacon id to `extra` changes the hash", async () => { + const record = { + body: "TypeError: grid.render is not a function", + resourceAttributes: { "hot.surface": "d", "hot.demo_id": "abc12345" }, + attributes: {}, + rawEventTime: "1700000000000", + }; + const withoutId = await hashRecord(record); + const withId = await hashRecord({ ...record, extra: { beacon_id: "aaaaaaaa" } }); + assert.notEqual(withId, withoutId, "adding a beacon id must change the hash"); +}); + +test("POST /telemetry/lite: a duplicated beacon (identical payload, redelivered) does not double-count its error.uncaught point", async () => { + const { env, ae } = freshEnv(); + const payload = litePayload(); + const first = await worker.fetch(liteRequest(payload), env, ctx); + await ctx.drain(); + assert.ok(first.status >= 200 && first.status < 300); + + const second = await worker.fetch(liteRequest(payload), env, ctx); + await ctx.drain(); + assert.ok(second.status >= 200 && second.status < 300, "a duplicate beacon must still answer 2xx"); + + const errorPoints = ae.points.filter((p) => p.indexes[0] === "error.uncaught"); + assert.equal(errorPoints.length, 1, "a redelivered beacon must write exactly one error.uncaught point, not two"); + + const ingestPoints = ae.points.filter((p) => p.indexes[0] === "o11y.ingest"); + assert.ok( + ingestPoints.some((p) => p.blobs[7] === "duplicate"), + "the second delivery's o11y.ingest point must say duplicate, matching the (now correct) single error.uncaught count", + ); +}); + +test("POST /telemetry/lite: an accepted error beacon writes an error.uncaught point with a fingerprint", async () => { + const { env, ae } = freshEnv(); + await worker.fetch(liteRequest(litePayload()), env, ctx); + await ctx.drain(); + const point = ae.points.find((p) => p.indexes[0] === "error.uncaught"); + assert.ok(point); + assert.equal(point.blobs[3], "d"); // hot.surface + assert.ok(point.blobs[10].startsWith("d:"), "fingerprint (blob11) is '<surface>:<hash>'"); // index 10 + assert.equal(point.blobs[11], "abc12345"); // demo_id +}); + +test("POST /telemetry/lite: the fingerprint ignores the stack — two rebuilds with the same n/m but a different hashed chunk in the stack must collapse to one fingerprint", async () => { + // Folding the stack into the fingerprint would + // mint a "new" fingerprint on every rebuild that shifts a chunk hash or line + // number in the first frame — the same DEV-2853 ladder problem + // `normalizeMonitorMessage` exists to collapse for the framed reporter, and + // `d`/`embed` surfaces (unlike `demo-runtime`) feed the new-fingerprint + // alert, so a false "new" here would page someone every deploy. + const { env: envA, ae: aeA } = freshEnv(); + const { env: envB, ae: aeB } = freshEnv(); + await worker.fetch( + liteRequest(litePayload({ st: "at Grid.render (chunk-abc123.js:10:4)" })), + envA, + ctx, + ); + await ctx.drain(); + await worker.fetch( + liteRequest(litePayload({ st: "at Grid.render (chunk-def456.js:12:9)" })), + envB, + ctx, + ); + await ctx.drain(); + const fpA = aeA.points.find((p) => p.indexes[0] === "error.uncaught").blobs[10]; + const fpB = aeB.points.find((p) => p.indexes[0] === "error.uncaught").blobs[10]; + assert.equal(fpA, fpB, "same n/m, different stack chunk hash — must be the same fingerprint"); +}); diff --git a/runner/pipeline/lite-inject.test.mjs b/runner/pipeline/lite-inject.test.mjs new file mode 100644 index 0000000000..a4c807c9f8 --- /dev/null +++ b/runner/pipeline/lite-inject.test.mjs @@ -0,0 +1,232 @@ +// Injecting the lite reporter into `/d`/`/embed` HTML documents +// (`workers/api/src/monitor-inject.ts#injectLiteHtml`, `htMajorFromVersion`) +// and `workers/api/src/share.ts#serveOutcome`. Companion to +// `pipeline/lite-beacon.test.mjs`, which covers the reporter's own +// behaviour and the o11y ingest route. +// +// DEV-2580 (the same rule `inject-html.test.mjs`/`monitor-inject.test.mjs` +// pin for the framed reporter and the scheme receiver): a document an SSR +// framework already rendered must come out of injection with its own +// head/body markup byte-identical, because a strict hydrator (Remix's +// `hydrateRoot(document, …)` on React 18) throws away the whole document +// over one unexpected `<head>` child. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { LITE_REPORTER_MARKER, injectLiteReporterIntoHtml } from "../packages/runtime/dist/monitor.js"; +import { demoRow, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { htMajorFromVersion, injectLiteHtml } = await import("../workers/api/src/monitor-inject.ts"); +const { serveDemoAsset, serveOutcome } = await import("../workers/api/src/share.ts"); + +const CONFIG = { surface: "d", demo: "r-react-18-0-0", ht: "18", fw: "react" }; +const HTML = `<!doctype html> +<html> + <head><title>demo +
+ +`; + +// ---- htMajorFromVersion --------------------------------------------------------- + +test("htMajorFromVersion: an exact release maps to its major", () => { + assert.equal(htMajorFromVersion("18.1.1"), "18"); + assert.equal(htMajorFromVersion("15.0.0"), "15"); + assert.equal(htMajorFromVersion("19.2.3"), "19"); +}); + +test("htMajorFromVersion: a next-channel build maps to 'next', not 'none' or '0'", () => { + // The nightly shape — major is always 0 under plain semver parsing. + assert.equal(htMajorFromVersion("0.0.0-next-64139ae-20260219"), "next"); + // The dotted prerelease shape. + assert.equal(htMajorFromVersion("19.0.0-next.1"), "next"); +}); + +test("htMajorFromVersion: an unrecognised, empty, or out-of-range ref is 'none'", () => { + assert.equal(htMajorFromVersion(""), "none"); + assert.equal(htMajorFromVersion(null), "none"); + assert.equal(htMajorFromVersion(undefined), "none"); + assert.equal(htMajorFromVersion("latest"), "none"); // the pre-DEV-2565 sentinel, defensively + assert.equal(htMajorFromVersion("14.9.0"), "none"); // below DEFAULT_MIN_MAJOR + assert.equal(htMajorFromVersion("20.0.0"), "none"); // above DEFAULT_MAX_MAJOR +}); + +// ---- injectLiteHtml: the monitor-inject.ts guard -------------------------------- + +test("injects into an HTML document", () => { + const out = injectLiteHtml(HTML, "text/html; charset=utf-8", null, CONFIG); + assert.ok(out.includes(LITE_REPORTER_MARKER)); + assert.ok(out.includes("demo"), "the original document survives"); +}); + +test("only text/html is rewritten — a non-HTML content type passes through untouched", () => { + const js = "export const a = 1;"; + assert.equal(injectLiteHtml(js, "application/javascript", null, CONFIG), js); + assert.equal(injectLiteHtml(HTML, "application/json", null, CONFIG), HTML); +}); + +test("an encoded body passes through untouched — decoding to inject would risk corrupting it", () => { + // `serveDemoAsset`'s own R2 puts never set a `Content-Encoding` (confirmed: + // none of `share.ts`'s `ARTIFACTS.put` calls do), so its call site always + // passes `null` — but the guard is exercised here directly, the same + // defence-in-depth `injectMonitor` keeps for an arbitrary Tier-2 proxy + // response that could carry one. `identity` is the one encoding value that + // still means "plain text" and must still be injected. + assert.equal(injectLiteHtml(HTML, "text/html", "gzip", CONFIG).includes(LITE_REPORTER_MARKER), false); + assert.equal(injectLiteHtml(HTML, "text/html", "br", CONFIG).includes(LITE_REPORTER_MARKER), false); + assert.equal(injectLiteHtml(HTML, "text/html", "identity", CONFIG).includes(LITE_REPORTER_MARKER), true); + assert.equal(injectLiteHtml(HTML, "text/html", undefined, CONFIG).includes(LITE_REPORTER_MARKER), true); +}); + +test("a second pass is a no-op", () => { + const once = injectLiteHtml(HTML, "text/html", null, CONFIG); + const twice = injectLiteHtml(once, "text/html", null, CONFIG); + assert.equal(twice, once); + assert.equal(once.split(LITE_REPORTER_MARKER).length, twice.split(LITE_REPORTER_MARKER).length); +}); + +test("the framed reporter's own marker is untouched — the two injectors coexist", () => { + // `/d`/`/embed` never receive the framed reporter in production, but this + // pins that nothing about the lite injector's marker check accidentally + // keys on `MONITOR_MESSAGE_TYPE` or otherwise interferes if it did. + const withLite = injectLiteHtml(HTML, "text/html", null, CONFIG); + assert.equal(withLite.includes(LITE_REPORTER_MARKER), true); +}); + +// ---- DEV-2580: self-removing tag, no whitespace, hydration markup intact ------- + +test("the injected tag removes itself and adds no whitespace to ", () => { + const out = injectLiteReporterIntoHtml(HTML, CONFIG); + assert.match(out, /]*> + +`; + const head = /]*>([\s\S]*?)<\/head>/.exec(remixDoc)[1]; + const body = /]*>([\s\S]*?)<\/body>/.exec(remixDoc)[1]; + + const injected = injectLiteReporterIntoHtml(remixDoc, CONFIG); + const injectedHead = /]*>([\s\S]*?)<\/head>/.exec(injected)[1]; + const injectedBody = /]*>([\s\S]*?)<\/body>/.exec(injected)[1]; + + // The reporter's own ', + "/index.js": BASE_SOURCE, +}; + +/** A runtime with a fake client attached, skipping the bundler (same shape as + * `head-assets.test.mjs#published`). */ +function mountedParcel() { + const runtime = new SandpackRuntime(ENTRY, { iframe: {} }); + const pushes = []; + runtime.client = { + updateSandbox: (setup) => pushes.push(setup), + destroy() {}, + listen: () => () => {}, + }; + runtime.files = { ...FILES }; + const compileErrors = []; + const errors = []; + runtime.onCompileError((e) => compileErrors.push(e)); + runtime.onError((e) => errors.push(e)); + return { runtime, pushes, compileErrors, errors }; +} + +/** Let the async transpile chain of the pushes so far settle. Babel is loaded + * once (below), after which a transpile is a handful of microtasks. */ +const settle = () => new Promise((resolve) => setTimeout(resolve, 20)); + +test.before(async () => { + // Warm the lazily-imported @babel/standalone so `settle()` never races its load. + await transpileFilesForParcel({ "/warm.js": "1;" }); +}); + +test("the parcel transpile failure is marked as such, and is not a CompilerUnavailableError", async () => { + const failure = await transpileFilesForParcel({ "/index.js": "const R9C = ;" }).then( + () => null, + (e) => e, + ); + assert.ok(failure instanceof Error); + assert.ok(isTranspileFailure(failure)); + // Sentry parity: still a plain `Error` with the same message shape, so the linked + // `cause` of the mount path's `Tier1CompileError` capture is byte-identical. + assert.equal(failure.name, "Error"); + assert.match(failure.message, /^Failed to transpile \/index\.js for the parcel sandbox: /); + assert.ok(!isTranspileFailure(new Error("Failed to transpile /x.js for the parcel sandbox: fake")), "the marker, not the text"); +}); + +test("edit path: a syntax error reports one compile error, pushes nothing, and raises no error card", async () => { + const { runtime, pushes, compileErrors, errors } = mountedParcel(); + + runtime.writeFile("/index.js", BASE_SOURCE + "const R9C = ;\n"); + await settle(); + + assert.equal(pushes.length, 0, "the broken source never reaches the bundler (last good render stays)"); + assert.equal(compileErrors.length, 1, "but it is the preview's compile error"); + assert.match(compileErrors[0].message, /Failed to transpile \/index\.js for the parcel sandbox/); + assert.equal(compileErrors[0].origin, "transpile", "nothing was dispatched"); + assert.equal(errors.length, 0, "no onError: the card and the Sentry capture are unchanged"); +}); + +test("edit path: only the newest push reports — a superseded keystroke's failure is typed past", async () => { + const { runtime, pushes, compileErrors } = mountedParcel(); + + // Two keystrokes before either transpile settles: the first is broken, the + // second finishes the line. + runtime.writeFile("/index.js", BASE_SOURCE + "const R9C = \n"); + runtime.writeFile("/index.js", BASE_SOURCE + "const R9C = 1;\n"); + await settle(); + + assert.equal(compileErrors.length, 0, "the stale failure must not be reported"); + assert.equal(pushes.length, 1, "the newest edit compiled and was pushed"); +}); + +test("edit path: a runtime SyntaxError (JSON.parse) is not a compile error — it parses, and is pushed", async () => { + const { runtime, pushes, compileErrors } = mountedParcel(); + + runtime.writeFile("/index.js", BASE_SOURCE + "JSON.parse('{');\n"); + await settle(); + + assert.equal(compileErrors.length, 0); + assert.equal(pushes.length, 1, "it runs, and whatever it throws is the in-preview reporter's"); +}); + +test("edit path: a missing entry mid-rename (DEV-2130) is not a compile error", async () => { + const { runtime, compileErrors, pushes } = mountedParcel(); + + runtime.deleteFile("/index.html"); // the parcel sandbox entry + await settle(); + + assert.equal(pushes.length, 0); + assert.equal(compileErrors.length, 0); +}); + +test("mount: a demo whose source does not parse reports its compile error at once, and the mount still rejects with the same error", async () => { + const runtime = new SandpackRuntime(ENTRY, { iframe: {} }); + const compileErrors = []; + runtime.onCompileError((e) => compileErrors.push(e)); + + const rejection = await runtime.mount({ ...FILES, "/index.js": "const R9C = ;\n" }).then( + () => null, + (e) => e, + ); + + assert.ok(isTranspileFailure(rejection), "the mount rejects with the transpile failure, unchanged"); + assert.equal(compileErrors.length, 1); + assert.equal(compileErrors[0].message, rejection.message, "same (bounded) diagnostic"); +}); + +// ---- the typed ladder, end to end through the metrics wiring and the collapse -- + +/** A hand-driven timer, as in `demo-event-collapse.test.mjs`. */ +function fakeTimers() { + let now = 0; + let nextId = 1; + const timers = new Map(); + return { + setTimer(fn, ms) { + const id = nextId++; + timers.set(id, { at: now + ms, fn }); + return id; + }, + clearTimer(id) { + timers.delete(id); + }, + advance(ms) { + now += ms; + for (const [id, t] of [...timers]) { + if (t.at <= now) { + timers.delete(id); + t.fn(); + } + } + }, + }; +} + +const SERVICE = { service_name: "demos-authoring", service_version: "abc123", environment: "production" }; + +function ladderHarness() { + const clock = fakeTimers(); + const telemetry = recordingTelemetry(); + const collapse = createDemoEventCollapse({ + emit: (emitItem) => emitItem(), + setTimer: clock.setTimer, + clearTimer: clock.clearTimer, + }); + // Mirrors `sentry.ts#collapseCompileError`. + const collapseCompileError = (emit, origin) => + collapse.report("compile:sandpack.compile_error", emit, { replacesRun: true, fromBundler: origin === "bundler" }); + // Mirrors `sentry.ts#reportDemoEventUnguarded` → `emitCollapsedDemoEvent`. + const relayRuntimeError = (message) => { + const fp = fingerprint("demo-runtime", message); + collapse.report(fp, () => + telemetry.metric( + "preview.runtime_error", + { count: 1 }, + { surface: "demo-runtime", tier: "1", framework: "javascript", ht_major: "18", reason: "uncaught", fingerprint: fp }, + ), + ); + }; + const { runtime, pushes } = mountedParcel(); + wireRuntimeMetrics(runtime, { framework: "javascript", versionRef: "18.0.0" }, telemetry, { collapseCompileError }); + // Mirrors `App.tsx` → `sentry.ts#noteDemoPushOutcome`. + runtime.onPushOutcome((outcome) => collapse.pushOutcome(outcome)); + const points = (name) => telemetry.metrics.filter((m) => m.name === name); + return { clock, telemetry, collapse, runtime, pushes, relayRuntimeError, points }; +} + +test("a typed syntax-error ladder yields exactly 1 sandpack.compile_error and 0 preview.runtime_error", async () => { + const { clock, telemetry, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + const line = "const R9C = ;"; + + let inFlight = null; + for (let i = 1; i <= line.length; i++) { + const prefix = line.slice(0, i); + collapse.noteEdit(); // App.tsx#writeFile, on every non-quiet keystroke + // The previous keystroke's run relays its throw only now: compile slower + // than the typist (the case the collapse cannot see through on its own). + if (inFlight) relayRuntimeError(inFlight); + inFlight = null; + const before = pushes.length; + runtime.writeFile("/index.js", BASE_SOURCE + prefix + "\n"); + await settle(); + // `c`, `co`, `con`, `cons` parse and are pushed; that run throws a ReferenceError. + if (pushes.length > before) inFlight = `${prefix} is not defined`; + } + assert.equal(pushes.length, 4, "guard: the four identifier prefixes really ran, so the ladder has runtime rungs"); + clock.advance(DEMO_EDIT_SETTLE_MS); + + assert.equal(points("sandpack.compile_error").length, 1, "one compile error for the typed line"); + assert.equal(points("preview.runtime_error").length, 0, "and no runtime error from the rungs it was typed through"); + for (const { name, values, attrs } of telemetry.metrics) toAePoint(name, values, { ...SERVICE, ...attrs }); + const [point] = points("sandpack.compile_error"); + assert.deepEqual(Object.keys(point.attrs).sort(), ["fingerprint", "framework", "ht_major"]); + assert.equal(point.attrs.framework, "javascript"); + assert.equal(point.attrs.ht_major, "18"); +}); + +test("a first-load compile failure counts immediately, with no edit burst to wait for", async () => { + const clock = fakeTimers(); + const telemetry = recordingTelemetry(); + const collapse = createDemoEventCollapse({ emit: (f) => f(), setTimer: clock.setTimer, clearTimer: clock.clearTimer }); + const runtime = new SandpackRuntime(ENTRY, { iframe: {} }); + wireRuntimeMetrics(runtime, { framework: "javascript", versionRef: "18.0.0" }, telemetry, { + collapseCompileError: (emit) => collapse.report("compile:sandpack.compile_error", emit, { replacesRun: true }), + }); + + await runtime.mount({ ...FILES, "/index.js": "const R9C = ;\n" }).catch(() => {}); + + assert.equal(telemetry.metrics.filter((m) => m.name === "sandpack.compile_error").length, 1, "no clock advanced, already counted"); +}); + +test("a runtime JSON.parse SyntaxError stays a preview.runtime_error, never a compile error", async () => { + const { clock, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + + collapse.noteEdit(); + runtime.writeFile("/index.js", BASE_SOURCE + "JSON.parse('{');\n"); + await settle(); + assert.equal(pushes.length, 1, "guard: it compiled and ran"); + relayRuntimeError("SyntaxError: Expected property name or '}' in JSON at position 1"); + clock.advance(DEMO_EDIT_SETTLE_MS); + + assert.equal(points("preview.runtime_error").length, 1); + assert.equal(points("sandpack.compile_error").length, 0); +}); + +// ---- a throwing line typed key by key, as the code editor produces it ------- + +/** The documents CodeMirror's `closeBrackets` produces while `line` is typed + * one key at a time: an opener inserts its closer, and typing the closer + * that is already next steps over it without changing the document. */ +function typedDocuments(line) { + const PAIRS = { "(": ")", "[": "]", "{": "}", "'": "'", '"': '"' }; + const CLOSERS = new Set([")", "]", "}", "'", '"']); + let doc = ""; + let cursor = 0; + const docs = []; + for (const ch of line) { + const next = doc[cursor]; + if (CLOSERS.has(ch) && next === ch) { + cursor += 1; // steps over: no edit reaches the app + continue; + } + const insert = PAIRS[ch] && (next === undefined || /[\s)\]};:>]/.test(next)) ? ch + PAIRS[ch] : ch; + doc = doc.slice(0, cursor) + insert + doc.slice(cursor); + cursor += 1; + docs.push(doc); + } + assert.equal(doc, line, "guard: the typed document ends as the line itself"); + return docs; +} + +/** What the preview relays for a pushed sandbox: the pushed module is run + * (timers fire at once) and its uncaught throw, if any, is the relay. */ +function runPushed(setup) { + const code = setup.files["/index.js"].code; + const sandbox = { setTimeout: (fn) => typeof fn === "function" && fn(), JSON, Error, console: { log() {} } }; + try { + vm.runInNewContext(code, sandbox); + return null; + } catch (e) { + return `Uncaught ${e.name}: ${e.message}`; + } +} + +const TYPED_THROWS = [ + ["setTimeout(() => { throw new Error('typed runtime'); }, 100);", "Uncaught Error: typed runtime"], + ['setTimeout(() => JSON.parse("{typed"), 50);', `Uncaught SyntaxError: ${jsonParseMessage('{typed')}`], +]; + +function jsonParseMessage(text) { + try { + JSON.parse(text); + } catch (e) { + return e.message; + } + throw new Error("guard: expected a JSON.parse failure"); +} + +for (const [line, thrown] of TYPED_THROWS) { + test(`typed key by key, \`${line}\` counts its final run's error once, though the closing ';' re-runs nothing`, async () => { + const { clock, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + + for (const doc of typedDocuments(line)) { + const before = pushes.length; + runtime.writeFile("/index.js", BASE_SOURCE + doc + "\n"); + collapse.noteEdit(); // App.tsx#writeFile, right after the runtime write + await settle(); + // Each run starts and relays before the next keystroke (the typist is slower than the compile). + if (pushes.length > before) { + runtime.onMessage({ type: "start" }); + const relayed = runPushed(pushes.at(-1)); + if (relayed) relayRuntimeError(relayed); + } + } + assert.ok(pushes.length > 3, "guard: prefixes of the line ran"); + assert.equal(runPushed(pushes.at(-1)), thrown, "guard: the last run is the finished line's"); + clock.advance(DEMO_EDIT_SETTLE_MS); + + const runtimeErrors = points("preview.runtime_error"); + assert.equal(runtimeErrors.length, 1, "the error the finished line throws, once"); + assert.equal(runtimeErrors[0].attrs.fingerprint, fingerprint("demo-runtime", thrown)); + assert.equal(points("sandpack.compile_error").length, 0, "the finished line compiles"); + }); +} + +/** What the preview relays for a pushed sandbox's `console.error` calls, joined as + * the monitor joins its arguments. */ +function consoleErrorOf(setup) { + const logged = []; + const console = { log() {}, error: (...args) => logged.push(args.join(" ")) }; + try { + vm.runInNewContext(setup.files["/index.js"].code, { console }); + } catch { + return null; + } + return logged[0] ?? null; +} + +test("a console.error line typed key by key counts once when each run relays only after the next edit is dispatched", async () => { + const { clock, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + const line = "console.error('typed err');"; + // The bundler evaluates one compile at a time: a run's relay reaches the page after + // the next edit has been dispatched, and before the bundler's `start` for that edit. + let late = null; + for (const doc of typedDocuments(line)) { + const before = pushes.length; + runtime.writeFile("/index.js", BASE_SOURCE + doc + "\n"); + collapse.noteEdit(); + await settle(); + if (late) relayRuntimeError(late); + late = null; + if (pushes.length > before) { + runtime.onMessage({ type: "start" }); + late = consoleErrorOf(pushes.at(-1)); + } + } + if (late) relayRuntimeError(late); + assert.equal(consoleErrorOf(pushes.at(-1)), "typed err", "guard: the last run is the finished line's"); + assert.ok(pushes.length > 3, "guard: prefixes of the line ran"); + clock.advance(DEMO_EDIT_SETTLE_MS); + + const runtimeErrors = points("preview.runtime_error"); + assert.deepEqual( + runtimeErrors.map((p) => p.attrs.fingerprint), + [fingerprint("demo-runtime", "typed err")], + "the finished line's message, once, and no earlier rung", + ); +}); + +/** The bundler's frameless `show-error` for a pushed sandbox it cannot build. */ +function bundlerRejects(runtime, message) { + runtime.onMessage({ type: "action", action: "show-error", message, payload: {} }); +} + +for (const mode of ["typed", "pasted"]) { + test(`a bundler compile error of the running sandbox counts once when ${mode}, though the closing ';' re-runs nothing`, async () => { + const { clock, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + runtime.onError(() => {}); // the card; not what is measured here + const line = 'import "./missing.css";'; + const docs = mode === "typed" ? typedDocuments(line) : [line]; + + for (const doc of docs) { + const before = pushes.length; + runtime.writeFile("/index.js", BASE_SOURCE + doc + "\n"); + collapse.noteEdit(); + await settle(); + if (pushes.length === before) continue; + runtime.onMessage({ type: "start" }); + const code = pushes.at(-1).files["/index.js"].code; + const specifier = /import\s*"([^"]*)"/.exec(code)?.[1]; + if (specifier !== undefined) bundlerRejects(runtime, `ModuleNotFoundError: Could not find module in path: '${specifier}'`); + else { + const relayed = runPushed(pushes.at(-1)); + if (relayed) relayRuntimeError(relayed); + } + } + clock.advance(DEMO_EDIT_SETTLE_MS); + + assert.equal(points("sandpack.compile_error").length, 1, "the bundler's diagnostic for the finished line, once"); + assert.equal(points("preview.runtime_error").length, 0, "no rung of the line counts"); + }); +} + +test("an edit that transpiles to the running sandbox reports 'unchanged'; one that differs reports 'rerun' when the bundler starts it", async () => { + const { runtime, pushes } = mountedParcel(); + const outcomes = []; + runtime.onPushOutcome((o) => outcomes.push(o)); + + runtime.writeFile("/index.js", BASE_SOURCE + "f(1)\n"); + await settle(); + assert.deepEqual(outcomes, [], "dispatched, but the bundler has not started it yet"); + runtime.onMessage({ type: "start" }); + runtime.writeFile("/index.js", BASE_SOURCE + "f(1);\n"); + await settle(); + runtime.writeFile("/index.js", BASE_SOURCE + "f(1, \n"); // does not parse: no outcome + await settle(); + runtime.writeFile("/index.js", BASE_SOURCE + "f(1)\n"); // superseded before its transpile settles: no outcome + runtime.writeFile("/index.js", BASE_SOURCE + "f(2)\n"); + await settle(); + runtime.onMessage({ type: "start" }); + + assert.deepEqual(outcomes, ["rerun", "unchanged", "rerun"]); + assert.equal(pushes.length, 2); +}); + +test("a report in a burst whose pushed run never starts (a stalled bundler) is still counted once", async () => { + const { clock, collapse, runtime, pushes, relayRuntimeError, points } = ladderHarness(); + runtime.writeFile("/index.js", BASE_SOURCE + "console.error('stalled');\n"); + collapse.noteEdit(); + await settle(); + assert.equal(pushes.length, 1, "guard: the push was dispatched"); + relayRuntimeError("stalled"); // the running sandbox's report; the bundler never posts `start` + clock.advance(DEMO_EDIT_SETTLE_MS); + clock.advance(DEMO_EDIT_SETTLE_MS); + assert.deepEqual(points("preview.runtime_error").map((p) => p.attrs.fingerprint), [fingerprint("demo-runtime", "stalled")]); +}); + +test("a bundler start that no push asked for (the mount's own compile) is not a rerun", async () => { + const { runtime, pushes } = mountedParcel(); + const outcomes = []; + runtime.onPushOutcome((o) => outcomes.push(o)); + runtime.onMessage({ type: "start" }); + assert.deepEqual(outcomes, []); + runtime.writeFile("/index.js", BASE_SOURCE + "f(1)\n"); + await settle(); + runtime.onMessage({ type: "start" }); + runtime.onMessage({ type: "start" }); + assert.equal(pushes.length, 1); + assert.deepEqual(outcomes, ["rerun"], "one push, one rerun"); +}); + +test("each dispatched run re-arms the in-preview reporter first; an unchanged or failed push does not", async () => { + const order = []; + const runtime = new SandpackRuntime(ENTRY, { + iframe: { contentWindow: { postMessage: (data) => order.push(["preview", data]) } }, + monitor: true, + }); + runtime.client = { updateSandbox: () => order.push(["compile"]), destroy() {}, listen: () => () => {} }; + runtime.files = { ...FILES }; + + runtime.writeFile("/index.js", BASE_SOURCE + "f(1)\n"); + await settle(); + runtime.writeFile("/index.js", BASE_SOURCE + "f(1);\n"); // unchanged + await settle(); + await runtime.reload(); // the refresh button: a real run + runtime.writeFile("/index.js", BASE_SOURCE + "f(1, \n"); // does not parse + await settle(); + + const reset = ["preview", { type: MONITOR_MESSAGE_TYPE, reset: MONITOR_RESET }]; + assert.deepEqual(order, [reset, ["compile"], reset, ["compile"]]); +}); + +test("without the monitor injected, no reset is posted into the preview", async () => { + const posted = []; + const runtime = new SandpackRuntime(ENTRY, { iframe: { contentWindow: { postMessage: (d) => posted.push(d) } } }); + runtime.client = { updateSandbox() {}, destroy() {}, listen: () => () => {} }; + runtime.files = { ...FILES }; + runtime.writeFile("/index.js", BASE_SOURCE + "f(1)\n"); + await settle(); + assert.deepEqual(posted, []); +}); diff --git a/runner/pipeline/sandpack-reload.test.mjs b/runner/pipeline/sandpack-reload.test.mjs index 744a2ac0c0..a02ffccacb 100644 --- a/runner/pipeline/sandpack-reload.test.mjs +++ b/runner/pipeline/sandpack-reload.test.mjs @@ -579,3 +579,100 @@ test("an ordinary mid-edit transpile failure still reaches nobody", async () => assert.equal(client.pushes.length, 0); assert.deepEqual(runtime.published, injectSchemeReceiver({ ...FILES }, ENTRY.entry), "the last good sandbox stays published"); }); + +// --------------------------------------------------------------------------- +// The compile-timing hooks (`onCompileTiming`/`onCompileError`), driven +// against the real runtime rather than a fake (`pipeline/browser-metrics.test.mjs` +// covers `apps/authoring/src/telemetry/metrics.ts`'s own emission logic; this +// covers whether sandpack.ts's own hooks fire — once, paired to the right +// dispatch, and only for a real compile diagnostic). +// +// `onBundlerUnreachable` (the `loadSandpackClient` rejection inside `mount()`) is +// not covered here: `loadSandpackClient` is a direct top-level import, and +// Node's `node:test` module mocking needs +// `--experimental-test-module-mocks`, which `pnpm test`'s script does not +// pass. It is covered against a fake hook in +// `pipeline/browser-metrics.test.mjs` instead. + +test("onCompileTiming: an edit that reaches the bundler and comes back done reports ok, once", async () => { + const { runtime, client } = mounted(); + const events = []; + runtime.onCompileTiming((e) => events.push(e)); + + runtime.writeFile("/src/main.js", "console.log('edited');"); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(client.pushes.length, 1, "sanity: the edit must have dispatched"); + assert.equal(events.length, 0, "not resolved until the bundler answers"); + + runtime.onMessage({ type: "done" }); + + assert.equal(events.length, 1); + assert.equal(events[0].outcome, "ok"); + assert.ok(events[0].durationMs >= 0); + + // A `done` with nothing dispatched must not re-fire. + runtime.onMessage({ type: "done" }); + assert.equal(events.length, 1, "guard: an already-resolved compile clock must not re-emit"); +}); + +test("onCompileTiming: a no-change compile (the sameFiles skip) never dispatches, so it never times", async () => { + const { runtime } = mounted(); + const events = []; + runtime.onCompileTiming((e) => events.push(e)); + + // Byte-identical to what `mounted()` already published — the sameFiles skip. + runtime.writeFile("/src/main.js", FILES["/src/main.js"]); + await new Promise((resolve) => setImmediate(resolve)); + // Even if the bundler still answers something for an unrelated reason, there is + // no dispatch clock running for this handler to pair it with. + runtime.onMessage({ type: "done" }); + + assert.equal(events.length, 0, "guard: nothing was dispatched, so nothing may be timed"); +}); + +test("onCompileTiming + onCompileError: a show-error with no frames (a real compile diagnostic) reports error once and fires onCompileError", async () => { + const { runtime, client } = mounted(); + const timing = []; + const errors = []; + runtime.onCompileTiming((e) => timing.push(e)); + runtime.onCompileError((e) => errors.push(e)); + + runtime.writeFile("/src/main.js", "const broken = ("); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(client.pushes.length, 1); + + runtime.onMessage({ type: "action", action: "show-error", message: "SyntaxError: Unexpected token" }); + + assert.equal(timing.length, 1); + assert.equal(timing[0].outcome, "error"); + assert.equal(errors.length, 1); + assert.match(errors[0].message, /SyntaxError/); +}); + +test("onCompileTiming + onCompileError: an evaluation error (frames present) does not re-resolve an already-ok compile, and is not a compile_error", async () => { + const { runtime } = mounted(); + const timing = []; + const errors = []; + runtime.onCompileTiming((e) => timing.push(e)); + runtime.onCompileError((e) => errors.push(e)); + + runtime.writeFile("/src/main.js", "console.log('edited');"); + await new Promise((resolve) => setImmediate(resolve)); + + // The module compiled fine (done, ok) … + runtime.onMessage({ type: "done" }); + assert.equal(timing.length, 1); + assert.equal(timing[0].outcome, "ok"); + + // … and only then threw at runtime — DEV-2552's evaluation-error split, reported + // through the same show-error channel, with frames. + runtime.onMessage({ + type: "action", + action: "show-error", + message: "TypeError: x is not a function", + payload: { frames: [{}] }, + }); + + assert.equal(timing.length, 1, "guard: an evaluation error must not re-resolve an already-ok compile"); + assert.equal(errors.length, 0, "guard: sandpack.compile_error is for compile diagnostics, not runtime throws (T06's territory)"); +}); diff --git a/runner/pipeline/scrub-telemetry.test.mjs b/runner/pipeline/scrub-telemetry.test.mjs new file mode 100644 index 0000000000..4a6ca762bd --- /dev/null +++ b/runner/pipeline/scrub-telemetry.test.mjs @@ -0,0 +1,289 @@ +// Observability contract §3 / ADR §E.4 — `scrubTelemetry`, one case per rule. +// Every input is realistic (a real Babel code frame, a real Tier-2 preview +// host, a real Chrome user-agent, an OTLP-shaped record with `url.full` and +// geo), and each case is isolated to the one rule it proves: a query-string +// case runs on a field the attribute allowlist would never touch by itself +// (`meta.page.url`, not `attributes["url.full"]`), so removing the +// query-stripping rule alone is what turns that case red — not a different +// rule accidentally covering for it. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { scrubTelemetry, stripQueryAndFragment } from "../packages/runtime/dist/telemetry/index.js"; + +// A real Babel code-frame throw, captured from +// `babel.transform("const x = ;", { presets: ["env"] })` (see +// telemetry-fingerprint.test.mjs for the two-line variant). +const CODE_FRAME_MESSAGE = "unknown: Unexpected token (1:10)\n\n> 1 | const x = ;\n | ^"; + +// A real Tier-2 preview host shape (`--.demos.handsontable.com`). +const PREVIEW_HOST = "3000-sbx7f2a-tok9xQ.demos.handsontable.com"; + +// A real Chrome-on-Windows UA string. +const CHROME_UA = + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36"; + +function faroLog(overrides = {}) { + return { + type: "log", + payload: { message: "hello", ...overrides.payload }, + meta: { ...overrides.meta }, + }; +} + +// ---- console items are dropped ------------------------------------------------ + +test("drops a Faro log item tagged as relayed console output", () => { + const item = faroLog({ payload: { context: { "hot.relay": "console-error" } } }); + assert.equal(scrubTelemetry(item), null); +}); + +test("does not drop an ordinary log item", () => { + const item = faroLog({ payload: { context: { "hot.surface": "authoring" } } }); + assert.notEqual(scrubTelemetry(item), null); +}); + +// ---- drop `meta.user` ----------------------------------------------------------- + +test("drops meta.user entirely", () => { + const item = faroLog({ meta: { user: { email: "artur.medrygal@handsontable.com", id: "u1" } } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.meta.user, undefined); +}); + +// ---- reduce browser meta to device/browser classes ------------------------------ + +test("reduces meta.browser (a real Chrome UA) to the browser and device classes, dropping the raw UA", () => { + const item = faroLog({ meta: { browser: { userAgent: CHROME_UA, viewportWidth: "1920", viewportHeight: "1080" } } }); + const scrubbed = scrubTelemetry(item); + assert.deepEqual(scrubbed.meta.browser, { browser: "chrome", device: "desktop" }); +}); + +test("drops meta.os and meta.device (fingerprint-shaped fields)", () => { + const item = faroLog({ meta: { os: { name: "Windows", version: "10" }, device: { model_name: "Pixel 7" } } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.meta.os, undefined); + assert.equal(scrubbed.meta.device, undefined); +}); + +// ---- strip query/fragment on a URL-valued field, isolated from the attribute +// allowlist: url.full would be dropped by the allowlist rule regardless, +// which would keep this case green even with query-stripping removed — +// meta.page.url is not an "attribute" at all, so only the query-stripping +// rule can make this pass ----------------------------------------------------- + +test("strips the query string and fragment from meta.page.url", () => { + const item = faroLog({ meta: { page: { url: "https://demos.handsontable.com/share/abc123?secret=1#frag" } } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.meta.page.url, "https://demos.handsontable.com/share/abc123"); +}); + +test("stripQueryAndFragment falls back to a cut at the first ?/# for a non-URL value", () => { + assert.equal(stripQueryAndFragment("/share/abc123?secret=1#frag"), "/share/abc123"); + assert.equal(stripQueryAndFragment("/share/abc123"), "/share/abc123"); +}); + +// ---- redact preview hosts, on a field the query-stripping rule never touches --- + +test("redacts a real Tier-2 preview host in a stack-frame filename with no query string", () => { + const item = { + type: "exception", + payload: { + value: "boom", + stacktrace: { frames: [{ filename: `https://${PREVIEW_HOST}/src/main.js`, function: "render" }] }, + }, + meta: {}, + }; + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.stacktrace.frames[0].filename, "https:///src/main.js"); +}); + +test("strips a bundler cache-busting query string from a stack-frame filename too", () => { + const item = { + type: "exception", + payload: { + value: "boom", + stacktrace: { frames: [{ filename: "https://demos.handsontable.com/src/main.js?t=1758625200001", function: "render" }] }, + }, + meta: {}, + }; + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.stacktrace.frames[0].filename, "https://demos.handsontable.com/src/main.js"); +}); + +test("redacts a preview host on meta.page.url together with the query strip", () => { + const item = faroLog({ meta: { page: { url: `https://${PREVIEW_HOST}/src/main.js?x=1` } } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.meta.page.url, "https:///src/main.js"); +}); + +// ---- strip Babel code frames from message-bearing fields ----------------------- + +test("strips a real Babel code frame from a log item's message", () => { + const item = faroLog({ payload: { message: CODE_FRAME_MESSAGE } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.message, "unknown: Unexpected token (1:10)"); +}); + +test("strips a real Babel code frame from an exception's value", () => { + const item = { type: "exception", payload: { value: CODE_FRAME_MESSAGE }, meta: {} }; + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.value, "unknown: Unexpected token (1:10)"); +}); + +// ---- strip a query string embedded in message-bearing text ----------------- + +test("strips a query string off a URL embedded in a log item's message text, keeping the surrounding text", () => { + const item = faroLog({ + payload: { message: "fetch failed for https://example.com/api/versions?token=SECRET after 3 retries" }, + }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.message, "fetch failed for https://example.com/api/versions after 3 retries"); +}); + +test("strips a query string off an embedded preview-host URL in a message, after the host itself is redacted", () => { + const item = faroLog({ + payload: { message: `stale preview at https://${PREVIEW_HOST}/src/main.js?t=12345` }, + }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.message, "stale preview at https:///src/main.js"); +}); + +// ---- redact an IP embedded in message text, browser-side defense-in-depth -- + +test("redacts an IPv4 address embedded in a log item's message text", () => { + const item = faroLog({ payload: { message: "connection from 192.0.2.55 refused" } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.message, "connection from refused"); +}); + +test("redacts an IPv6 address embedded in an exception's value", () => { + const item = { type: "exception", payload: { value: "failed for 2001:db8::8a2e:370:7334" }, meta: {} }; + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.value, "failed for "); +}); + +test("does not redact a version string that merely looks IP-shaped", () => { + const item = faroLog({ payload: { message: "Handsontable 18.1.1, build 1.2.3.4-beta" } }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.message, "Handsontable 18.1.1, build 1.2.3.4-beta"); +}); + +// ---- allowlist attributes/context (drops forbidden attrs, §3) ------------------ + +test("drops forbidden Faro context attributes (url.full, geo), keeps allowlisted hot.* ones", () => { + const item = faroLog({ + payload: { + context: { + "hot.surface": "authoring", + "url.full": "https://demos.handsontable.com/api/versions?x=1", + "geo.country": "PL", + "client.address": "127.0.0.1", + }, + }, + }); + const scrubbed = scrubTelemetry(item); + assert.deepEqual(scrubbed.payload.context, { "hot.surface": "authoring" }); +}); + +// ---- redactPreviewHosts on every string, not only the fields named above ------- + +test("redacts a preview host inside an allowlisted context value (session.id), which no targeted rule above touches", () => { + const item = faroLog({ + payload: { context: { "session.id": `plid-from-https://${PREVIEW_HOST}/leaked` } }, + }); + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.context["session.id"], "plid-from-https:///leaked"); +}); + +test("redacts a preview host inside payload.type (an exception's error-class name), a field no targeted rule names", () => { + const item = { + type: "exception", + payload: { type: `Error`, value: "boom" }, + meta: {}, + }; + const scrubbed = scrubTelemetry(item); + assert.equal(scrubbed.payload.type, "Error>"); +}); + +// ---- the OTLP-shape branch (ingest, on an already-normalised record) ----------- + +test("scrubs an OTLP-shaped record: drops url.full and geo attributes, strips code frame and preview host from the body", () => { + const record = { + body: `${CODE_FRAME_MESSAGE}\nat https://${PREVIEW_HOST}/src/main.js`, + attributes: { "hot.demo_id": "r-react-18-0-0" }, + resourceAttributes: { + "hot.surface": "api", + "url.full": "https://demos.handsontable.com/api/versions?x=1", + "geo.country": "PL", + "geo.asn": "12345", + }, + }; + const scrubbed = scrubTelemetry(record); + assert.deepEqual(scrubbed.resourceAttributes, { "hot.surface": "api" }); + assert.deepEqual(scrubbed.attributes, { "hot.demo_id": "r-react-18-0-0" }); + assert.equal(scrubbed.body, "unknown: Unexpected token (1:10)\nat https:///src/main.js"); +}); + +test("redacts a preview host in an object KEY, not only values (a MeasurementEvent's metric-name keys reach the OTLP body verbatim via JSON.stringify)", () => { + const item = { + type: "measurement", + payload: { values: { [`lcp_https://${PREVIEW_HOST}/leak`]: 1200 }, timestamp: new Date().toISOString() }, + meta: {}, + }; + const scrubbed = scrubTelemetry(item); + const keys = Object.keys(scrubbed.payload.values); + assert.equal(keys.length, 1); + assert.equal(keys[0], "lcp_https:///leak"); +}); + +// ---- never mutates the input ---------------------------------------------------- + +test("never mutates its argument", () => { + const item = faroLog({ meta: { user: { email: "x@y.z" } } }); + const before = JSON.stringify(item); + scrubTelemetry(item); + assert.equal(JSON.stringify(item), before); +}); + +// ---- a malformed stack frame must never throw ------------------------------- + +test("scrubTelemetry does not throw on a null entry inside stacktrace.frames — the exact `500` probe", () => { + const item = { + type: "exception", + payload: { type: "TypeError", value: "x", stacktrace: { frames: [null] } }, + meta: {}, + }; + // Before the fix, `frame.filename` on the `null` entry threw a + // `TypeError` that escaped this function entirely — an unauthenticated + // `{"exceptions":[{"stacktrace":{"frames":[null]}}]}` POST to + // `/telemetry/collect` became an uncaught `500`. + assert.doesNotThrow(() => scrubTelemetry(item)); +}); + +test("scrubTelemetry skips a non-object stack frame but still scrubs the real frames around it", () => { + const item = { + type: "exception", + payload: { + type: "TypeError", + value: "x", + stacktrace: { + frames: [ + { filename: `https://${PREVIEW_HOST}/a.js?t=1` }, + null, + undefined, + "not-an-object", + { filename: `https://${PREVIEW_HOST}/b.js` }, + ], + }, + }, + meta: {}, + }; + const scrubbed = scrubTelemetry(item); + const filenames = scrubbed.payload.stacktrace.frames.map((f) => (f && typeof f === "object" ? f.filename : f)); + assert.equal(filenames[0], "https:///a.js"); + assert.equal(filenames[4], "https:///b.js"); +}); diff --git a/runner/pipeline/sentry-gating.test.mjs b/runner/pipeline/sentry-gating.test.mjs index d4f1eca8c7..ea269a3974 100644 --- a/runner/pipeline/sentry-gating.test.mjs +++ b/runner/pipeline/sentry-gating.test.mjs @@ -1,5 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { join } from "node:path"; import { PRODUCTION_HOST, resolveReporting } from "../apps/authoring/src/reportingGate.ts"; import { apiSentryDsn, @@ -7,9 +9,15 @@ import { rehomeBudgetAlert, } from "../workers/api/src/sentry-gate.ts"; import { + applyFaroTee, isEdgelessForeignSessionStart, + isForeignUnhandled, isOfficeScannerRejection, + isUnhandledNoise, + withoutMessageEchoFrames, } from "../apps/authoring/src/eventGate.ts"; +import { resolveSentryScope, reportsDiagnosticToSentry } from "../apps/authoring/src/sentryScope.ts"; +import { safeInit } from "../apps/authoring/src/bootGuard.ts"; // DEV-2540. Three classes of traffic reached the production Sentry project that had // no business being there — local dev sessions, a Playwright run pointed at @@ -148,6 +156,181 @@ test("ordinary events pass through untouched", () => { } }); +// ── isUnhandledNoise / isForeignUnhandled ──────────────────────────────────────── +// +// Lives in `eventGate.ts`, not private to `sentry.ts`, so +// `telemetry/faro.ts`'s `beforeSend` can apply the same predicates Sentry's +// own `beforeSend` does — contract §6's "shared noise gates" requirement. + +test("isUnhandledNoise: a ResizeObserver loop warning, unhandled, is noise", () => { + const event = { + exception: { values: [{ type: "Error", value: "ResizeObserver loop completed with undelivered notifications.", mechanism: { handled: false } }] }, + }; + assert.equal(isUnhandledNoise(event), true); +}); + +test("isUnhandledNoise: a navigation-abort Failed to fetch, unhandled, is noise", () => { + const event = { + exception: { values: [{ type: "TypeError", value: "Failed to fetch", mechanism: { handled: false } }] }, + }; + assert.equal(isUnhandledNoise(event), true); +}); + +test("isUnhandledNoise: the SAME message, but handled (an explicit report), is NOT noise", () => { + const event = { + exception: { values: [{ type: "TypeError", value: "Failed to fetch", mechanism: { handled: true } }] }, + }; + assert.equal(isUnhandledNoise(event), false, "an explicit reportError('Failed to fetch') must still be reported"); +}); + +test("isUnhandledNoise: an unrelated unhandled error is NOT noise", () => { + const event = { + exception: { values: [{ type: "TypeError", value: "x is not a function", mechanism: { handled: false } }] }, + }; + assert.equal(isUnhandledNoise(event), false); +}); + +test("isForeignUnhandled: an unhandled error whose stack is entirely outside this origin is dropped", () => { + const event = { + exception: { + values: [ + { + type: "TypeError", + value: "boom", + mechanism: { handled: false }, + stacktrace: { frames: [{ filename: "https://sandpack-bundler.codesandbox.io/bundle.js" }] }, + }, + ], + }, + }; + assert.equal(isForeignUnhandled(event, "https://demos.handsontable.com"), true); +}); + +test("isForeignUnhandled: an unhandled error with an own-origin frame is NOT dropped", () => { + const event = { + exception: { + values: [ + { + type: "TypeError", + value: "boom", + mechanism: { handled: false }, + stacktrace: { frames: [{ filename: "https://demos.handsontable.com/assets/index.js" }] }, + }, + ], + }, + }; + assert.equal(isForeignUnhandled(event, "https://demos.handsontable.com"), false); +}); + +test("isForeignUnhandled: a HANDLED report through a foreign frame is NOT dropped", () => { + const event = { + exception: { + values: [ + { + type: "TypeError", + value: "boom", + mechanism: { handled: true }, + stacktrace: { frames: [{ filename: "https://sandpack-bundler.codesandbox.io/bundle.js" }] }, + }, + ], + }, + }; + assert.equal(isForeignUnhandled(event, "https://demos.handsontable.com"), false); +}); + +test("isUnhandledNoise / isForeignUnhandled: no exception values -> false, not thrown on", () => { + assert.equal(isUnhandledNoise({}), false); + assert.equal(isForeignUnhandled({ exception: { values: [] } }, "https://x"), false); +}); + +// ── withoutMessageEchoFrames — Faro's gecko-regex message-echo frame ──────────── +// +// Faro's stack parser can turn the `Error: ` line itself into a fake +// frame (no `lineno`, `filename` = a URL quoted in the message). Un-gated, that +// fake frame reaches `isForeignUnhandled` and drops the whole event — the exact +// finding input. + +test("withoutMessageEchoFrames: drops the exact fake frame Faro produced for the pii finding", () => { + const message = + "HAIKU1 pii jane.doe@example.com 192.0.2.55 https://x.test/p?token=SECRET123"; + const frames = [ + { + filename: "https://x.test/p?token=SECRET123", + function: "Error: HAIKU1 pii jane.doe@example.com 192.0.2.55 ", + }, + ]; + assert.deepEqual(withoutMessageEchoFrames(message, frames), []); +}); + +test("withoutMessageEchoFrames: keeps a real frame (has a lineno) even if its filename appears in the message", () => { + const message = "boom at https://app.test/main.js"; + const frames = [ + { filename: "https://app.test/main.js", function: "doThing", lineno: 12, colno: 3 }, + ]; + assert.deepEqual(withoutMessageEchoFrames(message, frames), frames); +}); + +test("withoutMessageEchoFrames: keeps a real, genuinely foreign frame not quoted in the message", () => { + // The full pipeline case for the second requirement: a real extension/ + // third-party frame must still be droppable by isForeignUnhandled downstream — + // this helper must not touch it. + const message = "boom"; + const frames = [ + { filename: "https://sandpack-bundler.codesandbox.io/bundle.js", function: "run", lineno: 4 }, + ]; + assert.deepEqual(withoutMessageEchoFrames(message, frames), frames); +}); + +test("withoutMessageEchoFrames: no message or no frames -> frames returned unchanged", () => { + const frames = [{ filename: "https://x.test/p", function: "f" }]; + assert.equal(withoutMessageEchoFrames(undefined, frames), frames); + assert.equal(withoutMessageEchoFrames("boom", undefined), undefined); +}); + +test("end to end: the pii finding's message no longer makes isForeignUnhandled drop the event", () => { + const message = + "HAIKU1 pii jane.doe@example.com 192.0.2.55 https://x.test/p?token=SECRET123"; + const rawFrames = [ + { + filename: "https://x.test/p?token=SECRET123", + function: "Error: HAIKU1 pii jane.doe@example.com 192.0.2.55 ", + }, + ]; + const event = { + exception: { + values: [ + { + type: "Error", + value: message, + mechanism: { handled: false }, + stacktrace: { frames: withoutMessageEchoFrames(message, rawFrames) }, + }, + ], + }, + }; + assert.equal(isForeignUnhandled(event, "http://localhost:5173"), false); +}); + +test("a genuinely foreign third-party script error is still dropped", () => { + const event = { + exception: { + values: [ + { + type: "TypeError", + value: "boom", + mechanism: { handled: false }, + stacktrace: { + frames: withoutMessageEchoFrames("boom", [ + { filename: "https://sandpack-bundler.codesandbox.io/bundle.js", function: "inject", lineno: 7 }, + ]), + }, + }, + ], + }, + }; + assert.equal(isForeignUnhandled(event, "http://localhost:5173"), true); +}); + // ── DEV-2858. beforeSend suppression gates for two NOT-OURS populations ───────── // // `isOfficeScannerRejection` (DEMOS-5F) and `isEdgelessForeignSessionStart` @@ -342,3 +525,166 @@ test("N7: an event with no tags at all does not throw and is not dropped", () => assert.equal(isEdgelessForeignSessionStart({}), false); assert.equal(isEdgelessForeignSessionStart({ tags: {} }), false); }); + +// ── contract §11 / ADR §E.3: VITE_SENTRY_SCOPE ────────────────────────────────────── +// +// `sentryScope.ts` is import-free for the same reason as `reportingGate.ts` — see +// its own header. The truth table: `"full"` is the default for every input other +// than the exact string `"uncaught"`, and `reportsDiagnosticToSentry` only ever +// narrows `reportingEnabled`, never widens it. + +test("resolveSentryScope: only the literal 'uncaught' opens the narrow scope", () => { + assert.equal(resolveSentryScope("uncaught"), "uncaught"); +}); + +test("resolveSentryScope: absent, empty, or any other string stays 'full'", () => { + for (const raw of [undefined, "", "Uncaught", "UNCAUGHT", "full", "off", "uncaught "]) { + assert.equal(resolveSentryScope(raw), "full", `raw=${JSON.stringify(raw)}`); + } +}); + +test("reportsDiagnosticToSentry: full scope + reporting enabled -> true", () => { + assert.equal(reportsDiagnosticToSentry(true, "full"), true); +}); + +test("reportsDiagnosticToSentry: uncaught scope closes it even though reporting is enabled", () => { + // The launch-plan flip (ADR §E.3): once this ships, a handled diagnostic no + // longer reaches Sentry at all, on the production host, with reporting on. + assert.equal(reportsDiagnosticToSentry(true, "uncaught"), false); +}); + +test("reportsDiagnosticToSentry: never widens a closed reportingEnabled gate", () => { + // The regression this guards: a scope switch must not become a second way to + // turn Sentry on when the production/automation gate (reportingGate.ts) is + // already closed — full scope on a closed gate still reports nothing. + assert.equal(reportsDiagnosticToSentry(false, "full"), false); + assert.equal(reportsDiagnosticToSentry(false, "uncaught"), false); +}); + +// ── unguarded browser telemetry ─────────────────────────────────────────────────── +// +// `main.tsx`'s `initTelemetry()` call and `sentry.ts`'s ADR §E.2 tee were both +// unguarded — a synchronous throw in either must not propagate out (blanking the +// app before `createRoot`, or making the SDK drop the whole Sentry event). Both +// fixes are thin call sites around the two guarded functions below; these tests +// exercise the actual guarding logic. Reverting either `try`/`catch` in +// `bootGuard.ts#safeInit` / `eventGate.ts#applyFaroTee` back to an unguarded call +// makes the matching "still renders" / "still returns the event" test below throw +// instead of passing. + +test("safeInit: a throwing init is swallowed and reported, never propagates (main.tsx still renders)", () => { + let reported; + assert.doesNotThrow(() => { + safeInit( + () => { + throw new Error("Faro client construction failed"); + }, + (err) => { + reported = err; + }, + ); + }); + assert.ok(reported instanceof Error); + assert.equal(reported.message, "Faro client construction failed"); +}); + +test("safeInit: a non-throwing init runs normally and onError is never called", () => { + let ran = false; + let reported; + safeInit( + () => { + ran = true; + }, + (err) => { + reported = err; + }, + ); + assert.equal(ran, true); + assert.equal(reported, undefined); +}); + +test("applyFaroTee: sets the page_load_id tag and pushes a sentry.event, returns the event", () => { + const event = { event_id: "abc123", tags: { existing: "x" } }; + const telemetry = { + pageLoadId: () => "plid-1", + event(name, attrs) { + this.calls = this.calls ?? []; + this.calls.push({ name, attrs }); + }, + }; + const out = applyFaroTee(event, telemetry); + assert.equal(out, event, "must return the same event, never null/undefined"); + assert.equal(out.tags.page_load_id, "plid-1"); + assert.equal(out.tags.existing, "x", "existing tags must be preserved"); + assert.deepEqual(telemetry.calls, [{ name: "sentry.event", attrs: { sentry_event_id: "abc123" } }]); +}); + +test("applyFaroTee: a throwing telemetry.pageLoadId() is swallowed — the event still ships", () => { + const event = { event_id: "abc123", tags: {} }; + const telemetry = { + pageLoadId: () => { + throw new Error("Faro client not ready"); + }, + event: () => { + throw new Error("must not be reached"); + }, + }; + let out; + assert.doesNotThrow(() => { + out = applyFaroTee(event, telemetry); + }); + assert.equal(out, event, "the event must still be returned, not dropped"); +}); + +test("applyFaroTee: a throwing telemetry.event() is swallowed — the event still ships with its tag set", () => { + const event = { event_id: "abc123", tags: {} }; + const telemetry = { + pageLoadId: () => "plid-2", + event: () => { + throw new Error("Faro push failed"); + }, + }; + let out; + assert.doesNotThrow(() => { + out = applyFaroTee(event, telemetry); + }); + assert.equal(out, event); + assert.equal(out.tags.page_load_id, "plid-2", "the tag set before the throw is kept, not rolled back"); +}); + +// Advisor follow-up on minor triage item 5: the tests above prove +// `safeInit`/`applyFaroTee` THEMSELVES never throw — they do NOT prove the +// real call sites still call them. `main.tsx`/`sentry.ts` cannot be +// imported here (they pull in React/`@sentry/react`/`import.meta.env`, the +// same constraint this file's own header note gives for `sentry.ts` and +// `index.ts`), so the wiring is pinned structurally instead, the same +// pattern `pipeline/mcp-create.test.mjs`'s "the update route calls +// isMcpCreated()" test uses. Reverting either call site (back to a bare +// `initTelemetry();`, or the inline try/catch instead of +// `applyFaroTee(event, telemetry)`) makes the matching assertion fail even +// though every test above it stays green. +test("main.tsx calls initTelemetry() through safeInit(), not bare", () => { + const root = join(import.meta.dirname, ".."); + const source = readFileSync(join(root, "apps/authoring/src/main.tsx"), "utf8"); + assert.match(source, /safeInit\(\s*initTelemetry\s*,/, "main.tsx must call initTelemetry() through safeInit()"); + assert.doesNotMatch( + source, + /^\s*initTelemetry\(\);\s*$/m, + "a bare, unguarded initTelemetry(); call on its own line would defeat the guard entirely", + ); +}); + +test("sentry.ts's beforeSend returns applyFaroTee(event, telemetry), not an inline tee", () => { + const root = join(import.meta.dirname, ".."); + const source = readFileSync(join(root, "apps/authoring/src/sentry.ts"), "utf8"); + assert.match( + source, + /return applyFaroTee\(event, telemetry\);/, + "beforeSend must return applyFaroTee(event, telemetry), the guarded tee", + ); + assert.doesNotMatch( + source, + /telemetry\.pageLoadId\(\)/, + "sentry.ts itself must not call telemetry.pageLoadId() directly — that belongs entirely to eventGate.ts#applyFaroTee now", + ); +}); diff --git a/runner/pipeline/serve-share-point.test.mjs b/runner/pipeline/serve-share-point.test.mjs new file mode 100644 index 0000000000..94956a7420 --- /dev/null +++ b/runner/pipeline/serve-share-point.test.mjs @@ -0,0 +1,93 @@ +// `GET /api/demos/:id` must not emit a §5 `serve.share` point +// unconditionally on every 2xx/4xx answer: that route is the metadata +// load for three different callers (App.tsx's edit/share loader in either +// mode, `FullMode`'s own fetch, and any ad hoc `GET /api/demos/`), not +// only the share-page document view — an existence-check-shaped +// `GET /api/demos/` call is not a page view either. +// +// The fix: `App.tsx`'s share-mode loader appends `?view=share` on this one +// fetch only, and the server only counts a point when that marker is +// present — never for an edit-mode load, a `FullMode` load, or an +// unmarked probe. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { ctx, makeEnv, demoRow } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +/** Same `bindingSink`-routing idiom as `lite-inject.test.mjs`/ + * `session-end-framework.test.mjs`'s own `countingEnv()`. */ +function countingEnv(seedRows = []) { + const { env } = makeEnv(seedRows); + const points = []; + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, points }; +} + +const sharePoints = (points) => points.filter((p) => p.indexes[0] === "serve.share"); +const metaRequest = (id, { share = false } = {}) => + new Request(`https://demos.handsontable.com/api/demos/${id}${share ? "?view=share" : ""}`); + +// blob/double slot positions (§4, AE_COLUMNS): outcome=blob8 (index 7), +// demo_id=blob12 (index 11), count=double1 (index 0). +const OUTCOME_SLOT = 7; +const DEMO_ID_SLOT = 11; +const COUNT_SLOT = 0; + +test("GET /api/demos/:id WITHOUT ?view=share (the edit-page/FullMode shape) answers 200 but emits NO serve.share point", async () => { + const { env, points } = countingEnv([demoRow({ id: "abc123" })]); + const res = await worker.fetch(metaRequest("abc123"), env, ctx); + assert.equal(res.status, 200); + assert.equal(sharePoints(points).length, 0, "an unmarked metadata fetch must never count as a share view"); +}); + +test("GET /api/demos/:id?view=share (the actual share-page load) emits exactly one serve.share 2xx point", async () => { + const { env, points } = countingEnv([demoRow({ id: "abc123" })]); + const res = await worker.fetch(metaRequest("abc123", { share: true }), env, ctx); + assert.equal(res.status, 200); + const sb = sharePoints(points); + assert.equal(sb.length, 1, `expected exactly 1 serve.share point, got ${sb.length}`); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "2xx"); + assert.equal(sb[0].blobs[DEMO_ID_SLOT], "abc123"); + assert.equal(sb[0].doubles[COUNT_SLOT], 1); +}); + +test("GET /api/demos/ WITHOUT ?view=share (an ad hoc existence-check probe) answers 404 and emits NO serve.share point", async () => { + const { env, points } = countingEnv([]); + const res = await worker.fetch(metaRequest("does-not-exist"), env, ctx); + assert.equal(res.status, 404); + assert.equal(sharePoints(points).length, 0, "an unmarked 404 must never inflate the share 4xx rate"); +}); + +test("GET /api/demos/?view=share (a genuinely broken share link) still counts as a real serve.share 4xx", async () => { + const { env, points } = countingEnv([]); + const res = await worker.fetch(metaRequest("does-not-exist", { share: true }), env, ctx); + assert.equal(res.status, 404); + const sb = sharePoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "4xx"); +}); + +test("GET /api/demos/:id?view=share on a revoked demo answers 410 and emits one serve.share 4xx point", async () => { + const { env, points } = countingEnv([demoRow({ id: "abc123", revoked: 1 })]); + const res = await worker.fetch(metaRequest("abc123", { share: true }), env, ctx); + assert.equal(res.status, 410); + const sb = sharePoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "4xx"); +}); + +test("GET /api/demos/:id?view=share on a revoked demo WITHOUT the marker emits nothing (same rule, revoked branch)", async () => { + const { env, points } = countingEnv([demoRow({ id: "abc123", revoked: 1 })]); + const res = await worker.fetch(metaRequest("abc123"), env, ctx); + assert.equal(res.status, 410); + assert.equal(sharePoints(points).length, 0); +}); diff --git a/runner/pipeline/session-end-framework.test.mjs b/runner/pipeline/session-end-framework.test.mjs new file mode 100644 index 0000000000..53aaaccba3 --- /dev/null +++ b/runner/pipeline/session-end-framework.test.mjs @@ -0,0 +1,128 @@ +// `session.end` must not be emitted with `framework: ""`: +// `teardownLiveSession` and `sessionSubrouteGuard` only ever have a +// `sessionId`, never the framework a session was created with, so the +// `tier2-sessions` dashboard panel ("session.end awake seconds, p95 by +// reason") — which filters `blob6 IN (${framework:sqlstring})` — would stay +// permanently empty, the same failure shape as `container.boot_ms`'s. +// +// The fix carries the framework on the per-session KV meter +// (`workers/api/src/budget.ts#SessionMeter.framework`, set by +// `startSessionMeter` at create) and reads it back through +// `meterSession`'s return value at teardown, before the `final: true` +// flush deletes the KV entry it lives in. This file pins that round trip +// through the real `POST /api/session` -> `DELETE /api/session/:id` route +// pair, using the same harness +// `pipeline/session-create-container-starting.test.mjs` and +// `pipeline/container-boot-ms.test.mjs` use. + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; + +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; +import { setSandboxFactory } from "./fixtures/cloudflare-sandbox-stub.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const FILES = { "package.json": JSON.stringify({ name: "demo" }) }; + +const sessionRequest = (body) => + new Request("https://demos.handsontable.com/api/session", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body), + }); + +const deleteRequest = (id) => + new Request(`https://demos.handsontable.com/api/session/${id}`, { method: "DELETE" }); + +/** See `pipeline/lite-inject.test.mjs#makeCountingEnv` / `container-boot-ms + * .test.mjs#countingEnv`: flips `getSink()` to its `bindingSink` branch (a + * real `RUNNER_EVENTS` fake) instead of the local-mode ClickHouse HTTP + * fetch, which has nothing to talk to in this sandbox. */ +function countingEnv() { + const { env } = makeEnv(); + const points = []; + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, points }; +} + +const endPoints = (points) => points.filter((p) => p.indexes[0] === "session.end"); + +function fakeSandbox({ destroyError } = {}) { + return { + async mkdir() {}, + async writeFile() {}, + deleteFile: async () => {}, + exec: async () => ({ success: true, stdout: "", stderr: "" }), + async startProcess() {}, + async exposePort() { + return { url: "https://preview.test/session" }; + }, + async setFramework() {}, + async destroy() { + if (destroyError) throw destroyError; + }, + }; +} + +test("a clean pagehide teardown carries the session's own framework, not ''", async () => { + const { env, points } = countingEnv(); + const sandbox = fakeSandbox(); + setSandboxFactory(() => sandbox); + + const createRes = await worker.fetch(sessionRequest({ framework: "angular", files: FILES }), env, ctx); + const { sessionId } = await createRes.json(); + assert.ok(sessionId, "expected a sessionId from the create"); + + const deleteRes = await worker.fetch(deleteRequest(sessionId), env, ctx); + assert.equal(deleteRes.status, 204); + + const ends = endPoints(points); + assert.equal(ends.length, 1, "expected exactly one session.end point"); + assert.equal(ends[0].blobs[8], "pagehide", "blob9 reason"); + assert.equal( + ends[0].blobs[5], + "angular", + "blob6 framework — what the tier2-sessions panel's session.end filter reads", + ); +}); + +test("a declined destroy() still carries the framework on its teardown_failed point", async () => { + const { env, points } = countingEnv(); + // "The container service is unreachable" — one of `isExpectedTeardownFailure`'s + // recognised platform messages (session-lifecycle.ts), so `destroy()`'s + // throw degrades to `teardown_failed` instead of escaping. + const sandbox = fakeSandbox({ destroyError: new Error("The container service is unreachable, try again later") }); + setSandboxFactory(() => sandbox); + + const createRes = await worker.fetch(sessionRequest({ framework: "vue", files: FILES }), env, ctx); + const { sessionId } = await createRes.json(); + + const deleteRes = await worker.fetch(deleteRequest(sessionId), env, ctx); + assert.equal(deleteRes.status, 204, "a declined destroy still answers the fire-and-forget keepalive with 204"); + + const ends = endPoints(points); + assert.equal(ends.length, 1); + assert.equal(ends[0].blobs[8], "teardown_failed"); + assert.equal(ends[0].blobs[5], "vue", "framework must survive onto the teardown_failed point too"); +}); + +test("a session id that was never created still tears down (no meter) with an empty framework — no throw", async () => { + const { env, points } = countingEnv(); + const sandbox = fakeSandbox(); + setSandboxFactory(() => sandbox); + + // No POST /api/session first — this id has no KV meter at all, the + // "an id someone invented" case `hasSessionMeter`'s own doc describes. + const deleteRes = await worker.fetch(deleteRequest("react-js-invented-id"), env, ctx); + assert.equal(deleteRes.status, 204); + + const ends = endPoints(points); + assert.equal(ends.length, 1); + assert.equal(ends[0].blobs[8], "pagehide"); + assert.equal(ends[0].blobs[5], "", "no meter ever existed, so this degrades to today's framework-less point"); +}); diff --git a/runner/pipeline/session-malformed-json.test.mjs b/runner/pipeline/session-malformed-json.test.mjs new file mode 100644 index 0000000000..5bd2b9fa3b --- /dev/null +++ b/runner/pipeline/session-malformed-json.test.mjs @@ -0,0 +1,73 @@ +// Malformed JSON on the session-family POST routes must not reach the +// generic fetch catch-all: an uncaught `SyntaxError` from `request.json()`, +// answered as a 500, pollutes the `api.request` 5xx rate and the +// `api-5xx-rate` alert with what is really a 400-shaped client mistake. +// `POST /api/session` is public and unauthenticated, so it is the easiest +// route on this Worker for arbitrary client garbage to reach. +// +// Both routes below already had the shape check in place +// (`isPlainRecord`/`validateFileWrite`, whose 400 the fetch catch-all +// already answers via `InvalidFilePathError`) — the gap was that a body +// which doesn't even parse as JSON never reached that check. The fix is +// the same one-line `.catch(() => null)` `POST /api/theme` and +// `/api/chat` use for the same reason. +// +// Driven through the real router (`workers/api/src/index.ts`'s default +// export) — a hand-rolled re-check of the body would not catch a +// regression in the actual route. +// Run: node --experimental-strip-types --test pipeline/session-malformed-json.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const HOST = "https://demos.handsontable.com"; + +function malformedJsonRequest(method, path) { + return new Request(`${HOST}${path}`, { + method, + headers: { "Content-Type": "application/json" }, + // Not valid JSON — `request.json()` rejects on this body. + body: "{ this is not json", + }); +} + +/** `api.request` (recordRequestSignal) writes an AE point for every request; with + * the bare `makeEnv()` env this falls through to the local-ClickHouse-HTTP sink + * (`serviceEnvironment`'s non-production branch) and makes a real network call to + * whatever is on :8123 on this machine — same fix as `snapshot-build-point.test.mjs`'s + * own `envWithPointCapture` helper: an in-memory sink routes the write away from + * the network entirely. */ +function envWithPointCapture() { + const { env, ...rest } = makeEnv(); + env.RUNNER_EVENTS = { writeDataPoint() {} }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, ...rest }; +} + +test("a malformed POST /api/session body is a 400, not the fetch catch-all's 500", async () => { + const { env } = envWithPointCapture(); + const res = await worker.fetch(malformedJsonRequest("POST", "/api/session"), env, ctx); + assert.equal(res.status, 400, "must not reach the generic 500 catch-all"); + const body = await res.json(); + // The existing error JSON shape (`isPlainRecord`'s own 400 for a body that + // fails the record check) — a parse failure now lands in that same branch + // rather than getting its own new shape. + assert.equal(body.error, "request body must be a plain record"); +}); + +test("a malformed POST /api/session/:id/file body is a 400, not a 500", async () => { + const { env } = envWithPointCapture(); + const res = await worker.fetch(malformedJsonRequest("POST", "/api/session/sess-1/file"), env, ctx); + assert.equal(res.status, 400, "must not reach the generic 500 catch-all"); + const body = await res.json(); + // validateFileWrite's own InvalidFilePathError message — unchanged by this + // fix, just reachable for an unparseable body now instead of only a + // wrong-shaped one. + assert.equal(body.error, "file write must be a plain record"); +}); diff --git a/runner/pipeline/session-start-failure.test.mjs b/runner/pipeline/session-start-failure.test.mjs index 1ab6a82419..8069a94cc1 100644 --- a/runner/pipeline/session-start-failure.test.mjs +++ b/runner/pipeline/session-start-failure.test.mjs @@ -537,3 +537,257 @@ test("the header-name list is capped, and says it was", async () => { // `extra` is not free — a response can carry arbitrarily many headers, and this // is the same bounding discipline `FAILURE_TEXT_MAX` applies to the body. }); + +// --------------------------------------------------------------------------- +// `onSessionStart` (§5 `session.start_ms`). `pipeline/browser-metrics.test.mjs` +// covers `apps/authoring/src/telemetry/metrics.ts`'s own emission logic against a +// fake hook; this covers whether `container.ts`'s classification and the +// create-clock reuse are correct against the REAL `mount()`. + +/** + * Drive `mount()` through a fake `POST /api/session` and return what + * `onSessionStart` saw. `respond(url, init)` answers the create POST; every other + * request (the status poll `mount()` kicks off right after returning, the cleanup + * DELETE on a failure) gets a quiet default that ends polling without further + * network activity. `window`/`fetch` are restored and the runtime disposed in + * `finally`, mirroring `sessionStartError` above. + */ +async function mountAndCollectSessionStart(respond) { + const fetchBefore = globalThis.fetch; + const windowBefore = globalThis.window; + globalThis.window = { addEventListener() {}, removeEventListener() {} }; + globalThis.fetch = (url, init = {}) => { + if (url.endsWith("/api/session") && init.method === "POST") return Promise.resolve(respond(url, init)); + // 410 stops the status-poll loop outright — the simplest "nothing more + // happens" shape for a unit test; a bare `ok:true` also serves the cleanup DELETE. + return Promise.resolve({ ok: false, status: 410, headers: new Headers(), text: () => Promise.resolve("") }); + }; + + const runtime = new ContainerRuntime(ENTRY, { iframe: {}, apiBase: "https://api.test" }); + const events = []; + runtime.onSessionStart((e) => events.push(e)); + try { + await runtime.mount({ ...FILES }).catch(() => {}); + return events; + } finally { + runtime.dispose(); + globalThis.fetch = fetchBefore; + globalThis.window = windowBefore; + } +} + +test("onSessionStart: a successful create reports outcome ready, once, with the create-clock's own elapsed time", async () => { + const events = await mountAndCollectSessionStart(() => ({ + ok: true, + status: 200, + type: "basic", + headers: new Headers(), + json: () => Promise.resolve({ previewUrl: "https://1234-sess-tok.demos.handsontable.com", port: 3000 }), + })); + + assert.equal(events.length, 1, "guard: exactly one session.start_ms per mount(), not one per internal step"); + assert.equal(events[0].outcome, "ready"); + assert.ok(events[0].elapsedMs >= 0, "reuses the same create-POST clock as SessionStartDiagnostics.elapsedMs"); +}); + +test("onSessionStart: each refusal code maps to its own outcome, and fires before dispose() clears the listener", async () => { + const cases = [ + { code: "at_capacity", status: 503, outcome: "at_capacity" }, + { code: "container_starting", status: 503, outcome: "container_starting" }, + { code: "budget_exhausted", status: 410, outcome: "budget_denied" }, + { code: "budget_login_required", status: 401, outcome: "budget_denied" }, + ]; + for (const { code, status, outcome } of cases) { + const events = await mountAndCollectSessionStart(() => ({ + ok: false, + status, + type: "basic", + headers: new Headers(), + text: () => Promise.resolve(JSON.stringify({ error: code, message: "refused" })), + })); + // If `dispose()` (called inside `mount()`'s own catch) cleared `sessionStartCbs` + // before this emission, `events` would be empty — the guard this asserts. + assert.equal(events.length, 1, `${code}: guard: emission must land before dispose() clears the listener`); + assert.equal(events[0].outcome, outcome, `${code} should map to ${outcome}`); + } +}); + +test("onSessionStart: an envelope-less timeout maps to boot_timeout", async () => { + const events = await mountAndCollectSessionStart(() => ({ + ok: false, + status: 522, + type: "basic", + headers: new Headers({ "cf-ray": RAY }), + text: () => Promise.resolve(""), + })); + + assert.equal(events.length, 1); + assert.equal(events[0].outcome, "boot_timeout"); +}); + +test("onSessionStart: the DEMOS-9 interception 504 (envelope-less, ray-less, headers readable) maps to error, not boot_timeout", async () => { + // Mirrors `classifySessionStartOutcome`'s own precedence, checked before the + // generic timeout tier since 504 is a member of both sets: "the response + // carries no sign of having come from our servers" is not a boot timeout. + const events = await mountAndCollectSessionStart(() => ({ + ok: false, + status: 504, + type: "basic", + headers: new Headers(), + text: () => Promise.resolve(""), + })); + + assert.equal(events.length, 1); + assert.equal(events[0].outcome, "error", "guard: DEMOS-9 interception must not read as our own boot timing out"); +}); + +test("onSessionStart: an enveloped 504 (our own words) still reads as a generic error, not boot_timeout", async () => { + // An envelope means the response came from our own Worker, not from the + // platform above it — `classifySessionStartOutcome`'s timeout/interception + // tiers are both gated on `!envelope`. + const events = await mountAndCollectSessionStart(() => ({ + ok: false, + status: 504, + type: "basic", + headers: new Headers(), + text: () => Promise.resolve(JSON.stringify({ error: "boom", message: "the pool is full" })), + })); + + assert.equal(events.length, 1); + assert.equal(events[0].outcome, "error"); +}); + +test("onSessionStart: fetch() itself throwing (no response at all) still reports outcome error with a real elapsed time", async () => { + const events = await mountAndCollectSessionStart(() => { + throw new TypeError("Failed to fetch"); + }); + + assert.equal(events.length, 1, "guard: a network error with no SessionStartError must still emit exactly once"); + assert.equal(events[0].outcome, "error"); + assert.ok(events[0].elapsedMs >= 0); +}); + +// --------------------------------------------------------------------------- +// `onHmr` (§5 `hmr.roundtrip_ms`). Drives the private `onFrameLoad` handler +// directly (the same style `sandpack-reload.test.mjs` drives `onMessage`) rather +// than through a full `poll()`/iframe simulation — see `HmrRoundtripEvent`'s own +// doc comment for what this hook does and does not observe (only a dev server +// that does a full page reload on an edit; genuine in-place HMR never fires an +// iframe `load` at all and stays invisible either way). + +/** `dispose()` unconditionally reaches for `window.removeEventListener` — needed + * even though these tests never call `mount()` (which is what normally stubs it). + * Restores `window` only after `fn()`'s promise settles, not synchronously after + * it returns one — the last of these tests disposes inside a `.finally()`. */ +async function withFakeWindow(fn) { + const windowBefore = globalThis.window; + globalThis.window = { addEventListener() {}, removeEventListener() {} }; + try { + return await fn(); + } finally { + globalThis.window = windowBefore; + } +} + +test("onHmr: a post-ready frame load following an edit flush reports the flush-to-load duration", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + const events = []; + runtime.onHmr((e) => events.push(e)); + runtime.didReady = true; + runtime.lastEditFlushDispatchedAt = performance.now() - 25; + + runtime.onFrameLoad(); + + assert.equal(events.length, 1); + assert.ok(events[0].durationMs >= 0); + runtime.dispose(); + })); + +test("onHmr: the very first (pre-ready) navigation never reports — that is the boot page, not an edit", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + const events = []; + runtime.onHmr((e) => events.push(e)); + // didReady is still false: this is the initial pointing navigation. + runtime.lastEditFlushDispatchedAt = performance.now() - 25; + + runtime.onFrameLoad(); + + assert.equal(events.length, 0, "guard: the pre-ready navigation must not be mistaken for an HMR round trip"); + runtime.dispose(); + })); + +test("onHmr: our own reload() navigation does not report", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + const events = []; + runtime.onHmr((e) => events.push(e)); + runtime.didReady = true; + runtime.reloadInFlight = true; + runtime.lastEditFlushDispatchedAt = performance.now() - 25; + + runtime.onFrameLoad(); + + assert.equal(events.length, 0, "guard: an explicit refresh must not read as a dev-server HMR reload"); + runtime.dispose(); + })); + +test("onHmr: a post-ready load with no preceding edit flush does not report (nothing to time)", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + const events = []; + runtime.onHmr((e) => events.push(e)); + runtime.didReady = true; + // lastEditFlushDispatchedAt stays null — no edit was made. + + runtime.onFrameLoad(); + + assert.equal(events.length, 0); + runtime.dispose(); + })); + +// Real in-place HMR never reaches `onFrameLoad` at all, so +// `lastEditFlushDispatchedAt` could otherwise sit set for minutes until an +// unrelated later reload reported that stale gap as the round-trip duration. +test("onHmr: a load arriving long after the flush (stale dispatch clock) does not report", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + const events = []; + runtime.onHmr((e) => events.push(e)); + runtime.didReady = true; + // Well past the bounded window (30s) — an unrelated reload long after the + // edit was flushed, not that edit's own round trip. + runtime.lastEditFlushDispatchedAt = performance.now() - 45_000; + + runtime.onFrameLoad(); + + assert.equal(events.length, 0, "guard: a stale dispatch timestamp must not be reported as this load's duration"); + runtime.dispose(); + })); + +test("flush() only starts the HMR dispatch clock once the preview is already ready", () => + withFakeWindow(() => { + const runtime = new ContainerRuntime(ENTRY, { iframe: {} }); + // Simulate the pre-ready buffered flush mount() triggers for edits made mid-create. + runtime.mounted = true; + runtime.sessionId = "sess-test"; + runtime.pending.set("/src/main.ts", "x"); + const before = runtime.lastEditFlushDispatchedAt; + + const fetchBefore = globalThis.fetch; + globalThis.fetch = () => Promise.resolve({ ok: true, status: 200, json: () => Promise.resolve({}) }); + return runtime + .flush() + .then(() => { + assert.equal( + runtime.lastEditFlushDispatchedAt, + before, + "guard: a pre-ready flush must not start the HMR clock — didReady is still false", + ); + }) + .finally(() => { + globalThis.fetch = fetchBefore; + runtime.dispose(); + }); + })); diff --git a/runner/pipeline/snapshot-build-bytes.test.mjs b/runner/pipeline/snapshot-build-bytes.test.mjs new file mode 100644 index 0000000000..9d9dac9f03 --- /dev/null +++ b/runner/pipeline/snapshot-build-bytes.test.mjs @@ -0,0 +1,224 @@ +// The §5 `snapshot.build` point's `bytes` field must be populated on +// either build path — `withSnapshotBuildPoint` (share.ts) must not emit +// `{ count, duration_ms }` with no `bytes` key, regardless of whether the +// built artifact is 2 KB or 20 MB. +// +// `withSnapshotBuildPoint` hands its callback an `addBytes` accumulator; +// `createDemo`/`updateDemo` call it with the real byte length of every +// object they write to R2 (a fresh build's own output, or a `build_cache` +// hit's copied objects — `R2Object.size`), and the point's `bytes` field +// carries the running total. +// +// Companion to `snapshot-build-point.test.mjs` (which proves the point +// fires at all) — this proves the one field that test never checked. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/snapshot-build-bytes.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { makeEnv } from "./fixtures/worker-harness.mjs"; +import { setSandboxFactory } from "./fixtures/cloudflare-sandbox-stub.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { createDemo, updateDemo } = await import("../workers/api/src/share.ts"); + +const ENTRY = { + framework: "javascript", + tier: 1, + installCommand: "pnpm install --frozen-lockfile", + buildCommand: "vite build", + outputDir: "dist", + outputGlob: null, + entry: "/index.js", + htmlEntry: "/index.html", +}; +const FILES = { "/package.json": JSON.stringify({ dependencies: { handsontable: "18.1.0" } }) }; + +// Deliberately uneven lengths so a bug that reports only the LAST file's size, +// or a hardcoded file count instead of a real sum, cannot pass by accident. +const BUILT = { + "index.html": "hello, bytes", + "assets/index-a1b2c3.js": "console.log('a real, if tiny, bundle');", +}; +const EXPECTED_BYTES = Object.values(BUILT).reduce((n, s) => n + new TextEncoder().encode(s).length, 0); + +/** A sandbox whose install and build both succeed and whose `dist/` holds `BUILT`, + * with real (non-empty, differently-sized) per-file content — `snapshot-build. + * test.mjs`'s own `buildEmitting` always answers `readFile` with `""`, which + * would make every file 0 bytes and pass a `bytes` field that is silently + * wrong just as easily as one that's silently absent. */ +function buildEmitting(built) { + return () => ({ + mkdir: async () => {}, + writeFile: async () => {}, + async readFile(path) { + const rel = Object.keys(built).find((r) => path.endsWith(r)); + return rel ? built[rel] : ""; + }, + destroy: async () => {}, + async exec(cmd) { + if (cmd.includes("find . -type f")) { + return { success: true, exitCode: 0, stdout: Object.keys(built).map((r) => `./${r}`).join("\n") }; + } + return { success: true, exitCode: 0, stdout: "", stderr: "" }; + }, + }); +} + +function envWithPointCapture(opts, seedArtifacts = {}) { + const points = []; + const { env, ...rest } = makeEnv([], [], seedArtifacts, opts); + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, points, ...rest }; +} + +function snapshotBuildPoints(points) { + return points.filter((p) => p.indexes[0] === "snapshot.build"); +} + +// §4 (AE_COLUMNS, packages/runtime/src/telemetry/metrics.ts): bytes=double7 +// (index 6) — a fixed slot shared by every metric that measures it, same as +// count=double1 (index 0) and duration_ms=double2 (index 1), the two +// `snapshot-build-point.test.mjs` already asserts on. +const BYTES_SLOT = 6; + +test("createDemo (fresh build, no cache) reports the real total bytes written to R2", async () => { + setSandboxFactory(buildEmitting(BUILT)); + try { + const { env, points } = envWithPointCapture({ buildCacheHit: false }); + await createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[7], "ok"); // outcome + assert.ok(EXPECTED_BYTES > 0, "sanity: the fixture itself is non-empty"); + assert.equal(sb[0].doubles[BYTES_SLOT], EXPECTED_BYTES); + } finally { + setSandboxFactory(null); + } +}); + +test("updateDemo (fresh build, no cache) reports the real total bytes written to R2", async () => { + setSandboxFactory(buildEmitting(BUILT)); + try { + const { env, points } = envWithPointCapture( + { buildCacheHit: false }, + ); + // Seed the row updateDemo expects to find. + const seeded = makeEnv( + [{ id: "abc123", framework: "javascript", tier: 1, ht_version: "18.1.0", files_hash: "old", r2_prefix: "demos/abc123/" }], + [], + {}, + { buildCacheHit: false }, + ); + seeded.env.RUNNER_EVENTS = env.RUNNER_EVENTS; + seeded.env.PREVIEW_HOST = env.PREVIEW_HOST; + await updateDemo(seeded.env, { + id: "abc123", + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + now: new Date().toISOString(), + }); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[7], "ok"); + assert.equal(sb[0].doubles[BYTES_SLOT], EXPECTED_BYTES); + } finally { + setSandboxFactory(null); + } +}); + +test("createDemo (build_cache hit) counts the copied artifact's bytes but not __source.json's", async () => { + // `fixtures/worker-harness.mjs`'s `fakeD1`, on a cache hit, always answers + // `{ r2_prefix: "demos/_prior-identical-build/" }` — see its own comment. + const CACHED_PREFIX = "demos/_prior-identical-build/"; + const seedArtifacts = { + [`${CACHED_PREFIX}index.html`]: "the real artifact", + // A prior demo's private source snapshot, sitting in the same cached + // directory (share.ts writes one next to every build). Deliberately + // larger than the artifact above, so a regression that counts it would + // move `bytes` by more than a rounding error — not just barely wrong. + [`${CACHED_PREFIX}__source.json`]: JSON.stringify({ + framework: "javascript", + files: { "/src/app.js": "x".repeat(1000) }, + }), + }; + const { env, points } = envWithPointCapture({ buildCacheHit: true }, seedArtifacts); + // `fakeR2.list()` always answers `{ objects: [] }` (a harness limitation + // unrelated to this — see snapshot-build-point.test.mjs's own cache-hit + // tests, which never reach the copy loop for the same reason). Overridden + // here, for this test only, so the copy loop under test actually runs. + env.ARTIFACTS.list = async ({ prefix }) => ({ + objects: Object.keys(seedArtifacts) + .filter((key) => key.startsWith(prefix)) + .map((key) => ({ key })), + }); + + await createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }); + + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[7], "ok"); + const artifactBytes = new TextEncoder().encode(seedArtifacts[`${CACHED_PREFIX}index.html`]).length; + assert.ok(artifactBytes > 0, "sanity: the fixture artifact is non-empty"); + assert.equal( + sb[0].doubles[BYTES_SLOT], + artifactBytes, + "must equal the copied artifact's own bytes, excluding the copied __source.json", + ); +}); + +// A guard test, not revert-check evidence for the fix itself (the three tests +// above already carry that: bytes was always 0/absent before the fix, so this +// assertion holds either way). Kept anyway — it pins the "never positive on +// failure" direction against a future change to the failure path specifically, +// which the other tests don't exercise. +test("a failed build reports bytes as 0 (or absent), never a partial/wrong total", async () => { + setSandboxFactory(() => ({ + mkdir: async () => {}, + writeFile: async () => {}, + readFile: async () => "", + destroy: async () => {}, + async exec(cmd) { + if (cmd.includes("install")) return { success: true, exitCode: 0, stdout: "", stderr: "" }; + return { success: false, exitCode: 1, stdout: "", stderr: "build blew up" }; + }, + })); + try { + const { env, points } = envWithPointCapture({ buildCacheHit: false }); + await assert.rejects(() => + createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }), + ); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1); + assert.equal(sb[0].blobs[7], "failed"); + assert.ok(!(sb[0].doubles[BYTES_SLOT] > 0), "a failed build never reports a positive byte count"); + } finally { + setSandboxFactory(null); + } +}); diff --git a/runner/pipeline/snapshot-build-point.test.mjs b/runner/pipeline/snapshot-build-point.test.mjs new file mode 100644 index 0000000000..35cfcfd3b7 --- /dev/null +++ b/runner/pipeline/snapshot-build-point.test.mjs @@ -0,0 +1,175 @@ +// The §5 `snapshot.build` point must be emitted on the synchronous build +// path too, not only the detached DO alarm path +// (`snapshot-jobs.ts#runSnapshotJob`): every real `vite build` run through +// the synchronous `createDemo`/`updateDemo` in `share.ts` left +// `SELECT count() FROM runner_events WHERE index1='snapshot.build'` at 0. +// +// The fix moves emission into `createDemo`/`updateDemo` themselves +// (`share.ts#withSnapshotBuildPoint`), so every caller gets exactly one +// `ok`/`failed` point regardless of whether it hit `build_cache` (a fast +// R2 copy) or ran a real container build — timed end to end, tagged +// `reason: "inline"` by default (the synchronous request-path callers in +// index.ts) or `reason: "detached"` (passed explicitly by +// `snapshot-jobs.ts`'s alarm) — this specs that consolidation doesn't turn +// into a double-emission. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { makeEnv } from "./fixtures/worker-harness.mjs"; +import { setSandboxFactory } from "./fixtures/cloudflare-sandbox-stub.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { createDemo, updateDemo } = await import("../workers/api/src/share.ts"); +const { runSnapshotJob } = await import("../workers/api/src/snapshot-jobs.ts"); + +const ENTRY = { framework: "react", tier: 1, buildCommand: "vite build", outputDir: "dist" }; +const FILES = { "/package.json": JSON.stringify({ dependencies: { handsontable: "18.1.0" } }) }; + +/** Routes AE writes through the `bindingSink` branch into an in-memory array + * (`getSink`'s production branch — see `lite-inject.test.mjs` for the same + * pattern), instead of the local-ClickHouse-HTTP fallback `serviceEnvironment` + * otherwise selects for a non-production `PREVIEW_HOST`. */ +function envWithPointCapture(...makeEnvArgs) { + const points = []; + const { env, ...rest } = makeEnv(...makeEnvArgs); + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + env.PREVIEW_HOST = "demos.handsontable.com"; + return { env, points, ...rest }; +} + +function snapshotBuildPoints(points) { + return points.filter((p) => p.indexes[0] === "snapshot.build"); +} + +// blob/double slot positions, §4 (AE_COLUMNS, packages/runtime/src/telemetry/metrics.ts): +// framework=blob6 (index 5), outcome=blob8 (index 7), reason=blob9 (index 8), +// count=double1 (index 0), duration_ms=double2 (index 1). +const FRAMEWORK_SLOT = 5; +const OUTCOME_SLOT = 7; +const REASON_SLOT = 8; +const COUNT_SLOT = 0; +const DURATION_SLOT = 1; + +test("createDemo (inline, build_cache hit — no container) emits exactly one snapshot.build ok/inline point", async () => { + const { env, points } = envWithPointCapture([], [], {}, { buildCacheHit: true }); + await createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1, `expected exactly 1 snapshot.build point, got ${sb.length}`); + assert.equal(sb[0].blobs[FRAMEWORK_SLOT], "react"); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "ok"); + assert.equal(sb[0].blobs[REASON_SLOT], "inline"); + assert.equal(sb[0].doubles[COUNT_SLOT], 1); + assert.ok(sb[0].doubles[DURATION_SLOT] >= 0, "duration_ms must be a real, non-negative number"); +}); + +test("updateDemo (inline, build_cache hit) emits exactly one snapshot.build ok/inline point", async () => { + const { env, points } = envWithPointCapture( + [{ id: "abc123", framework: "react", tier: 1, ht_version: "18.1.0", files_hash: "old", r2_prefix: "demos/abc123/" }], + [], + {}, + { buildCacheHit: true }, + ); + await updateDemo(env, { + id: "abc123", + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + now: new Date().toISOString(), + }); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1, `expected exactly 1 snapshot.build point, got ${sb.length}`); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "ok"); + assert.equal(sb[0].blobs[REASON_SLOT], "inline"); +}); + +test("createDemo's snapshot.build point survives past the call returning — it is awaited, not fire-and-forget (would be silently cancellable via ctx.waitUntil-less code otherwise)", async () => { + // A slow sink write: if withSnapshotBuildPoint used `void emitPoint(...)` + // instead of `await`, this test's assertion below would race the write + // and could observe 0 points instead of 1 — this is the revert-evidence + // for that specific regression risk (flagged before landing the fix). + let resolveWrite; + const writeDone = new Promise((r) => { resolveWrite = r; }); + const points = []; + const { env } = makeEnv([], [], {}, { buildCacheHit: true }); + env.PREVIEW_HOST = "demos.handsontable.com"; + env.RUNNER_EVENTS = { + writeDataPoint: (p) => { + points.push(p); + resolveWrite(); + }, + }; + await createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }); + // If the write were unawaited, this would need a `await writeDone` race — + // asserting immediately after `createDemo` resolves is the point: the + // write must already be done. + assert.equal(snapshotBuildPoints(points).length, 1); + await writeDone; // never hangs if the point above is already there +}); + +test("a build failure (real container build, no cache) emits snapshot.build failed/inline and still rejects the caller", async () => { + setSandboxFactory(() => ({ + mkdir: async () => {}, + writeFile: async () => {}, + readFile: async () => "", + destroy: async () => {}, + async exec(cmd) { + if (cmd.includes("install")) return { success: true, exitCode: 0, stdout: "", stderr: "" }; + return { success: false, exitCode: 1, stdout: "", stderr: "build blew up" }; + }, + })); + try { + const { env, points } = envWithPointCapture([], [], {}, { buildCacheHit: false }); + await assert.rejects(() => + createDemo(env, { + entry: ENTRY, + files: FILES, + htVersion: "18.1.0", + title: "A demo", + createdBy: "dev@handsontable.com", + now: new Date().toISOString(), + }), + ); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1, `expected exactly 1 snapshot.build point even on failure, got ${sb.length}`); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "failed"); + assert.equal(sb[0].blobs[REASON_SLOT], "inline"); + } finally { + setSandboxFactory(null); + } +}); + +test("runSnapshotJob (the DO alarm's detached path) still emits exactly one snapshot.build point, tagged reason=detached — not two, after moving emission into updateDemo", async () => { + const payload = JSON.stringify({ framework: "next.js", files: FILES }); + const { env, points } = envWithPointCapture([], [], { "demos/u1/__job.json": payload }); + await runSnapshotJob(env, { + demoId: "u1", + framework: "next.js", + htVersion: "16.2.0", + filesKey: "demos/u1/__job.json", + attempt: 0, + }); + const sb = snapshotBuildPoints(points); + assert.equal(sb.length, 1, `expected exactly 1 snapshot.build point (not 0, not 2), got ${sb.length}`); + assert.equal(sb[0].blobs[OUTCOME_SLOT], "ok"); + assert.equal(sb[0].blobs[REASON_SLOT], "detached"); + assert.equal(sb[0].blobs[FRAMEWORK_SLOT], "next.js"); +}); diff --git a/runner/pipeline/telemetry-ae-only-attrs.test.mjs b/runner/pipeline/telemetry-ae-only-attrs.test.mjs new file mode 100644 index 0000000000..eff8a5da21 --- /dev/null +++ b/runner/pipeline/telemetry-ae-only-attrs.test.mjs @@ -0,0 +1,97 @@ +// `hot.bucket`/`hot.reason`/`hot.fingerprint` survive the browser scrub +// (`scrubTelemetry`, `attrs.ts#AE_ONLY_ATTRIBUTE_KEYS` — never hoisted into +// a stored record) and land in their AE slots through `readAeOnlyAttrs` → +// `toAePoint`: `bucket` in `blob16`, `reason` in `blob9`, `fingerprint` in +// `blob11` (contract §3 AE-only keys, §4 slots). +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/telemetry-ae-only-attrs.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { scrubTelemetry, toAePoint } from "../packages/runtime/dist/telemetry/index.js"; + +register("./fixtures/o11y-worker-hooks.mjs", import.meta.url); +const { readAeOnlyAttrs } = await import("../workers/o11y/src/normalise/browser-attrs.ts"); + +const SERVICE = { service_name: "demos-authoring", service_version: "abc123", environment: "production" }; + +/** A Faro measurement item, shaped the way `attrsToContext` + Faro's + * `pushMeasurement` produce one — `context` uses the dotted keys + * `DOTTED_ATTR_KEY` now maps `bucket`/`reason`/`fingerprint` to. */ +function faroMeasurement(context) { + return { + type: "measurement", + payload: { values: { duration_ms: 1 }, timestamp: new Date().toISOString(), context }, + meta: { app: { name: "demos-authoring", version: "abc123", environment: "production" } }, + }; +} + +test("scrubTelemetry keeps hot.bucket/hot.reason/hot.fingerprint (the browser/re-run scrub allowlist)", () => { + const item = faroMeasurement({ + "hot.surface": "authoring", + "hot.bucket": "18.1", + "hot.reason": "17", + "hot.fingerprint": "sandpack.compile_error:deadbeefcafefeed", + }); + + const scrubbed = scrubTelemetry(item); + + assert.ok(scrubbed, "scrubTelemetry must not drop the whole item"); + assert.equal(scrubbed.payload.context?.["hot.surface"], "authoring", "sanity: an already-working key still survives"); + assert.equal(scrubbed.payload.context?.["hot.bucket"], "18.1", "guard: hot.bucket must survive the scrub allowlist"); + assert.equal(scrubbed.payload.context?.["hot.reason"], "17", "guard: hot.reason must survive the scrub allowlist"); + assert.equal( + scrubbed.payload.context?.["hot.fingerprint"], + "sandpack.compile_error:deadbeefcafefeed", + "guard: hot.fingerprint must survive the scrub allowlist", + ); +}); + +test("bucket/reason/fingerprint reach their §4 AE slots (blob16/blob9/blob11) through the real ingest conversion", () => { + // Each built from the SAME scrubbed context a real Faro request would now + // carry — `scrubTelemetry` runs first, exactly as `processOneItem`'s + // T00-D6 order does server-side (and as Faro's `beforeSend` does in the + // browser), so this exercises the real allowlist, not a hand-picked bag. + const readyScrubbed = scrubTelemetry( + faroMeasurement({ + "hot.surface": "authoring", + "hot.tier": "1", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.outcome": "ready", + "hot.bucket": "18.1", + }), + ); + const readyPoint = toAePoint( + "preview.ready_ms", + { duration_ms: 842 }, + { ...SERVICE, ...readAeOnlyAttrs(readyScrubbed.payload.context), surface: "authoring", tier: "1", framework: "react", ht_major: "18", outcome: "ready" }, + ); + assert.equal(readyPoint.blobs[15], "18.1", "guard: bucket must land in blob16 (index 15)"); + + const switchScrubbed = scrubTelemetry( + faroMeasurement({ "hot.framework": "react", "hot.ht_major": "18", "hot.reason": "17", "hot.bucket": "18.1" }), + ); + const switchPoint = toAePoint( + "version.switch", + { count: 1 }, + { ...SERVICE, ...readAeOnlyAttrs(switchScrubbed.payload.context), framework: "react", ht_major: "18" }, + ); + assert.equal(switchPoint.blobs[8], "17", "guard: reason must land in blob9 (index 8)"); + + const errorScrubbed = scrubTelemetry( + faroMeasurement({ + "hot.framework": "vue", + "hot.ht_major": "17", + "hot.fingerprint": "sandpack.compile_error:deadbeefcafefeed", + }), + ); + const errorPoint = toAePoint( + "sandpack.compile_error", + {}, + { ...SERVICE, ...readAeOnlyAttrs(errorScrubbed.payload.context), framework: "vue", ht_major: "17" }, + ); + assert.equal(errorPoint.blobs[10], "sandpack.compile_error:deadbeefcafefeed", "guard: fingerprint must land in blob11 (index 10)"); +}); diff --git a/runner/pipeline/telemetry-contract.test.mjs b/runner/pipeline/telemetry-contract.test.mjs new file mode 100644 index 0000000000..dd912b8966 --- /dev/null +++ b/runner/pipeline/telemetry-contract.test.mjs @@ -0,0 +1,367 @@ +// Proves `packages/runtime/src/telemetry/{attrs,metrics}.ts` agrees with +// `docs/observability-contract.md` §3, §4 and §5 — slot for slot, outcome +// for outcome. Parses the doc from disk (not a hand-copied fixture), so +// editing either side alone fails this file. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build` — +// this file imports the telemetry module from `packages/runtime/dist/` +// (the `../packages/runtime/dist/...` convention `dep-shims.test.mjs` and +// friends use). +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + AE_COLUMNS, + AE_ONLY_ATTRIBUTE_KEYS, + ALLOWED_ATTRIBUTE_KEYS, + DIAGNOSTIC_TAG_KEYS, + ENVIRONMENTS, + HOT_KINDS, + HT_MAJORS, + KNOWN_FRAMEWORKS, + METRICS, + METRIC_NAMES, + RESOURCE_ATTRS, + SERVICE_NAMES, + STRUCTURED_METADATA_KEYS, + SURFACES, + TIERS, +} from "../packages/runtime/dist/telemetry/index.js"; + +const dir = path.dirname(fileURLToPath(import.meta.url)); +const doc = fs.readFileSync(path.join(dir, "..", "docs", "observability-contract.md"), "utf8"); + +// ---- Generic markdown-table helpers --------------------------------------------- + +function section(startHeading, endHeading) { + const start = doc.indexOf(startHeading); + assert.notEqual(start, -1, `heading not found: ${startHeading}`); + const from = start + startHeading.length; + const end = doc.indexOf(endHeading, from); + assert.notEqual(end, -1, `heading not found: ${endHeading}`); + return doc.slice(from, end); +} + +/** Rows of a single markdown table within `text`: header + separator skipped, + * each data row split into trimmed cells. */ +function tableRows(text) { + const lines = text + .split("\n") + .map((l) => l.trim()) + .filter((l) => l.startsWith("|") && l.endsWith("|")); + return lines.slice(2).map((line) => line.slice(1, -1).split("|").map((c) => c.trim())); +} + +/** Split on `delimiter` at paren-depth 0 — a comma inside `(...)` (e.g. + * `reason (\`live\`, \`builder\`)`) does not split. */ +function splitTopLevel(cell, delimiter) { + const parts = []; + let depth = 0; + let current = ""; + for (const ch of cell) { + if (ch === "(") depth++; + if (ch === ")") depth--; + if (ch === delimiter && depth === 0) { + parts.push(current); + current = ""; + } else { + current += ch; + } + } + parts.push(current); + return parts.map((p) => p.trim()).filter((p) => p.length > 0); +} + +function backtickTokens(cell) { + return [...cell.matchAll(/`([^`]+)`/g)].map((m) => m[1]); +} + +// ---- §3: attributes --------------------------------------------------------- + +test("§3 resource attributes match attrs.ts, key for key, slot for slot, label for label", () => { + const body = section("## 3. Attributes", "## 4."); + const rows = tableRows(body); + const parsed = rows.map(([keyCell, valuesCell, labelCell, slotCell]) => ({ + key: keyCell.replace(/`/g, ""), + valuesCell, + lokiLabel: labelCell === "no" ? undefined : labelCell.replace(/`/g, ""), + aeSlot: slotCell.replace(/`/g, ""), + })); + + const byKey = new Map(RESOURCE_ATTRS.map((a) => [a.key, a])); + assert.deepEqual( + new Set(parsed.map((p) => p.key)), + new Set(byKey.keys()), + "§3 attribute keys and RESOURCE_ATTRS must name exactly the same set", + ); + + for (const p of parsed) { + const mod = byKey.get(p.key); + assert.equal(mod.lokiLabel, p.lokiLabel, `${p.key}: Loki label`); + assert.equal(mod.aeSlot, p.aeSlot, `${p.key}: AE slot`); + } + + // Closed-set values, only for the attributes the doc actually enumerates + // (hot.framework and hot.outcome are prose — "a key of config/frameworks.json…" + // and "per metric, see §5" — not enumerable here). + const closedSets = { + "service.name": SERVICE_NAMES, + "deployment.environment.name": ENVIRONMENTS, + "hot.surface": SURFACES, + "hot.tier": TIERS, + "hot.ht_major": HT_MAJORS, + }; + for (const p of parsed) { + const expected = closedSets[p.key]; + if (!expected) continue; + const range = /`(\d+)`\s*…\s*`(\d+)`/.exec(p.valuesCell); + let values = backtickTokens(p.valuesCell); + if (range) { + const lo = Number(range[1]); + const hi = Number(range[2]); + const expanded = []; + for (let n = lo; n <= hi; n++) expanded.push(String(n)); + values = [...expanded, ...values.filter((v) => v !== range[1] && v !== range[2])]; + } + assert.deepEqual(new Set(values), new Set(expected), `${p.key}: closed-set values`); + } +}); + +test("§3: KNOWN_FRAMEWORKS is exactly the config/frameworks.json keys plus none", () => { + const catalog = JSON.parse(fs.readFileSync(path.join(dir, "..", "config", "frameworks.json"), "utf8")); + assert.deepEqual( + new Set(KNOWN_FRAMEWORKS), + new Set([...Object.keys(catalog.frameworks), "none"]), + "a new frameworks.json key must be added to attrs.ts#KNOWN_FRAMEWORKS, or its records store hot.framework=other", + ); +}); + +/** A marker paragraph's text (up to the next blank line), for the §3 + * categories below that are prose, not a table. Called inside each test + * (never at module load) so a missing/renamed marker fails just that one + * test, rather than throwing during collection and skipping every test + * after it in the file. */ +function markerParagraph(marker) { + const start = doc.indexOf(marker); + assert.notEqual(start, -1, `${marker} paragraph not found`); + const end = doc.indexOf("\n\n", start); + return doc.slice(start, end === -1 ? undefined : end); +} + +function structuredMetadataDocKeys() { + return backtickTokens(markerParagraph("Structured metadata only")).filter((t) => /^[a-z]+\.[a-z_]+$/.test(t)); +} + +// Flat, non-dotted keys only (excludes the structured-metadata paragraph's +// dotted `hot.*`/`session.id`/`cf.ray` keys by construction — they never +// match this shape). +function diagnosticTagsDocKeys() { + return backtickTokens(markerParagraph("Diagnostic tags")).filter((t) => /^[a-z][a-z_]*$/.test(t)); +} + +function aeOnlyDocKeys() { + return backtickTokens(markerParagraph("AE-only transport keys")).filter((t) => /^hot\.[a-z_]+$/.test(t)); +} + +function resourceAttrDocKeys() { + return tableRows(section("## 3. Attributes", "## 4.")).map(([keyCell]) => keyCell.replace(/`/g, "")); +} + +test("§3 structured-metadata-only keys match STRUCTURED_METADATA_KEYS", () => { + assert.deepEqual(new Set(structuredMetadataDocKeys()), new Set(STRUCTURED_METADATA_KEYS)); + + // "`hot.kind` (the Faro item kind: `exception`, `log`, `event`, `measurement`)" + // — HOT_KINDS is attrs.ts's closed set for this key; pin it to the doc too. + const paragraph = markerParagraph("Structured metadata only"); + const hotKindIdx = paragraph.indexOf("`hot.kind`"); + assert.notEqual(hotKindIdx, -1, "hot.kind not found in the structured-metadata paragraph"); + const hotKindValues = backtickTokens(paragraph.slice(hotKindIdx + "`hot.kind`".length)); + assert.deepEqual(new Set(hotKindValues), new Set(HOT_KINDS)); +}); + +// A third §3 category, flat/non-dotted, distinct from +// STRUCTURED_METADATA_KEYS — see attrs.ts's own doc comment on +// DIAGNOSTIC_TAG_KEYS for how these are hoisted alongside structured metadata. +test("§3 diagnostic tag keys match DIAGNOSTIC_TAG_KEYS", () => { + assert.deepEqual(new Set(diagnosticTagsDocKeys()), new Set(DIAGNOSTIC_TAG_KEYS)); +}); + +// A fourth §3 category — keys that survive the browser/ingest attribute +// allowlist (`ALLOWED_ATTRIBUTE_KEYS`) but are never hoisted to a resource +// attribute or structured metadata at all (`attrs.ts`'s own doc comment on +// `AE_ONLY_ATTRIBUTE_KEYS`). Pinned separately from the two tests above so +// the doc's "AE-only transport keys" paragraph cannot drift from +// `AE_ONLY_ATTRIBUTE_KEYS` unnoticed. +test("§3 AE-only transport keys match AE_ONLY_ATTRIBUTE_KEYS", () => { + assert.deepEqual(new Set(aeOnlyDocKeys()), new Set(AE_ONLY_ATTRIBUTE_KEYS)); +}); + +// The allowlist `scrub.ts#allowlistAttributes` actually enforces is exactly +// the union of every §3 category documented above (resource attrs table + +// the three prose paragraphs) — proving each category against the doc +// separately (the tests above) does not by itself prove the code's allowlist +// has no fifth, undocumented member, or that a category was accidentally +// left out of `ALLOWED_ATTRIBUTE_KEYS`. Built from the DOC-PARSED key lists +// above, not from the module's own constants, so this only passes when the +// code's `ALLOWED_ATTRIBUTE_KEYS` matches what the doc actually says. +test("ALLOWED_ATTRIBUTE_KEYS equals the union of every documented §3 category", () => { + const documentedUnion = new Set([ + ...resourceAttrDocKeys(), + ...structuredMetadataDocKeys(), + ...diagnosticTagsDocKeys(), + ...aeOnlyDocKeys(), + ]); + assert.deepEqual(new Set(ALLOWED_ATTRIBUTE_KEYS), documentedUnion); +}); + +// ---- §4: Analytics Engine layout ---------------------------------------------- + +test("§4 Analytics Engine layout matches AE_COLUMNS, column for column, slot for slot", () => { + const body = section("## 4. Analytics Engine layout", "## 5."); + const rows = tableRows(body); + const parsed = {}; + for (const [slotCell, columnCell] of rows) { + if (columnCell.trim() === "—") continue; // an unassigned slot (or range) + parsed[columnCell.replace(/`/g, "")] = slotCell.replace(/`/g, ""); + } + assert.deepEqual(parsed, { ...AE_COLUMNS }); +}); + +// ---- §5: metric registry ------------------------------------------------------- + +function parseBlobsCell(cell) { + if (cell.trim() === "—") return { blobs: [], values: {} }; + const blobs = []; + const values = {}; + for (const token of splitTopLevel(cell, ",")) { + const eq = /^([a-z_.]+)\s*=\s*`([^`]+)`$/i.exec(token); + if (eq) { + blobs.push(eq[1]); + values[eq[1]] = [eq[2]]; + continue; + } + const paren = /^([a-z_.]+)\s*\(([^)]*)\)$/i.exec(token); + if (paren) { + blobs.push(paren[1]); + const inner = backtickTokens(paren[2]); + if (inner.length > 0) values[paren[1]] = inner; + continue; + } + const plain = /^([a-z_.]+)$/i.exec(token); + if (plain) { + blobs.push(plain[1]); + continue; + } + throw new Error(`telemetry-contract.test.mjs: unparseable Blobs token "${token}" in "${cell}"`); + } + return { blobs, values }; +} + +function parseDoublesCell(cell) { + if (cell.trim() === "—") return []; + return splitTopLevel(cell, ",").map((token) => { + const m = /^([a-z_]+)/i.exec(token); + if (!m) throw new Error(`telemetry-contract.test.mjs: unparseable Doubles token "${token}" in "${cell}"`); + return m[1]; + }); +} + +function parseOutcomesCell(cell, blobs) { + const result = { outcome: undefined, reason: undefined, outcomeAlias: undefined }; + if (cell.trim() === "—") return result; + const hasOutcome = blobs.includes("outcome"); + const hasReason = blobs.includes("reason"); + let unlabeledUsed = false; + for (const clause of splitTopLevel(cell, ";")) { + const alias = /^outcomes\s+as\s+`([^`]+)`$/i.exec(clause); + if (alias) { + result.outcomeAlias = alias[1]; + continue; + } + if (/^reason\s*=/i.test(clause)) { + // "reason = gate" — explicitly open, no fixed values. + result.reason = result.reason ?? []; + continue; + } + if (/^reason:?\s+/i.test(clause)) { + result.reason = [...(result.reason ?? []), ...backtickTokens(clause)]; + continue; + } + if (/^outcome:?\s+/i.test(clause)) { + result.outcome = [...(result.outcome ?? []), ...backtickTokens(clause)]; + continue; + } + const tokens = backtickTokens(clause); + if (tokens.length === 0) { + throw new Error(`telemetry-contract.test.mjs: unparseable Outcomes/reason clause "${clause}"`); + } + if (unlabeledUsed) { + throw new Error(`telemetry-contract.test.mjs: two unlabeled clauses in "${cell}"`); + } + unlabeledUsed = true; + if (hasOutcome) result.outcome = [...(result.outcome ?? []), ...tokens]; + else if (hasReason) result.reason = [...(result.reason ?? []), ...tokens]; + else throw new Error(`telemetry-contract.test.mjs: unlabeled values but no outcome/reason blob: "${cell}"`); + } + return result; +} + +function parseMetricRow([namesCell, , blobsCell, doublesCell, outcomesCell]) { + const names = splitTopLevel(namesCell, ",").map((n) => n.replace(/`/g, "")); + const { blobs, values: blobValues } = parseBlobsCell(blobsCell); + const doubles = parseDoublesCell(doublesCell); + const parsedOutcomes = parseOutcomesCell(outcomesCell, blobs); + + const values = { ...blobValues }; + if (parsedOutcomes.outcome) values.outcome = [...(values.outcome ?? []), ...parsedOutcomes.outcome]; + if (parsedOutcomes.reason) values.reason = [...(values.reason ?? []), ...parsedOutcomes.reason]; + + return { names, blobs, doubles, values, outcomeAlias: parsedOutcomes.outcomeAlias }; +} + +/** Drop empty-array entries (an explicit "open, no fixed values" marker, e.g. + * "reason = gate") so they compare equal to a genuinely absent key, and sort + * every surviving array so order never matters. */ +function normalizeValues(values) { + const out = {}; + for (const [k, v] of Object.entries(values)) { + if (Array.isArray(v) && v.length > 0) out[k] = [...v].sort(); + } + return out; +} + +test("§5 metric registry matches METRICS, slot for slot and outcome for outcome", () => { + const body = section("## 5. Metric registry", "## 6."); + const rawRows = tableRows(body).map(parseMetricRow); + + const byName = new Map(); + for (const row of rawRows) { + for (const name of row.names) { + byName.set(name, { blobs: row.blobs, doubles: row.doubles, values: row.values, outcomeAlias: row.outcomeAlias }); + } + } + + for (const [name, entry] of byName) { + if (!entry.outcomeAlias) continue; + const target = byName.get(entry.outcomeAlias); + assert.ok(target, `${name}: outcome alias "${entry.outcomeAlias}" not found in the table`); + entry.values = { ...entry.values, outcome: target.values.outcome }; + } + + assert.deepEqual( + new Set(byName.keys()), + new Set(METRIC_NAMES), + "§5 metric names and MetricName must name exactly the same set", + ); + + for (const [name, entry] of byName) { + const mod = METRICS[name]; + assert.deepEqual(new Set(entry.blobs), new Set(mod.blobs), `${name}: blobs`); + assert.deepEqual(new Set(entry.doubles), new Set(mod.doubles), `${name}: doubles`); + assert.deepEqual(normalizeValues(entry.values), normalizeValues(mod.values ?? {}), `${name}: values (outcome/reason/fixed)`); + } +}); diff --git a/runner/pipeline/telemetry-convert.test.mjs b/runner/pipeline/telemetry-convert.test.mjs new file mode 100644 index 0000000000..7288c0169a --- /dev/null +++ b/runner/pipeline/telemetry-convert.test.mjs @@ -0,0 +1,343 @@ +// Observability contract §6 (Faro item → OTLP log record) and §9 (beacon → OTLP +// log record). +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { + beaconToRecord, + clampTimestampMs, + faroItemToRecord, + formatStackFrame, + hoistAttributes, + isValidOpenAttrValue, + msToUnixNano, + sanitizeResourceAttributes, + scrubTelemetry, +} from "../packages/runtime/dist/telemetry/index.js"; + +const SERVICE = { name: "demos-authoring", version: "abc123def456", environment: "production" }; +const RECEIVED_AT_MS = Date.UTC(2026, 8, 23, 12, 0, 0); // fixed, so "twice" never races a clock +const PREVIEW_HOST = "3000-sbx7f2a-tok9xQ.demos.handsontable.com"; + +test("clampTimestampMs keeps a candidate inside the 5-minute window", () => { + const candidate = RECEIVED_AT_MS - 60_000; // 1 minute earlier + assert.equal(clampTimestampMs(candidate, RECEIVED_AT_MS), candidate); +}); + +test("clampTimestampMs falls back outside the 5-minute window", () => { + const tooOld = RECEIVED_AT_MS - 6 * 60_000; + assert.equal(clampTimestampMs(tooOld, RECEIVED_AT_MS), RECEIVED_AT_MS); +}); + +test("clampTimestampMs falls back when the candidate is absent", () => { + assert.equal(clampTimestampMs(undefined, RECEIVED_AT_MS), RECEIVED_AT_MS); +}); + +test("msToUnixNano converts exactly, at ordinary timestamp magnitudes", () => { + const ms = 1_695_463_200_123; + assert.equal(msToUnixNano(ms), "1695463200123000000"); +}); + +test("msToUnixNano stays an exact integer string even where `ms * 1e6` as a plain double would not", () => { + // `ms` alone is a safe integer (well under 2^53); `ms * 1e6` computed in + // double precision overflows the mantissa and prints in exponential + // notation ("8e+21", `String(8_000_000_000_000_000 * 1e6)`) — a value OTLP's + // JSON `fixed64` mapping cannot parse back as a timestamp. BigInt keeps it + // exact. Not a realistic wall-clock date, but a real boundary the function + // must not get wrong. + const ms = 8_000_000_000_000_000; + assert.equal(msToUnixNano(ms), "8000000000000000000000"); +}); + +test("hoistAttributes splits a merged bag into resource attrs vs structured metadata, dropping anything else", () => { + const { resourceAttributes, attributes } = hoistAttributes({ + "hot.surface": "authoring", + "hot.ht_major": "18", + "hot.demo_id": "r-react-18-0-0", + "session.id": "plid-1", + "not.a.contract.key": "should be dropped", + }); + assert.deepEqual(resourceAttributes, { "hot.surface": "authoring", "hot.ht_major": "18" }); + assert.deepEqual(attributes, { "hot.demo_id": "r-react-18-0-0", "session.id": "plid-1" }); +}); + +test("hoistAttributes also keeps the authoring app's diagnostic tag keys as structured metadata", () => { + // `scrub.ts#allowlistAttributes` (via `attrs.ts#ALLOWED_ATTRIBUTE_KEYS`) has + // allowed `handled`/`context`/`sentry_event_id`/the `versions-fetch` tags + // through, but `hoistAttributes` — the very next + // step in both the Faro and OTLP ingest paths — had its own narrower key + // set and silently dropped them again. + const { resourceAttributes, attributes } = hoistAttributes({ + "hot.surface": "authoring", + handled: "true", + context: "tier1-compiler-asset", + sentry_event_id: "abc123def456", + versions_fetch_outcome: "ok", + "not.a.contract.key": "still dropped", + }); + assert.deepEqual(resourceAttributes, { "hot.surface": "authoring" }); + assert.deepEqual(attributes, { + handled: "true", + context: "tier1-compiler-asset", + sentry_event_id: "abc123def456", + versions_fetch_outcome: "ok", + }); +}); + +// ---- label/service forgery on the ingest path ------------------------------- + +test("sanitizeResourceAttributes drops an out-of-enum closed-set value, keeps a valid one", () => { + const out = sanitizeResourceAttributes({ + "hot.surface": "o11y", + "hot.tier": "zzz", // not in TIERS + "hot.ht_major": "18", + }); + assert.deepEqual(out, { "hot.surface": "o11y", "hot.ht_major": "18" }); +}); + +test("sanitizeResourceAttributes stores an over-length/bad-charset hot.framework or hot.outcome as \"other\"", () => { + const tooLong = "F".repeat(3000); + const out = sanitizeResourceAttributes({ "hot.framework": tooLong, "hot.outcome": "attacker-", + }, + }, + }); + const record = faroItemToRecord(spoofed, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + // The route's own identity, never the client's. + assert.equal(record.resourceAttributes["service.name"], "demos-authoring"); + assert.equal(record.resourceAttributes["deployment.environment.name"], "production"); + // Out-of-enum / over-length hot.* values never reach the record verbatim. + assert.equal(record.resourceAttributes["hot.tier"], undefined); + assert.equal(record.resourceAttributes["hot.framework"], "other"); + // A log never carries a metric outcome; ingest fills the `none` default. + assert.equal(record.resourceAttributes["hot.outcome"], undefined); + // A legitimate closed-set value survives untouched. + assert.equal(record.resourceAttributes["hot.surface"], "o11y"); +}); + +test("beaconToRecord: an over-length hot.framework (fw) is stored as \"other\", not verbatim — lite path", () => { + const record = beaconToRecord( + { + v: 1, + s: "embed", + demo: "r-react-18-0-0", + ht: "18", + fw: "F".repeat(3000), + n: "TypeError", + m: "x", + val: null, + dev: "desktop", + ts: RECEIVED_AT_MS, + }, + { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }, + ); + assert.equal(record.resourceAttributes["hot.framework"], "other"); +}); + +function faroLogItem(overrides = {}) { + return { + type: "log", + payload: { + message: "session started", + timestamp: new Date(RECEIVED_AT_MS - 1000).toISOString(), + context: { + "hot.surface": "authoring", + "hot.tier": "1", + "hot.framework": "react", + "hot.ht_major": "18", + "hot.demo_id": "r-react-18-0-0", + "session.id": "plid-abc", + }, + ...overrides.payload, + }, + meta: {}, + ...overrides, + }; +} + +test("faroItemToRecord hoists hot.* to resourceAttributes and adds the service.* resource attrs", () => { + const record = faroItemToRecord(faroLogItem(), { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.resourceAttributes["service.name"], "demos-authoring"); + assert.equal(record.resourceAttributes["service.version"], "abc123def456"); + assert.equal(record.resourceAttributes["deployment.environment.name"], "production"); + assert.equal(record.resourceAttributes["hot.surface"], "authoring"); + assert.equal(record.resourceAttributes["hot.tier"], "1"); + assert.equal(record.resourceAttributes["hot.framework"], "react"); + assert.equal(record.resourceAttributes["hot.ht_major"], "18"); + // Structured metadata never lands as a resource attribute. + assert.equal(record.resourceAttributes["hot.demo_id"], undefined); + assert.equal(record.resourceAttributes["session.id"], undefined); + assert.deepEqual(record.attributes, { + "hot.demo_id": "r-react-18-0-0", + "session.id": "plid-abc", + "hot.kind": "log", + }); +}); + +test("faroItemToRecord always sets hot.kind from item.type (§3's closed set), overwriting a client-sent value", () => { + const item = faroLogItem({ payload: { context: { "hot.kind": "not-a-real-kind" } } }); + const record = faroItemToRecord(item, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.attributes["hot.kind"], "log"); +}); + +test("faroItemToRecord throws on an item.type outside §3's closed set, e.g. 'trace' (Faro's own enum allows it; the contract does not)", () => { + const traceItem = { type: "trace", payload: {}, meta: {} }; + assert.throws(() => faroItemToRecord(traceItem, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }), /not a valid hot\.kind/); + assert.throws( + () => faroItemToRecord({ ...faroLogItem(), type: "not-a-real-kind" }, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }), + /not a valid hot\.kind/, + ); +}); + +test("faroItemToRecord clamps the event timestamp to the receive window", () => { + const farInPast = faroLogItem({ payload: { timestamp: new Date(RECEIVED_AT_MS - 3_600_000).toISOString() } }); + const record = faroItemToRecord(farInPast, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.timeUnixNano, msToUnixNano(RECEIVED_AT_MS)); +}); + +test("faroItemToRecord body: exception carries its type, log carries its message", () => { + const exceptionItem = { + type: "exception", + payload: { type: "TypeError", value: "x is not a function", timestamp: new Date(RECEIVED_AT_MS).toISOString() }, + meta: {}, + }; + const record = faroItemToRecord(exceptionItem, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.body, "TypeError: x is not a function"); + + const logRecord = faroItemToRecord(faroLogItem(), { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(logRecord.body, "session started"); +}); + +// ---- a lineno < 1 is not a real position ------------------------------------- +// +// `@jridgewell/trace-mapping#originalPositionFor` throws on `line: 0` at +// drain time (`workers/o11y/src/drain/symbolicate.ts`'s own guard is the real +// defence against that). This drops only the invalid position on the ingest side, +// so the drain-time regex (`symbolicate.ts#STACK_LINE_RE`) never even +// captures a `line: 0` for a normalised record — without changing the +// contract shape (a frame with no numeric position already renders exactly +// this way, see the "no real position" case just below). +test("formatStackFrame: a lineno of 0 is dropped (no position rendered), not passed through as an invalid source position", () => { + const line = formatStackFrame({ + filename: "https://demos.handsontable.com/assets/index-abc123.js", + function: "f", + lineno: 0, + colno: 5, + }); + assert.equal(line, " at f (https://demos.handsontable.com/assets/index-abc123.js)"); +}); + +test("formatStackFrame: a negative or non-finite lineno is also dropped", () => { + const negative = formatStackFrame({ filename: "https://demos.handsontable.com/x.js", function: "f", lineno: -5, colno: 1 }); + const nonFinite = formatStackFrame({ filename: "https://demos.handsontable.com/x.js", function: "f", lineno: Infinity, colno: 1 }); + assert.equal(negative, " at f (https://demos.handsontable.com/x.js)"); + assert.equal(nonFinite, " at f (https://demos.handsontable.com/x.js)"); +}); + +test("formatStackFrame: a normal, valid lineno/colno still renders its position exactly as before", () => { + const line = formatStackFrame({ + filename: "https://demos.handsontable.com/assets/index-abc123.js", + function: "f", + lineno: 12, + colno: 34, + }); + assert.equal(line, " at f (https://demos.handsontable.com/assets/index-abc123.js:12:34)"); +}); + +test("faroItemToRecord is byte-identical converting the same item twice", () => { + const item = faroLogItem(); + const a = faroItemToRecord(item, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + const b = faroItemToRecord(item, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(JSON.stringify(a), JSON.stringify(b)); +}); + +function litePayload(overrides = {}) { + return { + v: 1, + t: "err", + s: "embed", + demo: "r-react-18-0-0", + ht: "18", + fw: "react", + n: "TypeError", + m: "x is not a function", + val: null, + dev: "desktop", + ts: RECEIVED_AT_MS - 500, + ...overrides, + }; +} + +test("beaconToRecord hoists s/ht/fw to resourceAttributes, hot.tier is always static", () => { + const record = beaconToRecord(litePayload(), { service: { ...SERVICE, name: "demos-embed" }, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.resourceAttributes["hot.surface"], "embed"); + assert.equal(record.resourceAttributes["hot.tier"], "static"); + assert.equal(record.resourceAttributes["hot.framework"], "react"); + assert.equal(record.resourceAttributes["hot.ht_major"], "18"); + assert.equal(record.resourceAttributes["service.name"], "demos-embed"); + assert.deepEqual(record.attributes, { "hot.demo_id": "r-react-18-0-0", "hot.kind": "exception" }); +}); + +test("beaconToRecord clamps its own ts the same way", () => { + const stale = litePayload({ ts: RECEIVED_AT_MS - 3_600_000 }); + const record = beaconToRecord(stale, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.timeUnixNano, msToUnixNano(RECEIVED_AT_MS)); +}); + +test("beaconToRecord is byte-identical converting the same payload twice", () => { + const payload = litePayload(); + const a = beaconToRecord(payload, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + const b = beaconToRecord(payload, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(JSON.stringify(a), JSON.stringify(b)); +}); + +test("beaconToRecord marks a vital beacon's hot.kind as measurement, not exception", () => { + const vital = litePayload({ t: "vital", n: "LCP", m: undefined, val: 2200 }); + const record = beaconToRecord(vital, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.equal(record.attributes["hot.kind"], "measurement"); + assert.equal(record.body, "LCP=2200"); +}); + +// ---- a beacon does not typecheck as scrubTelemetry's argument at all (it +// is neither Faro- nor OTLP-shaped) — convert, then scrub, the opposite +// order from a Faro item. + +test("beaconToRecord -> scrubTelemetry cleans a code frame in m and a preview host in st", () => { + const dirty = litePayload({ + m: "unknown: Unexpected token (1:10)\n\n> 1 | const x = ;\n | ^", + st: `at https://${PREVIEW_HOST}/src/main.js`, + }); + const record = beaconToRecord(dirty, { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + const scrubbed = scrubTelemetry(record); + assert.equal(scrubbed.body, "TypeError: unknown: Unexpected token (1:10)\nat https:///src/main.js"); +}); + +test("scrubTelemetry is a no-op on an already-clean faroItemToRecord output (idempotent at the boundary)", () => { + const record = faroItemToRecord(faroLogItem(), { service: SERVICE, receivedAtMs: RECEIVED_AT_MS }); + assert.deepEqual(scrubTelemetry(record), record); +}); diff --git a/runner/pipeline/telemetry-facade-boot-safety.test.mjs b/runner/pipeline/telemetry-facade-boot-safety.test.mjs new file mode 100644 index 0000000000..055864b248 --- /dev/null +++ b/runner/pipeline/telemetry-facade-boot-safety.test.mjs @@ -0,0 +1,56 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { pathToFileURL } from "node:url"; +import { join } from "node:path"; + +// `pipeline/telemetry-facade-noop.test.mjs` pins `noopTelemetry.pageLoadId()`'s +// contract (stable across calls) but cannot reproduce or guard the actual +// boot crash: Node has no "no I/O in module scope" restriction the way +// workerd does, so a test that only calls `pageLoadId()` after import +// passes identically whether the mint is eager (module top level) or lazy +// (inside the function). This file closes that gap: stub +// `globalThis.crypto.randomUUID` before importing the module, and assert +// the import itself never calls it — the one thing that distinguishes +// "eager IIFE" from "lazy accessor". +// +// A fresh module evaluation is required for the "import alone" assertion +// to mean anything (a cached module from an earlier import would already +// have run its top-level code before this file's stub was installed) — +// cache-busted via a query string on the specifier, the same technique +// this repo uses elsewhere for a rotated chunk URL. + +const facadeUrl = pathToFileURL( + join(import.meta.dirname, "..", "packages/runtime/dist/telemetry/facade.js"), +).href; + +function stubRandomUUID() { + const original = globalThis.crypto.randomUUID; + let calls = 0; + globalThis.crypto.randomUUID = (...args) => { + calls += 1; + return original.apply(globalThis.crypto, args); + }; + return { + count: () => calls, + restore: () => { globalThis.crypto.randomUUID = original; }, + }; +} + +test("importing facade.js alone never calls crypto.randomUUID (the eager-IIFE regression)", async () => { + const stub = stubRandomUUID(); + try { + // Cache-busted: a fresh module instance, so its top-level code (if any) + // runs AFTER the stub above is installed, not before. + const mod = await import(`${facadeUrl}?bust=${Date.now()}-${Math.random()}`); + assert.equal(stub.count(), 0, "importing the module must not itself mint a page-load id"); + // The lazy accessor still has to work, and only now: first call mints + // (count -> 1), every later call reuses the same id (count stays 1). + const first = mod.noopTelemetry.pageLoadId(); + assert.equal(stub.count(), 1, "the first pageLoadId() call mints exactly once"); + const second = mod.noopTelemetry.pageLoadId(); + assert.equal(stub.count(), 1, "a second call must not mint again"); + assert.equal(first, second); + } finally { + stub.restore(); + } +}); diff --git a/runner/pipeline/telemetry-facade-noop.test.mjs b/runner/pipeline/telemetry-facade-noop.test.mjs new file mode 100644 index 0000000000..ee59c962bc --- /dev/null +++ b/runner/pipeline/telemetry-facade-noop.test.mjs @@ -0,0 +1,49 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { noopTelemetry, recordingTelemetry } from "../packages/runtime/dist/telemetry/facade.js"; + +// `noopTelemetry` (`packages/runtime/src/telemetry/facade.ts`) must not +// mint its page-load id in a module-top-level IIFE — `const pageLoadId = +// mintPageLoadId()` running at import time, calling `crypto.randomUUID()` +// outside any handler. Harmless under plain Node, but measured against a +// real `wrangler dev` running the API worker: workerd refuses +// "asynchronous I/O ... and generating random values ... within global +// scope" and the whole Worker fails to boot. `pipeline/`'s Node harness +// cannot reproduce that failure mode (Node has no such restriction) — this +// only pins the contract noopTelemetry must keep (mint once, stay stable), +// not the boot crash itself. + +test("noopTelemetry.pageLoadId() is stable across calls", () => { + const a = noopTelemetry.pageLoadId(); + const b = noopTelemetry.pageLoadId(); + assert.equal(a, b); + assert.ok(a.length > 0); +}); + +test("noopTelemetry.metric/event/error are no-ops that never throw", () => { + assert.doesNotThrow(() => noopTelemetry.metric("api.request", { count: 1 }, {})); + assert.doesNotThrow(() => noopTelemetry.event("example.open", {})); + assert.doesNotThrow(() => noopTelemetry.error(new Error("x"), "ctx")); +}); + +test("recordingTelemetry still mints its own id eagerly (a test double, not the boot path)", () => { + // Unlike noopTelemetry, recordingTelemetry's default id is minted at call + // time (inside a function, called by a test), never at module import — no + // fix needed here, and this pins that the two are not accidentally merged. + const rec = recordingTelemetry(); + assert.equal(typeof rec.pageLoadId(), "string"); + assert.ok(rec.pageLoadId().length > 0); +}); + +test("recordingTelemetry records what it is given", () => { + const rec = recordingTelemetry("fixed-id"); + rec.metric("api.request", { count: 1 }, { route_class: "api/versions", outcome: "2xx" }); + rec.event("example.open", { kind: "starter" }); + rec.error(new Error("boom"), "ctx", { surface: "api" }); + assert.equal(rec.pageLoadId(), "fixed-id"); + assert.deepEqual(rec.metrics, [ + { name: "api.request", values: { count: 1 }, attrs: { route_class: "api/versions", outcome: "2xx" } }, + ]); + assert.equal(rec.events.length, 1); + assert.equal(rec.errors.length, 1); +}); diff --git a/runner/pipeline/telemetry-fingerprint.test.mjs b/runner/pipeline/telemetry-fingerprint.test.mjs new file mode 100644 index 0000000000..416972932e --- /dev/null +++ b/runner/pipeline/telemetry-fingerprint.test.mjs @@ -0,0 +1,118 @@ +// Observability contract §7 — `fingerprint`, `stripCodeFrame`, +// `feedsNewFingerprintAlert`. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { + fingerprint, + isValidFingerprint, + stripCodeFrame, + feedsNewFingerprintAlert, +} from "../packages/runtime/dist/telemetry/index.js"; + +// Captured verbatim from a real `@babel/standalone` `transform()` throw +// (`babel.transform("const x = ;", { presets: ["env"] })` and a two-line +// variant), not hand-typed — the exact shape `codeFrameColumns` renders. +const SINGLE_LINE_FRAME = "unknown: Unexpected token (1:10)\n\n> 1 | const x = ;\n | ^"; +const MULTI_LINE_FRAME = + "unknown: Unexpected token (2:12)\n\n 1 | function f() {\n> 2 | return x +;\n | ^\n 3 | }\n 4 |"; + +test("stripCodeFrame removes a real Babel code frame, keeping the message", () => { + assert.equal(stripCodeFrame(SINGLE_LINE_FRAME), "unknown: Unexpected token (1:10)"); + assert.equal(stripCodeFrame(MULTI_LINE_FRAME), "unknown: Unexpected token (2:12)"); +}); + +test("stripCodeFrame is a no-op on text with no gutter or caret lines", () => { + assert.equal(stripCodeFrame("TypeError: x is not a function"), "TypeError: x is not a function"); +}); + +test("fingerprint pins FNV-1a 64 to the published test vectors, via messages normalizeMonitorMessage leaves untouched", () => { + // normalizeMonitorMessage leaves a message with no url/timestamp/quote/digit + // untouched, so these exercise the raw hash exactly. + assert.equal(fingerprint("ctx", ""), "ctx:cbf29ce484222325"); + assert.equal(fingerprint("ctx", "a"), "ctx:af63dc4c8601ec8c"); + assert.equal(fingerprint("ctx", "foobar"), "ctx:85944171f73967e8"); +}); + +test("fingerprint is deterministic for the same context and message", () => { + assert.equal(fingerprint("demo-runtime", "boom"), fingerprint("demo-runtime", "boom")); +}); + +test("fingerprint collapses a demo-runtime keystroke ladder to one shape", () => { + const rungs = ["t", "tr", "tru", "truthy"].map((id) => + fingerprint("demo-runtime", `${id} is not defined`), + ); + const distinct = new Set(rungs); + assert.equal(distinct.size, 1, `expected one fingerprint for the ladder, got ${distinct.size}: ${[...distinct]}`); +}); + +test("fingerprint does not collapse two genuinely different messages", () => { + const a = fingerprint("demo-runtime", "Cannot read properties of undefined (reading 'x')"); + const b = fingerprint("demo-runtime", "Maximum call stack size exceeded"); + assert.notEqual(a, b); +}); + +test("fingerprint strips a code frame before hashing, so a frame-bearing and a frame-free message with the same text fingerprint the same", () => { + const withFrame = fingerprint("api", `Unexpected token\n\n${SINGLE_LINE_FRAME.split("\n\n")[1]}`); + const withoutFrame = fingerprint("api", "Unexpected token"); + assert.equal(withFrame, withoutFrame); +}); + +test("feedsNewFingerprintAlert excludes only demo-runtime", () => { + assert.equal(feedsNewFingerprintAlert("demo-runtime"), false); + for (const surface of ["authoring", "share", "embed", "d", "api", "o11y"]) { + assert.equal(feedsNewFingerprintAlert(surface), true, surface); + } +}); + +// ---- one shared validator, `:` allowed inside `context` --------------------- +// +// A validator anchored on the first `:` and rejecting a second one +// anywhere in `context` would silently discard every real call site below, +// which sends a `:`-joined call-site path. Every literal context string +// here is grepped verbatim from the real call sites: +// `apps/authoring/src/App.tsx`'s `reportError(error, +// "docs-example-load:fetch" | "docs-example-load:path" | +// "docs-bucket-resolve:bucket" | "docs-bucket-resolve:fetch")`, and +// `workers/api/src/index.ts`'s `reportDiagnostic(..., { context: +// "npm-registry:version-exists" | "npm-registry:versions" })`. +const REAL_MULTI_SEGMENT_CONTEXTS = [ + "docs-example-load:fetch", + "docs-example-load:path", + "docs-bucket-resolve:bucket", + "docs-bucket-resolve:fetch", + "npm-registry:version-exists", + "npm-registry:versions", +]; + +test("isValidFingerprint accepts every real multi-segment context call site, computed through the real fingerprint() function (never a hand-typed hex string)", () => { + for (const ctx of REAL_MULTI_SEGMENT_CONTEXTS) { + const fp = fingerprint(ctx, "upstream request failed"); + assert.ok(isValidFingerprint(fp), `${ctx} -> ${fp} must be valid`); + // The context half must survive verbatim — this is what a forgotten + // `:`-anchor-on-the-FIRST-colon bug would silently truncate. + assert.ok(fp.startsWith(`${ctx}:`), `expected ${fp} to start with "${ctx}:"`); + } +}); + +test("isValidFingerprint still accepts a single-segment context (the common case, unchanged)", () => { + assert.ok(isValidFingerprint(fingerprint("authoring", "boom"))); + assert.ok(isValidFingerprint(fingerprint("sandpack.compile_error", "boom")), "a dotted metric-name context (metrics.ts)"); +}); + +test("isValidFingerprint rejects a forged/injection-shaped value", () => { + assert.equal(isValidFingerprint(" N "), false); + assert.equal(isValidFingerprint("authoring:not-hex-at-all!!"), false); + assert.equal(isValidFingerprint("AUTHORING:0123456789abcdef"), false, "uppercase context is not the contract's own charset"); + assert.equal(isValidFingerprint("authoring:0123456789ABCDEF"), false, "the hex half must be lowercase"); + assert.equal(isValidFingerprint(":0123456789abcdef"), false, "context must not be empty"); + assert.equal(isValidFingerprint("authoring:"), false, "hex half must not be empty"); +}); + +test("isValidFingerprint rejects an over-long value, even one that otherwise matches the shape", () => { + const overLong = `${"a".repeat(500)}:0123456789abcdef`; + assert.equal(isValidFingerprint(overLong), false); +}); diff --git a/runner/pipeline/telemetry-inbox.test.mjs b/runner/pipeline/telemetry-inbox.test.mjs new file mode 100644 index 0000000000..73f62516f2 --- /dev/null +++ b/runner/pipeline/telemetry-inbox.test.mjs @@ -0,0 +1,104 @@ +// Observability contract §8 — `inboxKey`/`parseInboxKey` (UTC, not local time), +// NDJSON encode/decode round-trip, and `buildResourceLogs`'s shape. +// +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { + buildResourceLogs, + cleanMarkerKey, + decodeNdjson, + encodeNdjson, + inboxKey, + parseInboxKey, +} from "../packages/runtime/dist/telemetry/index.js"; + +test("inboxKey builds the exact §8 shape", () => { + const date = new Date(Date.UTC(2026, 8, 23, 14, 5, 0)); + assert.equal(inboxKey("browser", date, 7), "inbox/browser/2026-09-23/14/000000000007.ndjson.gz"); +}); + +test("inboxKey uses UTC, not local time — this machine's zone (CEST, UTC+2) would put 23:30 UTC in tomorrow's local date", () => { + // 2026-09-23T23:30:00Z is 2026-09-24, 01:30 in CEST (UTC+2). + const date = new Date(Date.UTC(2026, 8, 23, 23, 30, 0)); + const key = inboxKey("worker", date, 1); + assert.equal(key, "inbox/worker/2026-09-23/23/000000000001.ndjson.gz"); +}); + +test("inboxKey pads the sequence to 12 digits", () => { + const date = new Date(Date.UTC(2026, 0, 1, 0, 0, 0)); + assert.equal(inboxKey("browser", date, 42), "inbox/browser/2026-01-01/00/000000000042.ndjson.gz"); +}); + +test("parseInboxKey is the exact inverse of inboxKey", () => { + const date = new Date(Date.UTC(2026, 8, 23, 14, 5, 0)); + const key = inboxKey("browser", date, 7); + assert.deepEqual(parseInboxKey(key), { tenant: "browser", date: "2026-09-23", hour: "14", seq: 7 }); +}); + +test("parseInboxKey rejects a key that does not match the shape", () => { + assert.equal(parseInboxKey("inbox/browser/2026-09-23/14/7.ndjson.gz"), null); // seq not 12 digits + assert.equal(parseInboxKey("inbox/other-tenant/2026-09-23/14/000000000007.ndjson.gz"), null); + assert.equal(parseInboxKey("not a key at all"), null); +}); + +test("cleanMarkerKey builds the §8 state path", () => { + assert.equal(cleanMarkerKey("wake-abc123"), "state/wakes/wake-abc123/clean"); +}); + +test("encodeNdjson / decodeNdjson round-trip exactly", () => { + const records = [ + buildResourceLogs({ + body: "hello", + timeUnixNano: "1695463200000000000", + resourceAttributes: { "service.name": "demos-o11y" }, + }), + buildResourceLogs({ + body: "world", + timeUnixNano: "1695463201000000000", + resourceAttributes: { "service.name": "demos-o11y" }, + attributes: { "hot.demo_id": "r-react-18-0-0" }, + }), + ]; + const encoded = encodeNdjson(records); + assert.equal(encoded.split("\n").filter(Boolean).length, 2); + assert.deepEqual(decodeNdjson(encoded), records); +}); + +test("encodeNdjson of an empty array is an empty string, not a bare newline", () => { + assert.equal(encodeNdjson([]), ""); +}); + +test("decodeNdjson skips blank lines", () => { + const one = buildResourceLogs({ body: "x", timeUnixNano: "1", resourceAttributes: {} }); + const withBlankLines = `\n${JSON.stringify(one)}\n\n`; + assert.deepEqual(decodeNdjson(withBlankLines), [one]); +}); + +test("buildResourceLogs wraps exactly one log record, resource attrs first", () => { + const rl = buildResourceLogs({ + body: "boom", + timeUnixNano: "1695463200000000000", + resourceAttributes: { "service.name": "demos-o11y", "hot.surface": "o11y" }, + attributes: { "cf.ray": "abc123" }, + severityText: "ERROR", + }); + assert.equal(rl.scopeLogs.length, 1); + assert.equal(rl.scopeLogs[0].logRecords.length, 1); + const record = rl.scopeLogs[0].logRecords[0]; + assert.equal(record.body.stringValue, "boom"); + assert.equal(record.timeUnixNano, "1695463200000000000"); + assert.equal(record.severityText, "ERROR"); + assert.deepEqual(record.attributes, [{ key: "cf.ray", value: { stringValue: "abc123" } }]); + assert.deepEqual(rl.resource.attributes, [ + { key: "service.name", value: { stringValue: "demos-o11y" } }, + { key: "hot.surface", value: { stringValue: "o11y" } }, + ]); +}); + +test("buildResourceLogs omits `attributes` entirely when there are none (not an empty array)", () => { + const rl = buildResourceLogs({ body: "x", timeUnixNano: "1", resourceAttributes: {} }); + assert.equal(rl.scopeLogs[0].logRecords[0].attributes, undefined); +}); diff --git a/runner/pipeline/telemetry-lite.test.mjs b/runner/pipeline/telemetry-lite.test.mjs new file mode 100644 index 0000000000..6b29261638 --- /dev/null +++ b/runner/pipeline/telemetry-lite.test.mjs @@ -0,0 +1,130 @@ +// Observability contract §9 — the lite beacon payload validator, including +// the total-size cap (2048 bytes, decisive over the per-field caps). +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { + LITE_MESSAGE_MAX, + LITE_PAYLOAD_MAX_BYTES, + LITE_STACK_MAX, + isValidLitePayload, +} from "../packages/runtime/dist/telemetry/index.js"; + +function errPayload(overrides = {}) { + return { + v: 1, + t: "err", + s: "embed", + demo: "r-react-18-0-0", + ht: "18", + fw: "react", + n: "TypeError", + m: "x is not a function", + val: null, + dev: "desktop", + ts: Date.now(), + ...overrides, + }; +} + +test("accepts a well-formed error payload", () => { + assert.equal(isValidLitePayload(errPayload()), true); +}); + +test("accepts a well-formed vital payload", () => { + assert.equal( + isValidLitePayload({ + v: 1, + t: "vital", + s: "d", + demo: "r-react-18-0-0", + ht: "18", + fw: "react", + n: "LCP", + val: 2200, + dev: "mobile", + ts: Date.now(), + }), + true, + ); +}); + +test("rejects a non-object, null, and a wrong v", () => { + assert.equal(isValidLitePayload(null), false); + assert.equal(isValidLitePayload("nope"), false); + assert.equal(isValidLitePayload(errPayload({ v: 2 })), false); +}); + +test("rejects an unknown surface", () => { + assert.equal(isValidLitePayload(errPayload({ s: "authoring" })), false); +}); + +test("rejects a vital name that isn't one of the four", () => { + const bad = errPayload({ t: "vital", n: "FID", val: 10 }); + assert.equal(isValidLitePayload(bad), false); +}); + +test("rejects err with a non-null val", () => { + assert.equal(isValidLitePayload(errPayload({ val: 1 })), false); +}); + +test("rejects a message over LITE_MESSAGE_MAX", () => { + assert.equal(isValidLitePayload(errPayload({ m: "x".repeat(LITE_MESSAGE_MAX + 1) })), false); + assert.equal(isValidLitePayload(errPayload({ m: "x".repeat(LITE_MESSAGE_MAX) })), true); +}); + +test("rejects a stack over LITE_STACK_MAX", () => { + assert.equal(isValidLitePayload(errPayload({ st: "x".repeat(LITE_STACK_MAX + 1) })), false); + // A stack well under the field cap, in an otherwise small payload, is fine — + // note a stack *at* LITE_STACK_MAX is not asserted valid here: at 2000 chars + // it already exceeds LITE_PAYLOAD_MAX_BYTES on its own once the rest of the + // payload's fields are counted (see the case below), which is exactly + // the "field caps don't promise they fit together" the module documents. + assert.equal(isValidLitePayload(errPayload({ st: "x".repeat(200) })), true); +}); + +test("rejects a payload whose individual fields all pass their own caps but whose total exceeds LITE_PAYLOAD_MAX_BYTES", () => { + // m at its own cap (500) + st at its own cap (2000) is already ~2500+ bytes + // of JSON — comfortably over the 2048-byte total. + const maxed = errPayload({ m: "m".repeat(LITE_MESSAGE_MAX), st: "s".repeat(LITE_STACK_MAX) }); + assert.equal(isValidLitePayload(maxed), false, "each field cap alone must not be enough"); + assert.ok(new TextEncoder().encode(JSON.stringify(maxed)).length > LITE_PAYLOAD_MAX_BYTES); +}); + +test("a payload built to actually fit under the total cap passes", () => { + const fits = errPayload({ m: "short message", st: "s".repeat(1500) }); + assert.ok(new TextEncoder().encode(JSON.stringify(fits)).length <= LITE_PAYLOAD_MAX_BYTES); + assert.equal(isValidLitePayload(fits), true); +}); + +// The per-beacon `id` prevents byte-identical beacons thrown in the same +// millisecond from being deduped as one (`workers/o11y/src/lite.ts`'s +// `hashRecord` call). Absent entirely for an old/cached reporter — must +// stay accepted — and, when present, a +// `Math.random().toString(36).slice(2,10)` value: 0-16 lowercase base-36 +// characters, with `""` a legitimate value (`Math.random()` landing on +// exactly 0), never a missing one. + +test("accepts a payload with no id at all (old reporter)", () => { + const noId = errPayload(); + assert.equal("id" in noId, false, "precondition: errPayload() carries no id field"); + assert.equal(isValidLitePayload(noId), true); +}); + +test("accepts a well-formed id", () => { + assert.equal(isValidLitePayload(errPayload({ id: "a1b2c3d4" })), true); +}); + +test("accepts an empty-string id (Math.random() landing on exactly 0)", () => { + assert.equal(isValidLitePayload(errPayload({ id: "" })), true); +}); + +test("rejects a malformed id — uppercase, over 16 characters, or non-string", () => { + assert.equal(isValidLitePayload(errPayload({ id: "ABCDEFGH" })), false, "uppercase must be rejected"); + assert.equal(isValidLitePayload(errPayload({ id: "a".repeat(16) })), true, "exactly 16 chars is still valid"); + assert.equal(isValidLitePayload(errPayload({ id: "a".repeat(17) })), false, "over 16 chars must be rejected"); + assert.equal(isValidLitePayload(errPayload({ id: 12345678 })), false, "a number must be rejected, not coerced"); + assert.equal(isValidLitePayload(errPayload({ id: null })), false, "null must be rejected, unlike undefined"); +}); diff --git a/runner/pipeline/telemetry-metrics.test.mjs b/runner/pipeline/telemetry-metrics.test.mjs new file mode 100644 index 0000000000..ef06fa8778 --- /dev/null +++ b/runner/pipeline/telemetry-metrics.test.mjs @@ -0,0 +1,115 @@ +// `toAePoint` (observability contract §4/§5) — the runtime behaviour +// `telemetry-contract.test.mjs` does not cover (that file checks the +// static registry data against the doc; this checks what building a point +// actually does with it): positional slot layout, and the validation that +// rejects an outcome/reason/value not listed for the metric. +// Build prerequisite: `pnpm --filter @handsontable/demo-runtime build`. +// Run: node --experimental-strip-types --test pipeline/*.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { AE_COLUMNS, toAePoint } from "../packages/runtime/dist/telemetry/index.js"; + +const SERVICE = { service_name: "demos-authoring", service_version: "abc123", environment: "production" }; + +test("toAePoint writes a fixed-width point (20 blobs, 20 doubles) regardless of how many are used", () => { + const point = toAePoint("hmr.roundtrip_ms", { duration_ms: 42 }, { ...SERVICE, framework: "react", ht_major: "18" }); + assert.equal(point.blobs.length, 20); + assert.equal(point.doubles.length, 20); + assert.equal(point.indexes.length, 1); + assert.equal(point.indexes[0], "hmr.roundtrip_ms"); +}); + +test("toAePoint places each value at its AE_COLUMNS slot", () => { + const point = toAePoint( + "preview.ready_ms", + { duration_ms: 1234 }, + { ...SERVICE, surface: "authoring", tier: "1", framework: "react", ht_major: "18", outcome: "ready", bucket: "18.1" }, + ); + const blobIndex = (col) => Number(AE_COLUMNS[col].replace("blob", "")) - 1; + const doubleIndex = (col) => Number(AE_COLUMNS[col].replace("double", "")) - 1; + + assert.equal(point.blobs[blobIndex("service_name")], "demos-authoring"); + assert.equal(point.blobs[blobIndex("surface")], "authoring"); + assert.equal(point.blobs[blobIndex("outcome")], "ready"); + assert.equal(point.blobs[blobIndex("bucket")], "18.1"); + assert.equal(point.doubles[doubleIndex("duration_ms")], 1234); + // Untouched blob slots stay at the documented default — but double1 (count) + // is never "untouched": see the dedicated tests below. + assert.equal(point.blobs[blobIndex("reason")], ""); +}); + +test("toAePoint defaults double1 (count) to 1, even for a metric whose own §5 row never lists count", () => { + // preview.ready_ms's Doubles column is just duration_ms (§5) — count is + // still universal (§4's reading rule: "1 per point unless pre-aggregated"), + // or every count-based query (alert thresholds, dashboard panels) reads + // zero for this metric forever. + const point = toAePoint( + "preview.ready_ms", + { duration_ms: 1234 }, + { ...SERVICE, surface: "authoring", tier: "1", framework: "react", ht_major: "18", outcome: "ready" }, + ); + const doubleIndex = (col) => Number(AE_COLUMNS[col].replace("double", "")) - 1; + assert.equal(point.doubles[doubleIndex("count")], 1); +}); + +test("toAePoint lets an explicit count override the default 1 (a pre-aggregated point)", () => { + const point = toAePoint("sandpack.compile_error", { count: 5 }, { ...SERVICE, framework: "react", ht_major: "18" }); + const doubleIndex = (col) => Number(AE_COLUMNS[col].replace("double", "")) - 1; + assert.equal(point.doubles[doubleIndex("count")], 5); +}); + +test("toAePoint rejects an outcome not in the metric's allowed set", () => { + assert.throws( + () => toAePoint("sandpack.compile_ms", {}, { ...SERVICE, tier: "1", framework: "react", ht_major: "18", outcome: "bogus" }), + /not an allowed "outcome"/, + ); +}); + +test("toAePoint rejects an outcome for a metric with no outcome slot", () => { + assert.throws( + () => toAePoint("hmr.roundtrip_ms", {}, { ...SERVICE, framework: "react", ht_major: "18", outcome: "ready" }), + /has no "outcome" slot/, + ); +}); + +test("toAePoint accepts an open (unenumerated) reason, e.g. import.url's provider-shaped reason", () => { + const point = toAePoint("import.url", { count: 1 }, { ...SERVICE, provider: "jsfiddle", outcome: "ok", reason: "anything goes here" }); + const blobIndex = (col) => Number(AE_COLUMNS[col].replace("blob", "")) - 1; + assert.equal(point.blobs[blobIndex("reason")], "anything goes here"); +}); + +test("toAePoint rejects an unknown metric name", () => { + assert.throws(() => toAePoint("not.a.real.metric", {}, SERVICE), /unknown metric/); +}); + +test("toAePoint rejects a value outside a repo-wide closed set (surface), for a metric that actually declares surface", () => { + assert.throws( + () => + toAePoint( + "preview.ready_ms", + {}, + { ...SERVICE, surface: "not-a-real-surface", tier: "1", framework: "react", ht_major: "18", outcome: "ready" }, + ), + /not an allowed "surface"/, + ); +}); + +test("toAePoint enforces preview.runtime_error's fixed surface value", () => { + assert.throws( + () => + toAePoint( + "preview.runtime_error", + { count: 1 }, + { ...SERVICE, surface: "authoring", tier: "1", framework: "react", ht_major: "18", reason: "uncaught" }, + ), + /not an allowed "surface"/, + ); + const point = toAePoint( + "preview.runtime_error", + { count: 1 }, + { ...SERVICE, surface: "demo-runtime", tier: "1", framework: "react", ht_major: "18", reason: "uncaught" }, + ); + const blobIndex = (col) => Number(AE_COLUMNS[col].replace("blob", "")) - 1; + assert.equal(point.blobs[blobIndex("surface")], "demo-runtime"); +}); diff --git a/runner/pipeline/telemetry-sink.test.mjs b/runner/pipeline/telemetry-sink.test.mjs new file mode 100644 index 0000000000..9fc55dc729 --- /dev/null +++ b/runner/pipeline/telemetry-sink.test.mjs @@ -0,0 +1,96 @@ +// `AeSink` (observability contract §4/§10) — `memorySink`, and +// `clickhouseSink`'s wire format against the real clickhouse-init.sql +// schema, cross-checked against a throwaway ClickHouse container (raw +// epoch-millisecond `timestamp`, not seconds or a string — see `sink.ts`'s +// doc comment). Credential headers and non-2xx rejection are asserted too, +// since a sink that only checks whether `fetch` threw resolves on a `403`. + +import test from "node:test"; +import assert from "node:assert/strict"; +import { clickhouseSink, clickhouseTimestamp, memorySink, toAePoint } from "../packages/runtime/dist/telemetry/index.js"; + +test("memorySink collects every point written, in order", () => { + const sink = memorySink(); + const a = toAePoint("chat.edit", { count: 1 }, { service_name: "demos-api", service_version: "x", environment: "production", outcome: "proposed" }); + const b = toAePoint("chat.edit", { count: 1 }, { service_name: "demos-api", service_version: "x", environment: "production", outcome: "applied" }); + sink.writeDataPoint(a); + sink.writeDataPoint(b); + assert.equal(sink.points.length, 2); + assert.deepEqual(sink.points, [a, b]); +}); + +test("clickhouseTimestamp is a raw epoch-millisecond integer, not a formatted string", () => { + const date = new Date(Date.UTC(2026, 8, 23, 12, 0, 0, 123)); + assert.equal(clickhouseTimestamp(date), 1790164800123); + assert.equal(typeof clickhouseTimestamp(date), "number"); +}); + +test("clickhouseSink POSTs one JSONEachRow line with the runner_events table's exact column names", async () => { + const calls = []; + const fetchImpl = async (url, init) => { + calls.push({ url, init }); + return { ok: true }; + }; + const sink = clickhouseSink("http://localhost:8123", { fetchImpl }); + const point = toAePoint( + "api.request", + { count: 3, duration_ms: 42 }, + { service_name: "demos-api", service_version: "x", environment: "production", route_class: "api/versions", outcome: "2xx" }, + ); + await sink.writeDataPoint(point); + + assert.equal(calls.length, 1); + const url = new URL(calls[0].url); + assert.equal(url.origin, "http://localhost:8123"); + assert.match(url.searchParams.get("query"), /INSERT INTO runner_events FORMAT JSONEachRow/); + + const row = JSON.parse(calls[0].init.body.trim()); + assert.equal(typeof row.timestamp, "number"); + assert.ok(row.timestamp > 1_700_000_000_000, `timestamp looks like epoch ms: ${row.timestamp}`); + assert.equal(row._sample_interval, 1); + assert.equal(row.index1, "api.request"); + assert.equal(row.blob1, "demos-api"); + assert.equal(row.blob8, "2xx"); + assert.equal(row.blob10, "api/versions"); + assert.equal(row.double1, 3); + assert.equal(row.double2, 42); + // Every column the DDL declares is a plain blobN/doubleN key — no + // friendly name (e.g. "outcome") ever appears as a JSON key. + assert.equal(row.outcome, undefined); + assert.equal(row.metric, undefined); +}); + +test("clickhouseSink sends X-ClickHouse-User/-Key when credentials are given", async () => { + const calls = []; + const fetchImpl = async (url, init) => { + calls.push({ url, init }); + return { ok: true }; + }; + const sink = clickhouseSink("http://localhost:8123", { fetchImpl, user: "default", password: "local-dev-token" }); + const point = toAePoint("chat.edit", { count: 1 }, { service_name: "demos-api", service_version: "x", environment: "production", outcome: "proposed" }); + await sink.writeDataPoint(point); + + assert.equal(calls[0].init.headers["X-ClickHouse-User"], "default"); + assert.equal(calls[0].init.headers["X-ClickHouse-Key"], "local-dev-token"); +}); + +test("clickhouseSink sends no credential headers when none are given", async () => { + const calls = []; + const fetchImpl = async (url, init) => { + calls.push({ url, init }); + return { ok: true }; + }; + const sink = clickhouseSink("http://localhost:8123", { fetchImpl }); + const point = toAePoint("chat.edit", { count: 1 }, { service_name: "demos-api", service_version: "x", environment: "production", outcome: "proposed" }); + await sink.writeDataPoint(point); + + assert.equal(calls[0].init.headers["X-ClickHouse-User"], undefined); + assert.equal(calls[0].init.headers["X-ClickHouse-Key"], undefined); +}); + +test("clickhouseSink rejects on a non-2xx response instead of resolving silently (the real bug: an auth failure was swallowed)", async () => { + const fetchImpl = async () => ({ ok: false, status: 403, text: async () => "Code: 516. DB::Exception: Authentication failed" }); + const sink = clickhouseSink("http://localhost:8123", { fetchImpl }); + const point = toAePoint("chat.edit", { count: 1 }, { service_name: "demos-api", service_version: "x", environment: "production", outcome: "proposed" }); + await assert.rejects(() => sink.writeDataPoint(point), /clickhouseSink: insert failed, 403/); +}); diff --git a/runner/pipeline/theme-ai-network-error.test.mjs b/runner/pipeline/theme-ai-network-error.test.mjs new file mode 100644 index 0000000000..a204b28436 --- /dev/null +++ b/runner/pipeline/theme-ai-network-error.test.mjs @@ -0,0 +1,86 @@ +// A network-level throw from the LiteLLM fetch in `/api/theme` must not +// vanish: `requestTheme` (theme-ai.ts) must wrap its own `fetch()` call, or +// a connection failure (DNS, refused, reset — a real `TypeError`, not a +// `ChatUnavailableError`) propagates straight past `index.ts`'s +// `if (err instanceof ChatUnavailableError)` guard to `throw err`, reaching +// the generic fetch catch-all with no `theme.ai` point at all — contract §5 +// promises one on every outcome, `error` included, matching +// `chat.answer`'s twin catch. +// +// Driven through the real router (`workers/api/src/index.ts`'s default +// export) — a re-declared copy of the catch would not catch this regressing. +// Run: node --experimental-strip-types --test pipeline/theme-ai-network-error.test.mjs + +import test from "node:test"; +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { ctx, makeEnv } from "./fixtures/worker-harness.mjs"; + +register("./fixtures/worker-hooks.mjs", import.meta.url); + +const { default: worker } = await import("../workers/api/src/index.ts"); + +const HOST = "https://demos.handsontable.com"; + +// blob/double slot positions, §4 (AE_COLUMNS, packages/runtime/src/telemetry/metrics.ts): +// model=blob13 (index 12), outcome=blob8 (index 7), count=double1 (index 0). +const MODEL_SLOT = 12; +const OUTCOME_SLOT = 7; +const COUNT_SLOT = 0; + +function envWithPointCapture() { + const points = []; + const { env, ...rest } = makeEnv(); + env.RUNNER_EVENTS = { writeDataPoint: (p) => points.push(p) }; + // Routes AE writes through the in-memory sink instead of the local-ClickHouse + // HTTP fallback `serviceEnvironment` selects for a non-production host — see + // the identical comment on `snapshot-build-point.test.mjs`'s own helper. + env.PREVIEW_HOST = "demos.handsontable.com"; + env.LITELLM_API_KEY = "test-key"; + return { env, points, ...rest }; +} + +function themeAiPoints(points) { + return points.filter((p) => p.indexes[0] === "theme.ai"); +} + +test("a network-level throw from the LiteLLM fetch still emits theme.ai outcome=error", async () => { + const { env, points } = envWithPointCapture(); + + const realFetch = globalThis.fetch; + globalThis.fetch = async (input) => { + const url = typeof input === "string" ? input : input.url; + if (url.includes("litellm")) { + // The exact failure mode this test exists for: `fetch()` itself + // rejects (connection refused / DNS / reset), never resolving to a + // Response — so `requestTheme`'s `if (!res.ok)` branch (the one that + // throws `ChatUnavailableError`) is never reached at all. + throw new TypeError("fetch failed: network connection lost"); + } + throw new Error(`unexpected network fetch in theme-ai-network-error.test.mjs: ${url}`); + }; + + try { + const res = await worker.fetch( + new Request(`${HOST}/api/theme`, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ prompt: "make it purple", current: {} }), + }), + env, + ctx, + ); + // The raw throw is not a ChatUnavailableError, so it still falls through + // to the generic fetch catch-all (a 500) — that part of the behaviour is + // pre-existing and out of scope here. The point is what to fix. + assert.equal(res.status, 500); + + const points_ = themeAiPoints(points); + assert.equal(points_.length, 1, `expected exactly 1 theme.ai point, got ${points_.length}`); + assert.equal(points_[0].blobs[OUTCOME_SLOT], "error"); + assert.equal(points_[0].blobs[MODEL_SLOT], "unknown"); + assert.equal(points_[0].doubles[COUNT_SLOT], 1); + } finally { + globalThis.fetch = realFetch; + } +}); diff --git a/runner/pnpm-lock.yaml b/runner/pnpm-lock.yaml index 40c1d27656..4b28b41fc0 100644 --- a/runner/pnpm-lock.yaml +++ b/runner/pnpm-lock.yaml @@ -29,6 +29,9 @@ importers: '@fontsource/fira-code': specifier: ^5.3.0 version: 5.3.0 + '@grafana/faro-web-sdk': + specifier: 2.12.1 + version: 2.12.1 '@handsontable/demo-editor-shell': specifier: workspace:* version: link:../../packages/editor-shell @@ -160,6 +163,37 @@ importers: specifier: ^4.108.0 version: 4.108.0(@cloudflare/workers-types@4.20260702.1) + workers/o11y: + dependencies: + '@bufbuild/protobuf': + specifier: 2.15.0 + version: 2.15.0 + '@cloudflare/containers': + specifier: 0.3.7 + version: 0.3.7 + '@handsontable/demo-runtime': + specifier: workspace:* + version: link:../../packages/runtime + '@jridgewell/trace-mapping': + specifier: 0.3.31 + version: 0.3.31 + jose: + specifier: 6.2.12 + version: 6.2.12 + devDependencies: + '@cloudflare/workers-types': + specifier: 5.20260923.1 + version: 5.20260923.1 + source-map-js: + specifier: 1.2.1 + version: 1.2.1 + typescript: + specifier: ~5.6.0 + version: 5.6.3 + wrangler: + specifier: 4.136.3 + version: 4.136.3(@cloudflare/workers-types@5.20260923.1)(@types/node@22.20.1) + packages: '@apm-js-collab/code-transformer-bundler-plugins@0.7.1': @@ -263,6 +297,9 @@ packages: resolution: {integrity: sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==} engines: {node: '>=6.9.0'} + '@bufbuild/protobuf@2.15.0': + resolution: {integrity: sha512-DAheWUkVr/SJTWCc+lg9dhY0eN4SaWlf4+bG1KzHeXbnqt0AfB/NX0Z+VunGlM1ki1B4zVvye27MpKh/svySUA==} + '@cloudflare/containers@0.3.7': resolution: {integrity: sha512-DM9dm3FnIBSyiSJ1FLavKwl/lk3oAmTaynCzZQ9pZR0ncRPquSxkxd8Nu2MFILxmDDsPkxKsSNEh9mHHMty4Fw==} @@ -293,39 +330,81 @@ packages: workerd: optional: true + '@cloudflare/unenv-preset@2.16.2': + resolution: {integrity: sha512-JBP1+Z7ZSNG/d4mRP+y8VC5dka3tZVMLEZRvS+rzQ4DGV1EoxRFQckcJTTkXbHSQiTj0DtNI01Zwb/V2fX0mvQ==} + peerDependencies: + unenv: 2.0.0-rc.24 + workerd: '>1.20260305.0 <2.0.0-0' + peerDependenciesMeta: + workerd: + optional: true + '@cloudflare/workerd-darwin-64@1.20260706.1': resolution: {integrity: sha512-61XleG5EaE+Zam2Y/fwOcMInMrjd3QMKIG0/+pHycZcVZC5Oxs5GfHSTnvyhcniigGW/bcIkvn1PptWbxlVleQ==} engines: {node: '>=16'} cpu: [x64] os: [darwin] + '@cloudflare/workerd-darwin-64@1.20260921.1': + resolution: {integrity: sha512-3iB2WnYOlZ29T+1zhCwbHFExCBp6E9bgmDUMryATYwrIGEQ1YbvR78m4ydm56XKN/d/yF3803ivMGfZMYDtiMg==} + engines: {node: '>=16'} + cpu: [x64] + os: [darwin] + '@cloudflare/workerd-darwin-arm64@1.20260706.1': resolution: {integrity: sha512-jbwuMEuhgrXEk+9MD8eHPRsXWlxqtNMqqCMBLJauTHk+NR+U2O6hJXElPqlAVlUTh0A0GPM+DfATupRXm7Ml8w==} engines: {node: '>=16'} cpu: [arm64] os: [darwin] + '@cloudflare/workerd-darwin-arm64@1.20260921.1': + resolution: {integrity: sha512-FpqVR7IQXVBmGtajyonEmhmb5UAsmV7dTaIkpemmHZXHEw7uYpkhkzKPjc4BOPhNQy8iwt2p+RZBPMY3Y7/bvQ==} + engines: {node: '>=16'} + cpu: [arm64] + os: [darwin] + '@cloudflare/workerd-linux-64@1.20260706.1': resolution: {integrity: sha512-kppCFkt6WHTNmcE83aUj4chWV13hdQKOgXxn53sqpLiVnPi6HEIOWB1DNRAljG9IDt3KMgni9HdzCYVzq2drJA==} engines: {node: '>=16'} cpu: [x64] os: [linux] + '@cloudflare/workerd-linux-64@1.20260921.1': + resolution: {integrity: sha512-riAJIohaVp5A8Sqy4yKlzHOaLPOICMf5oey+jC2rm45RVT+wK8+7UU0d31Dy/02Nc8YUkobAFwNVjX06P8WQ5g==} + engines: {node: '>=16'} + cpu: [x64] + os: [linux] + '@cloudflare/workerd-linux-arm64@1.20260706.1': resolution: {integrity: sha512-H4xpocM1vXNfIVrP/JiOhLooWBXi5erE81hBj2wmCvPIUcf+49CuvZEthpVwFYn9SFop/Y8+bUiV0vVrWf8xuw==} engines: {node: '>=16'} cpu: [arm64] os: [linux] + '@cloudflare/workerd-linux-arm64@1.20260921.1': + resolution: {integrity: sha512-tnJu08tT7s0XWDqp3O0H/vCp0voy9OqVAzspb89biMo1dh8IiEpnyXnoPmdJ7H4qBnXCmXgy0kuEphuvpDPj9w==} + engines: {node: '>=16'} + cpu: [arm64] + os: [linux] + '@cloudflare/workerd-windows-64@1.20260706.1': resolution: {integrity: sha512-SPWuI+kBNtr/mhcKE7XOECgRWOaidmzVmvtmDrRZX5Y4/+XqD/heRAU5Oa17wOZm3eQ1CB8Xzxjrr/6uQaP8YA==} engines: {node: '>=16'} cpu: [x64] os: [win32] + '@cloudflare/workerd-windows-64@1.20260921.1': + resolution: {integrity: sha512-VgNcRPstoZMb1G94JTrx+jU24GtkkazNfox0gnF/2fkuXpcfW/M0e0xvdMovYfwt8ZxG5AB2ZNvanD6ufBwiuQ==} + engines: {node: '>=16'} + cpu: [x64] + os: [win32] + '@cloudflare/workers-types@4.20260702.1': resolution: {integrity: sha512-mOhf5TUEB1m2vPrxtqoIGfz0fUC9xyxRDx5gWHy5s+OCo6dcV+g7wI1R7gYCMFohhqF/2y2xeKVwMwCJjfn/WA==} + '@cloudflare/workers-types@5.20260923.1': + resolution: {integrity: sha512-zuOqyxfhzhN2LlxxcpnAS79aGVm9ku3ZcWKyvVLHyjPV6oHsuHSFidzSHhHE9ho9UTcUxU1IdtpdVuFXluKB+g==} + '@codemirror/autocomplete@6.20.3': resolution: {integrity: sha512-tlosUqb+3BbxCxZdu4tKeRghPFC+QM7q4X5YhKV2eCmPG+1r2F3f4AaSz5sCrFqUtX4Jh20VFTKecl16MgiV9g==} @@ -378,6 +457,9 @@ packages: '@emnapi/runtime@1.11.2': resolution: {integrity: sha512-kyOl3X0DuTiT1h2ft8r2fYO8JYtU9a9Xis/zBSiGArNaagCOWx90N1k2wxp18czFDH+OgcWGb5ZP/XMt3dcyPA==} + '@emnapi/runtime@1.11.3': + resolution: {integrity: sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA==} + '@esbuild/aix-ppc64@0.21.5': resolution: {integrity: sha512-1SDgH6ZSPTlggy1yI6+Dbkiz8xzpHJEVAlF/AM1tHPLsf5STom9rwtjE4hKAF20FfXXNTFqEYXyJNWh1GiZedQ==} engines: {node: '>=12'} @@ -831,6 +913,12 @@ packages: '@fontsource/fira-code@5.3.0': resolution: {integrity: sha512-EJL968RJRkakubAj/coU8pSUaeTE5UNoRjtzAr6kGiSZ3jWuN8/AKWHwym/PFUaQL1q7IL/H+EXs4358YhrTBQ==} + '@grafana/faro-core@2.12.1': + resolution: {integrity: sha512-Pg0vjgsR1BxdmLzzX9r8YcZ8rI4aXLlDGtreppZev++BbnH2utKBwCI16Vx+IzfQqHKRT+7WGc63HOue8/9J6Q==} + + '@grafana/faro-web-sdk@2.12.1': + resolution: {integrity: sha512-JnpZy/ztfkP3UZ3AW0DzUxVgMRTBvYz+m0oZuVi+KRpGTYz0GGF0umi0S0pAemnaUbQBo6umGdCm5CL6n8BAWQ==} + '@img/colour@1.1.0': resolution: {integrity: sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ==} engines: {node: '>=18'} @@ -841,70 +929,145 @@ packages: cpu: [arm64] os: [darwin] + '@img/sharp-darwin-arm64@0.35.4': + resolution: {integrity: sha512-Uhfl4V4lhP2nbUVF9+hyH1+luj86f1gUFeo8ALYxFoULoU+G87D43BfeMP8XHsk9boxAnCY/bf2EHwhA7MuGsA==} + engines: {node: '>=20.9.0'} + cpu: [arm64] + os: [darwin] + '@img/sharp-darwin-x64@0.34.5': resolution: {integrity: sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [x64] os: [darwin] + '@img/sharp-darwin-x64@0.35.4': + resolution: {integrity: sha512-hWniXY3bG5qKpkKrAwPe4y+VTPmf086YQAnkxWh7uA1YrlRouWGa0M0Mxj3ZjnXFkv7/TD1bTy9lGUK26vRvWw==} + engines: {node: '>=20.9.0'} + cpu: [x64] + os: [darwin] + + '@img/sharp-freebsd-wasm32@0.35.4': + resolution: {integrity: sha512-lIsKw/BU+kjB4eZjxrYrZmwOJYi3Ajrv66iAlBmUPyKc3HpnloevB1g3wxGD9P/5BbQ1brBGl65VRRrCvQDEqA==} + engines: {node: '>=20.9.0'} + os: [freebsd] + '@img/sharp-libvips-darwin-arm64@1.2.4': resolution: {integrity: sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g==} cpu: [arm64] os: [darwin] + '@img/sharp-libvips-darwin-arm64@1.3.3': + resolution: {integrity: sha512-suTBPTDGrI9WodccaDdwZItTSaBYASlBk1NSfElSHrUfzu3szG6lvIF58+WiFvnfzuK8ZBFS5zE00PxqxnRiPg==} + cpu: [arm64] + os: [darwin] + '@img/sharp-libvips-darwin-x64@1.2.4': resolution: {integrity: sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg==} cpu: [x64] os: [darwin] + '@img/sharp-libvips-darwin-x64@1.3.3': + resolution: {integrity: sha512-FVJZ5mITMobmXIz/hPDTw0EintTW5H3WfrxwLqEqjiIihlu+hVRyGrFQ60xl0Lxn7Bt3zdpevPaQi0HEzqz9fw==} + cpu: [x64] + os: [darwin] + '@img/sharp-libvips-linux-arm64@1.2.4': resolution: {integrity: sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw==} cpu: [arm64] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-arm64@1.3.3': + resolution: {integrity: sha512-0DaL0A6Xu6sQSQFwe4iVCrKWU2cCTItnRsYsCdxAMm9NF6twAA9BKnoqy4hqz4+azQ0JHuA26qiUKsf1XJ/v5A==} + cpu: [arm64] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linux-arm@1.2.4': resolution: {integrity: sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A==} cpu: [arm] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-arm@1.3.3': + resolution: {integrity: sha512-3rbU4vqXXc3hY/OiXdl52xZvT0F1yEngWfvqudtPJg/KkyiaQw2DRsFrNzpmLvfavbwOq3qXn36GP8obHRULQA==} + cpu: [arm] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linux-ppc64@1.2.4': resolution: {integrity: sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA==} cpu: [ppc64] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-ppc64@1.3.3': + resolution: {integrity: sha512-cdn1OvUBwsXhbC0zSzJnNzf5MZ/mTrobawDvNXBTxe8VtqKAm0sRuEY2Evzovb/w9JMk4TvRxqt1mekSuJz64w==} + cpu: [ppc64] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linux-riscv64@1.2.4': resolution: {integrity: sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA==} cpu: [riscv64] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-riscv64@1.3.3': + resolution: {integrity: sha512-HjPVx7yKz+0lqdhDlTw1tt90wamBoxhiXpvl1XZpJLiHH4RCJ5yDTqH+VlYPv2fwFs89JFw4c1IexYOcQUi4IQ==} + cpu: [riscv64] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linux-s390x@1.2.4': resolution: {integrity: sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ==} cpu: [s390x] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-s390x@1.3.3': + resolution: {integrity: sha512-neWLh+3yCNThxnfy3c4BbVBeGgt9aftno+XbT56iK28RgeDs3UOFWviLWlUu0bArYVYJaFDK+RRohbicUNCm8Q==} + cpu: [s390x] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linux-x64@1.2.4': resolution: {integrity: sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw==} cpu: [x64] os: [linux] libc: [glibc] + '@img/sharp-libvips-linux-x64@1.3.3': + resolution: {integrity: sha512-4vKmvAst9nrowcqquKFAyZJUDolUaIp8uRiN0mWFguJ1IplC9/pitXtlnnlU4aa/eJw3J7i67V+pwUL+wZGdsA==} + cpu: [x64] + os: [linux] + libc: [glibc] + '@img/sharp-libvips-linuxmusl-arm64@1.2.4': resolution: {integrity: sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw==} cpu: [arm64] os: [linux] libc: [musl] + '@img/sharp-libvips-linuxmusl-arm64@1.3.3': + resolution: {integrity: sha512-Y9kQaLMuNoB0bPYOOdcZMaseNrFpPodIWWMrx+CZyydf2xn68j9WYc6sWWRrDwNkzCQjKYfc68L7jKjGlHMibw==} + cpu: [arm64] + os: [linux] + libc: [musl] + '@img/sharp-libvips-linuxmusl-x64@1.2.4': resolution: {integrity: sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg==} cpu: [x64] os: [linux] libc: [musl] + '@img/sharp-libvips-linuxmusl-x64@1.3.3': + resolution: {integrity: sha512-fj8Mv0HHfD1Rr+4I68+3agJynxDWtBFgicTbSOb9Bke6pIwzGcJ+RX/yHjmiEGFMCavY/dxvem7MyNaJF+wDiw==} + cpu: [x64] + os: [linux] + libc: [musl] + '@img/sharp-linux-arm64@0.34.5': resolution: {integrity: sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -912,6 +1075,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-arm64@0.35.4': + resolution: {integrity: sha512-De4jpEnAU8Hd5oT0j1G3uL4ZvTuipVMn7YC6vPaJhy6/7EwEae0SVAoBrUMYQbkLGDm85taVWwuPc1a44LTzCQ==} + engines: {node: '>=20.9.0'} + cpu: [arm64] + os: [linux] + libc: [glibc] + '@img/sharp-linux-arm@0.34.5': resolution: {integrity: sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -919,6 +1089,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-arm@0.35.4': + resolution: {integrity: sha512-7OAS8gI0EReKGVN2HssHlM6umJgxF5VI3xN0p9FA91p/YO+ou5hiNghLdZ5BEHztwaaK5+bLKRf8x/o2L2nk9A==} + engines: {node: '>=20.9.0'} + cpu: [arm] + os: [linux] + libc: [glibc] + '@img/sharp-linux-ppc64@0.34.5': resolution: {integrity: sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -926,6 +1103,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-ppc64@0.35.4': + resolution: {integrity: sha512-2oYZJeIl4kCcMGk4ouZVjnkCtFrpQFlNEtJ6GbxzhHQchwH0NH/qEb9ykmOl29dqwMq+JhFdZn+1ak2FKhI9fQ==} + engines: {node: '>=20.9.0'} + cpu: [ppc64] + os: [linux] + libc: [glibc] + '@img/sharp-linux-riscv64@0.34.5': resolution: {integrity: sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -933,6 +1117,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-riscv64@0.35.4': + resolution: {integrity: sha512-cPbNChoRURAWdebDIHSenxRpgEdy7JkPydSnUxRm9VvKD7m0/xVaR/8Fzlu81pk5nHEvHH87UZUA7cTtwnbJSA==} + engines: {node: '>=20.9.0'} + cpu: [riscv64] + os: [linux] + libc: [glibc] + '@img/sharp-linux-s390x@0.34.5': resolution: {integrity: sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -940,6 +1131,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-s390x@0.35.4': + resolution: {integrity: sha512-RY0JFY8Fd6RonCBtHz+DvadaPkXDSI1AUn6yWL9TipqkZ1vY8w8evqdgyDFnkm4/K1ve1TvZiaePP5oSd4+WVQ==} + engines: {node: '>=20.9.0'} + cpu: [s390x] + os: [linux] + libc: [glibc] + '@img/sharp-linux-x64@0.34.5': resolution: {integrity: sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -947,6 +1145,13 @@ packages: os: [linux] libc: [glibc] + '@img/sharp-linux-x64@0.35.4': + resolution: {integrity: sha512-9qvvEAuk8k89TfWUoX2htWjbAMX8p+NxCppjpcg5k6xMsjhBQPTsoIh36h9Qde4WRuGpJeYnOjdosDn/cnv+OA==} + engines: {node: '>=20.9.0'} + cpu: [x64] + os: [linux] + libc: [glibc] + '@img/sharp-linuxmusl-arm64@0.34.5': resolution: {integrity: sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -954,6 +1159,13 @@ packages: os: [linux] libc: [musl] + '@img/sharp-linuxmusl-arm64@0.35.4': + resolution: {integrity: sha512-KB5jxpfWQTr0nc3xdHtWChdbifHrBGsd2SM62Eyxrl8afikm+f5qGBU75SJIZBT/S1MC8XyacdlXBMSWq6OURA==} + engines: {node: '>=20.9.0'} + cpu: [arm64] + os: [linux] + libc: [musl] + '@img/sharp-linuxmusl-x64@0.34.5': resolution: {integrity: sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} @@ -961,29 +1173,63 @@ packages: os: [linux] libc: [musl] + '@img/sharp-linuxmusl-x64@0.35.4': + resolution: {integrity: sha512-f+eZJZIQNEEd26RPSW+76chwOf1XtA2Y/O+5ocVyLliHkeih3e+jhLVBdNTd2rS3IbNXK8+ug93Vf5ZXtF5Lxg==} + engines: {node: '>=20.9.0'} + cpu: [x64] + os: [linux] + libc: [musl] + '@img/sharp-wasm32@0.34.5': resolution: {integrity: sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [wasm32] + '@img/sharp-wasm32@0.35.4': + resolution: {integrity: sha512-zQnl4Kwp7Q6NHsENtU2T/00Zi+w3AQNwz3+UaTyVBy2FpXrzXzGjndpK61onhZjRtRpQXxCTeqw19bVyXOh7jA==} + engines: {node: '>=20.9.0'} + + '@img/sharp-webcontainers-wasm32@0.35.4': + resolution: {integrity: sha512-ESfNkywmCfPNyaZjxooddJQiQ+l/nTpGEOGthxiLnIHXC/CmcBixnfwUleX9mCz9ovrUUvKMap/pm8RYbzfwaA==} + engines: {node: '>=20.9.0'} + cpu: [wasm32] + '@img/sharp-win32-arm64@0.34.5': resolution: {integrity: sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [arm64] os: [win32] + '@img/sharp-win32-arm64@0.35.4': + resolution: {integrity: sha512-iNdlBX9gLVvqe2I3uIJSIKTq6wckP/DYxZtcqxm09x5Gi24DnFBmPAWZmr60ZyYMG0xlzo6goG3670ar+RXvRw==} + engines: {node: '>=20.9.0'} + cpu: [arm64] + os: [win32] + '@img/sharp-win32-ia32@0.34.5': resolution: {integrity: sha512-FV9m/7NmeCmSHDD5j4+4pNI8Cp3aW+JvLoXcTUo0IqyjSfAZJ8dIUmijx1qaJsIiU+Hosw6xM5KijAWRJCSgNg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [ia32] os: [win32] + '@img/sharp-win32-ia32@0.35.4': + resolution: {integrity: sha512-kqRsbaa5CS6KHlpxnN7WhE6vAAugXyZButpRdvDWetlv6Qv4N9WTcrWzF7tXfB9T7MsoadqdI8hmwLq6UlLvtw==} + engines: {node: ^20.9.0} + cpu: [ia32] + os: [win32] + '@img/sharp-win32-x64@0.34.5': resolution: {integrity: sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} cpu: [x64] os: [win32] + '@img/sharp-win32-x64@0.35.4': + resolution: {integrity: sha512-XtmnYhBcrORsJ4XJngyzr/EWP0hRZLAZRFaApdKuviyqF78+ylxh2y06ZmtULAMOnObJ3ucpN0AcwSWnMowTRg==} + engines: {node: '>=20.9.0'} + cpu: [x64] + os: [win32] + '@jridgewell/gen-mapping@0.3.13': resolution: {integrity: sha512-2kkt/7niJ6MgEPxF0bYdQ6etZaA+fQvDcLKckhy1yIQOzaoKjBBjSj63/aLVjYE3qhRt5dvM+uUyfCg6UKCBbA==} @@ -1030,10 +1276,54 @@ packages: '@open-draft/deferred-promise@2.2.0': resolution: {integrity: sha512-CecwLWx3rhxVQF6V4bAgPS5t+So2sTbPgAzafKkVizyi7tlwpcFpdFqq+wqF2OwNBmqFuu6tOyouTuxgpMfzmA==} + '@opentelemetry/api-logs@0.222.0': + resolution: {integrity: sha512-9mb1If+IF6u0ZVXkHQ6ogEae5HwA6ajIVUgpSDQyRASxft6BSXHvBvPooRle3yFN/fKnCdSOnuu0OC3PLcF6+g==} + engines: {node: '>=8.0.0'} + '@opentelemetry/api@1.9.1': resolution: {integrity: sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q==} engines: {node: '>=8.0.0'} + '@opentelemetry/core@2.11.0': + resolution: {integrity: sha512-7YP44XH0tV6+Mb54x2YGf84i7yi+31MBZlE8JwvozkxyTvXbSp10X7cI7YE49ChJ3shMJoBmCJF3+1QFBJctGA==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': '>=1.0.0 <1.10.0' + + '@opentelemetry/otlp-transformer@0.222.0': + resolution: {integrity: sha512-/F3BZ89+CJQnZkMh2tCrtcdB+XT2Dxhj4FFE+WPQ//413hmFL0/RfEX6vgOIWGhiSzrkHWTK3+6SiT7K5/g/jQ==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': ^1.3.0 + + '@opentelemetry/resources@2.11.0': + resolution: {integrity: sha512-Ie7+8q8MDF4FAEQCKVMTx3ReUvxiIAgIiiW3c9JdmP8+HMcDy20puT+AHjexnExgnbvBxjQ9fjkFDWrikJ2jQA==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': '>=1.3.0 <1.10.0' + + '@opentelemetry/sdk-logs@0.222.0': + resolution: {integrity: sha512-+19YHODIjaUCArxleaJtuufFZVpz/xvvK+VllQqE+W8hHolxdoRwHfK/s667zezwh1hkx6FFF+oYzetYgqK+Bg==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': '>=1.4.0 <1.10.0' + + '@opentelemetry/sdk-metrics@2.11.0': + resolution: {integrity: sha512-7GXXcObyHyDUUSG+L+kJoquty01bzm7ivE7+SSgXXJcHuPzGviptxwARmI2c+bnnxjexGQbJnyNlN8HxBP/Y7A==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': '>=1.9.0 <1.10.0' + + '@opentelemetry/sdk-trace@2.11.0': + resolution: {integrity: sha512-fFnTqGm8/G73GQVnxYi7LXa1ZVYEUvgL6XI1LpvV0bPC7WQ/ZGgKxCSl8FnlZBKto9JHHEFTO6s6CUpvvtwFrA==} + engines: {node: ^18.19.0 || >=20.6.0} + peerDependencies: + '@opentelemetry/api': '>=1.3.0 <1.10.0' + + '@opentelemetry/semantic-conventions@1.43.0': + resolution: {integrity: sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg==} + engines: {node: '>=14'} + '@playwright/test@1.61.1': resolution: {integrity: sha512-8nKv6+0RJSL9FE4jYOEGXnPeM/Hg12qZpmqzZjRh3qM0Y7c3z1mrOTfFLids72RDQYVh9WpLEfR5WdpNX4fkig==} engines: {node: '>=18'} @@ -1569,6 +1859,9 @@ packages: isexe@2.0.0: resolution: {integrity: sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==} + jose@6.2.12: + resolution: {integrity: sha512-9NiFmJEex0sy2Dk58j2UGBSHgUs2ypF9eZSu4L6vjOX3Dp96Sw1F3uL+H+D1sx02jZZdzUT0HgvCy59CuvXcWw==} + js-tokens@4.0.0: resolution: {integrity: sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ==} @@ -1613,6 +1906,10 @@ packages: engines: {node: '>=22.0.0'} hasBin: true + miniflare@5.20260921.0-alpha: + resolution: {integrity: sha512-vHH/unOYvV2jA1Q9SdkmzrQhhMoksdwg5jegu6ZeKaaRzgxZhVbt1NdTpQjHF2VTgiBjgP8SiUlUMfruB3N3SQ==} + engines: {node: '>=22.0.0'} + minimatch@10.2.5: resolution: {integrity: sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg==} engines: {node: 18 || 20 || >=22} @@ -1735,6 +2032,15 @@ packages: resolution: {integrity: sha512-Ou9I5Ft9WNcCbXrU9cMgPBcCK8LiwLqcbywW3t4oDV37n1pzpuNLsYiAV8eODnjbtQlSDwZ2cUEeQz4E54Hltg==} engines: {node: ^18.17.0 || ^20.3.0 || >=21.0.0} + sharp@0.35.4: + resolution: {integrity: sha512-n++8XWcj+jCOr2IOl7h8LbKnGBDY4aPbmprMONBNFdn0ImXqpGVv5zliDs0V9HbmbCQLpbuo2ej9rAoOQTvMDA==} + engines: {node: '>=20.9.0'} + peerDependencies: + '@types/node': '*' + peerDependenciesMeta: + '@types/node': + optional: true + source-map-js@1.2.1: resolution: {integrity: sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA==} engines: {node: '>=0.10.0'} @@ -1771,6 +2077,10 @@ packages: engines: {node: '>=14.17'} hasBin: true + ua-parser-js@1.0.41: + resolution: {integrity: sha512-LbBDqdIC5s8iROCUjMbW1f5dJQTEFB1+KO9ogbvlb3nm9n4YHa5p4KTvFPWvh2Hs8gZMBuiB1/8+pdfe/tDPug==} + hasBin: true + undici-types@6.21.0: resolution: {integrity: sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==} @@ -1778,6 +2088,10 @@ packages: resolution: {integrity: sha512-cRZYrTDwWznlnRiPjggAGxZXanty6M8RV1ff8Wm4LWXBp7/IG8v5DnOm74DtUBp9OONpK75YlPnIjQqX0dBDtA==} engines: {node: '>=20.18.1'} + undici@7.29.0: + resolution: {integrity: sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==} + engines: {node: '>=20.18.1'} + unenv@2.0.0-rc.24: resolution: {integrity: sha512-i7qRCmY42zmCwnYlh9H2SvLEypEFGye5iRmEMKjcGi7zk9UquigRjFtTLz0TYqr0ZGLZhaMHl/foy1bZR+Cwlw==} @@ -1861,6 +2175,9 @@ packages: w3c-keyname@2.2.8: resolution: {integrity: sha512-dpojBhNsCNN7T82Tm7k26A6G9ML3NkhDsnw9n/eoxSRlVBB4CEtIQ/KTCLI2Fwf3ataSXRhYFkQi3SlnFwPvPQ==} + web-vitals@6.2.2: + resolution: {integrity: sha512-oto5x6dLEgrRqfcWed+pZEUb2q6ikbFmqF54CRDhI/QGbn+qxC49b4C4OkbH+kb9C3a8shpFD3SgR9kRcs80ZQ==} + webidl-conversions@3.0.1: resolution: {integrity: sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ==} @@ -1877,6 +2194,11 @@ packages: engines: {node: '>=16'} hasBin: true + workerd@1.20260921.1: + resolution: {integrity: sha512-4HyG7G1W4ksa6tUZ8bV2jxDRWuL5PXnHm9+Z1sjFPb9OZNoYtXz4y7QQRh4ibi0BF/lOmlAVjbhkUqsAVZuUKA==} + engines: {node: '>=16'} + hasBin: true + wrangler@4.108.0: resolution: {integrity: sha512-mJnY6vf2Tminfo0Jy5OreBgUjIKW3iM1TQtOrZZRpFUyiQvAYUDVN0aRcDn8/aEfF0bmZgcixRdwwVHZMZaE7w==} engines: {node: '>=22.0.0'} @@ -1887,6 +2209,16 @@ packages: '@cloudflare/workers-types': optional: true + wrangler@4.136.3: + resolution: {integrity: sha512-L1bS8BI9xoEk3RRr5XoZn5q5k1jCRo8+37ZEEab/jsUFzE1Ea2sIBAbmP1CdwSeD6mZYmnUtR6xpi9esHIl6GA==} + engines: {node: '>=22.0.0'} + hasBin: true + peerDependencies: + '@cloudflare/workers-types': ^5.20260921.1 + peerDependenciesMeta: + '@cloudflare/workers-types': + optional: true + ws@8.21.0: resolution: {integrity: sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==} engines: {node: '>=10.0.0'} @@ -2054,6 +2386,8 @@ snapshots: '@babel/helper-string-parser': 7.29.7 '@babel/helper-validator-identifier': 7.29.7 + '@bufbuild/protobuf@2.15.0': {} + '@cloudflare/containers@0.3.7': {} '@cloudflare/kv-asset-handler@0.5.0': {} @@ -2071,23 +2405,46 @@ snapshots: optionalDependencies: workerd: 1.20260706.1 + '@cloudflare/unenv-preset@2.16.2(unenv@2.0.0-rc.24)(workerd@1.20260921.1)': + dependencies: + unenv: 2.0.0-rc.24 + optionalDependencies: + workerd: 1.20260921.1 + '@cloudflare/workerd-darwin-64@1.20260706.1': optional: true + '@cloudflare/workerd-darwin-64@1.20260921.1': + optional: true + '@cloudflare/workerd-darwin-arm64@1.20260706.1': optional: true + '@cloudflare/workerd-darwin-arm64@1.20260921.1': + optional: true + '@cloudflare/workerd-linux-64@1.20260706.1': optional: true + '@cloudflare/workerd-linux-64@1.20260921.1': + optional: true + '@cloudflare/workerd-linux-arm64@1.20260706.1': optional: true + '@cloudflare/workerd-linux-arm64@1.20260921.1': + optional: true + '@cloudflare/workerd-windows-64@1.20260706.1': optional: true + '@cloudflare/workerd-windows-64@1.20260921.1': + optional: true + '@cloudflare/workers-types@4.20260702.1': {} + '@cloudflare/workers-types@5.20260923.1': {} + '@codemirror/autocomplete@6.20.3': dependencies: '@codemirror/language': 6.12.4 @@ -2208,6 +2565,11 @@ snapshots: tslib: 2.8.1 optional: true + '@emnapi/runtime@1.11.3': + dependencies: + tslib: 2.8.1 + optional: true + '@esbuild/aix-ppc64@0.21.5': optional: true @@ -2435,6 +2797,17 @@ snapshots: '@fontsource/fira-code@5.3.0': {} + '@grafana/faro-core@2.12.1': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/otlp-transformer': 0.222.0(@opentelemetry/api@1.9.1) + + '@grafana/faro-web-sdk@2.12.1': + dependencies: + '@grafana/faro-core': 2.12.1 + ua-parser-js: 1.0.41 + web-vitals: 6.2.2 + '@img/colour@1.1.0': {} '@img/sharp-darwin-arm64@0.34.5': @@ -2442,95 +2815,199 @@ snapshots: '@img/sharp-libvips-darwin-arm64': 1.2.4 optional: true + '@img/sharp-darwin-arm64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-darwin-arm64': 1.3.3 + optional: true + '@img/sharp-darwin-x64@0.34.5': optionalDependencies: '@img/sharp-libvips-darwin-x64': 1.2.4 optional: true + '@img/sharp-darwin-x64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-darwin-x64': 1.3.3 + optional: true + + '@img/sharp-freebsd-wasm32@0.35.4': + dependencies: + '@img/sharp-wasm32': 0.35.4 + optional: true + '@img/sharp-libvips-darwin-arm64@1.2.4': optional: true + '@img/sharp-libvips-darwin-arm64@1.3.3': + optional: true + '@img/sharp-libvips-darwin-x64@1.2.4': optional: true + '@img/sharp-libvips-darwin-x64@1.3.3': + optional: true + '@img/sharp-libvips-linux-arm64@1.2.4': optional: true + '@img/sharp-libvips-linux-arm64@1.3.3': + optional: true + '@img/sharp-libvips-linux-arm@1.2.4': optional: true + '@img/sharp-libvips-linux-arm@1.3.3': + optional: true + '@img/sharp-libvips-linux-ppc64@1.2.4': optional: true + '@img/sharp-libvips-linux-ppc64@1.3.3': + optional: true + '@img/sharp-libvips-linux-riscv64@1.2.4': optional: true + '@img/sharp-libvips-linux-riscv64@1.3.3': + optional: true + '@img/sharp-libvips-linux-s390x@1.2.4': optional: true + '@img/sharp-libvips-linux-s390x@1.3.3': + optional: true + '@img/sharp-libvips-linux-x64@1.2.4': optional: true + '@img/sharp-libvips-linux-x64@1.3.3': + optional: true + '@img/sharp-libvips-linuxmusl-arm64@1.2.4': optional: true + '@img/sharp-libvips-linuxmusl-arm64@1.3.3': + optional: true + '@img/sharp-libvips-linuxmusl-x64@1.2.4': optional: true + '@img/sharp-libvips-linuxmusl-x64@1.3.3': + optional: true + '@img/sharp-linux-arm64@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-arm64': 1.2.4 optional: true + '@img/sharp-linux-arm64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-arm64': 1.3.3 + optional: true + '@img/sharp-linux-arm@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-arm': 1.2.4 optional: true + '@img/sharp-linux-arm@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-arm': 1.3.3 + optional: true + '@img/sharp-linux-ppc64@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-ppc64': 1.2.4 optional: true + '@img/sharp-linux-ppc64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-ppc64': 1.3.3 + optional: true + '@img/sharp-linux-riscv64@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-riscv64': 1.2.4 optional: true + '@img/sharp-linux-riscv64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-riscv64': 1.3.3 + optional: true + '@img/sharp-linux-s390x@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-s390x': 1.2.4 optional: true + '@img/sharp-linux-s390x@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-s390x': 1.3.3 + optional: true + '@img/sharp-linux-x64@0.34.5': optionalDependencies: '@img/sharp-libvips-linux-x64': 1.2.4 optional: true + '@img/sharp-linux-x64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linux-x64': 1.3.3 + optional: true + '@img/sharp-linuxmusl-arm64@0.34.5': optionalDependencies: '@img/sharp-libvips-linuxmusl-arm64': 1.2.4 optional: true + '@img/sharp-linuxmusl-arm64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linuxmusl-arm64': 1.3.3 + optional: true + '@img/sharp-linuxmusl-x64@0.34.5': optionalDependencies: '@img/sharp-libvips-linuxmusl-x64': 1.2.4 optional: true + '@img/sharp-linuxmusl-x64@0.35.4': + optionalDependencies: + '@img/sharp-libvips-linuxmusl-x64': 1.3.3 + optional: true + '@img/sharp-wasm32@0.34.5': dependencies: '@emnapi/runtime': 1.11.2 optional: true + '@img/sharp-wasm32@0.35.4': + dependencies: + '@emnapi/runtime': 1.11.3 + optional: true + + '@img/sharp-webcontainers-wasm32@0.35.4': + dependencies: + '@img/sharp-wasm32': 0.35.4 + optional: true + '@img/sharp-win32-arm64@0.34.5': optional: true + '@img/sharp-win32-arm64@0.35.4': + optional: true + '@img/sharp-win32-ia32@0.34.5': optional: true + '@img/sharp-win32-ia32@0.35.4': + optional: true + '@img/sharp-win32-x64@0.34.5': optional: true + '@img/sharp-win32-x64@0.35.4': + optional: true + '@jridgewell/gen-mapping@0.3.13': dependencies: '@jridgewell/sourcemap-codec': 1.5.5 @@ -2593,8 +3070,56 @@ snapshots: '@open-draft/deferred-promise@2.2.0': {} + '@opentelemetry/api-logs@0.222.0': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/api@1.9.1': {} + '@opentelemetry/core@2.11.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/semantic-conventions': 1.43.0 + + '@opentelemetry/otlp-transformer@0.222.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/api-logs': 0.222.0 + '@opentelemetry/core': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/resources': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/sdk-logs': 0.222.0(@opentelemetry/api@1.9.1) + '@opentelemetry/sdk-metrics': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/sdk-trace': 2.11.0(@opentelemetry/api@1.9.1) + + '@opentelemetry/resources@2.11.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/core': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/semantic-conventions': 1.43.0 + + '@opentelemetry/sdk-logs@0.222.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/api-logs': 0.222.0 + '@opentelemetry/core': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/resources': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/semantic-conventions': 1.43.0 + + '@opentelemetry/sdk-metrics@2.11.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/core': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/resources': 2.11.0(@opentelemetry/api@1.9.1) + + '@opentelemetry/sdk-trace@2.11.0(@opentelemetry/api@1.9.1)': + dependencies: + '@opentelemetry/api': 1.9.1 + '@opentelemetry/core': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/resources': 2.11.0(@opentelemetry/api@1.9.1) + '@opentelemetry/semantic-conventions': 1.43.0 + + '@opentelemetry/semantic-conventions@1.43.0': {} + '@playwright/test@1.61.1': dependencies: playwright: 1.61.1 @@ -3138,6 +3663,8 @@ snapshots: isexe@2.0.0: {} + jose@6.2.12: {} + js-tokens@4.0.0: {} jsesc@3.1.0: {} @@ -3176,6 +3703,19 @@ snapshots: - bufferutil - utf-8-validate + miniflare@5.20260921.0-alpha(@types/node@22.20.1): + dependencies: + '@cspotcode/source-map-support': 0.8.1 + sharp: 0.35.4(@types/node@22.20.1) + undici: 7.29.0 + workerd: 1.20260921.1 + ws: 8.21.0 + youch: 4.1.0-beta.10 + transitivePeerDependencies: + - '@types/node' + - bufferutil + - utf-8-validate + minimatch@10.2.5: dependencies: brace-expansion: 5.0.8 @@ -3316,6 +3856,39 @@ snapshots: '@img/sharp-win32-ia32': 0.34.5 '@img/sharp-win32-x64': 0.34.5 + sharp@0.35.4(@types/node@22.20.1): + dependencies: + '@img/colour': 1.1.0 + detect-libc: 2.1.2 + semver: 7.8.5 + optionalDependencies: + '@img/sharp-darwin-arm64': 0.35.4 + '@img/sharp-darwin-x64': 0.35.4 + '@img/sharp-freebsd-wasm32': 0.35.4 + '@img/sharp-libvips-darwin-arm64': 1.3.3 + '@img/sharp-libvips-darwin-x64': 1.3.3 + '@img/sharp-libvips-linux-arm': 1.3.3 + '@img/sharp-libvips-linux-arm64': 1.3.3 + '@img/sharp-libvips-linux-ppc64': 1.3.3 + '@img/sharp-libvips-linux-riscv64': 1.3.3 + '@img/sharp-libvips-linux-s390x': 1.3.3 + '@img/sharp-libvips-linux-x64': 1.3.3 + '@img/sharp-libvips-linuxmusl-arm64': 1.3.3 + '@img/sharp-libvips-linuxmusl-x64': 1.3.3 + '@img/sharp-linux-arm': 0.35.4 + '@img/sharp-linux-arm64': 0.35.4 + '@img/sharp-linux-ppc64': 0.35.4 + '@img/sharp-linux-riscv64': 0.35.4 + '@img/sharp-linux-s390x': 0.35.4 + '@img/sharp-linux-x64': 0.35.4 + '@img/sharp-linuxmusl-arm64': 0.35.4 + '@img/sharp-linuxmusl-x64': 0.35.4 + '@img/sharp-webcontainers-wasm32': 0.35.4 + '@img/sharp-win32-arm64': 0.35.4 + '@img/sharp-win32-ia32': 0.35.4 + '@img/sharp-win32-x64': 0.35.4 + '@types/node': 22.20.1 + source-map-js@1.2.1: {} source-map@0.6.1: {} @@ -3345,10 +3918,14 @@ snapshots: typescript@5.6.3: {} + ua-parser-js@1.0.41: {} + undici-types@6.21.0: {} undici@7.28.0: {} + undici@7.29.0: {} + unenv@2.0.0-rc.24: dependencies: pathe: 2.0.3 @@ -3382,6 +3959,8 @@ snapshots: w3c-keyname@2.2.8: {} + web-vitals@6.2.2: {} + webidl-conversions@3.0.1: {} whatwg-url@5.0.0: @@ -3401,6 +3980,14 @@ snapshots: '@cloudflare/workerd-linux-arm64': 1.20260706.1 '@cloudflare/workerd-windows-64': 1.20260706.1 + workerd@1.20260921.1: + optionalDependencies: + '@cloudflare/workerd-darwin-64': 1.20260921.1 + '@cloudflare/workerd-darwin-arm64': 1.20260921.1 + '@cloudflare/workerd-linux-64': 1.20260921.1 + '@cloudflare/workerd-linux-arm64': 1.20260921.1 + '@cloudflare/workerd-windows-64': 1.20260921.1 + wrangler@4.108.0(@cloudflare/workers-types@4.20260702.1): dependencies: '@cloudflare/kv-asset-handler': 0.5.0 @@ -3418,6 +4005,24 @@ snapshots: - bufferutil - utf-8-validate + wrangler@4.136.3(@cloudflare/workers-types@5.20260923.1)(@types/node@22.20.1): + dependencies: + '@cloudflare/kv-asset-handler': 0.5.0 + '@cloudflare/unenv-preset': 2.16.2(unenv@2.0.0-rc.24)(workerd@1.20260921.1) + blake3-wasm: 2.1.5 + esbuild: 0.28.1 + miniflare: 5.20260921.0-alpha(@types/node@22.20.1) + path-to-regexp: 6.3.0 + unenv: 2.0.0-rc.24 + workerd: 1.20260921.1 + optionalDependencies: + '@cloudflare/workers-types': 5.20260923.1 + fsevents: 2.3.3 + transitivePeerDependencies: + - '@types/node' + - bufferutil + - utf-8-validate + ws@8.21.0: {} yallist@3.1.1: {} diff --git a/runner/scripts/check-telemetry-leak.mjs b/runner/scripts/check-telemetry-leak.mjs new file mode 100644 index 0000000000..fc455b5886 --- /dev/null +++ b/runner/scripts/check-telemetry-leak.mjs @@ -0,0 +1,58 @@ +#!/usr/bin/env node +// A durable post-build leak check for the local-telemetry path (contract +// §10). Builds nothing — run against an already-built +// `apps/authoring/dist/` (CI wires it in right after the production +// build). Exit 0 ("ok") when no sentinel is found; exit 1 and lists every +// match otherwise. + +import { readFileSync, readdirSync, existsSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const dist = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../apps/authoring/dist"); + +if (!existsSync(dist)) { + console.error(`no build to check at ${dist} — run \`pnpm --filter @handsontable/demo-authoring build\` first`); + process.exit(1); +} + +const assetsDir = path.join(dist, "assets"); +if (!existsSync(assetsDir)) { + console.error(`${assetsDir} does not exist — the build did not finish`); + process.exit(1); +} + +// Each sentinel is a string that only survives in a production dist/ if +// dead-code elimination failed or the local-telemetry flag leaked in — +// see main.tsx/sentry.ts/faro.ts for where each one is gated. +const SENTINELS = [ + "__test_crash_boundary", + "T06 e2e render-crash probe", + "VITE_TELEMETRY_LOCAL", + "__t06SentryCapture", + "__t06ReportDemoEvent", + "__t06Telemetry", +]; + +const jsFiles = readdirSync(assetsDir).filter((f) => f.endsWith(".js")); +if (jsFiles.length === 0) { + console.error(`${assetsDir} has no .js assets — the build did not finish`); + process.exit(1); +} + +const hits = []; +for (const file of jsFiles) { + const code = readFileSync(path.join(assetsDir, file), "utf8"); + for (const sentinel of SENTINELS) { + if (code.includes(sentinel)) hits.push(`assets/${file}: "${sentinel}"`); + } +} + +if (hits.length) { + console.error("telemetry leak check failed — local-path sentinel(s) found in a production build:"); + for (const h of hits) console.error(` - ${h}`); + console.error("Rebuild with .env.local absent and VITE_TELEMETRY_LOCAL unset."); + process.exit(1); +} + +console.log(`telemetry leak check ok: no local-path sentinel found across ${jsFiles.length} JS asset(s)`); diff --git a/runner/scripts/ci/assert-e2e-ran.mjs b/runner/scripts/ci/assert-e2e-ran.mjs new file mode 100644 index 0000000000..c710a75aca --- /dev/null +++ b/runner/scripts/ci/assert-e2e-ran.mjs @@ -0,0 +1,53 @@ +// Guard for `.github/workflows/e2e-o11y-local.yml`: reads a Playwright JSON +// reporter output and fails loudly when the gated run did not actually +// exercise any tests, or silently skipped every one of them — the exact +// failure shape a mistyped/missing env gate (E2E_LIVE, E2E_TELEMETRY, +// E2E_O11Y_LOCAL) produces: `playwright test` still exits 0 (every test +// `test.skip()`s itself cleanly), so the step above this one reads as a +// pass with zero real coverage. docs/TESTING.md's "every gate must have a +// workflow home" rule is only true in practice if the workflow provably RAN +// the gated tests — this is that proof, not just a green exit code. +// +// Usage: node scripts/ci/assert-e2e-ran.mjs +// Run from `runner/` (the working directory Playwright's +// PLAYWRIGHT_JSON_OUTPUT_NAME wrote the file into). + +import { readFileSync } from "node:fs"; + +const [, , reportPath] = process.argv; +if (!reportPath) { + console.error("usage: node scripts/ci/assert-e2e-ran.mjs "); + process.exit(2); +} + +const report = JSON.parse(readFileSync(reportPath, "utf8")); +const stats = report.stats ?? {}; +const expected = stats.expected ?? 0; +const skipped = stats.skipped ?? 0; +const unexpected = stats.unexpected ?? 0; + +console.log(`e2e report: expected=${expected} skipped=${skipped} unexpected=${unexpected} (${reportPath})`); + +if (expected === 0) { + console.error( + "::error::0 tests were expected to run — the gate env var is almost certainly missing or misspelled " + + "(every test.skip()'d itself, and `playwright test` still exits 0 for that).", + ); + process.exit(1); +} +if (skipped > 0) { + console.error( + `::error::${skipped} test(s) were skipped — a gate condition inside the spec file itself did not ` + + "evaluate the way this workflow's env vars intend. Check the spec's test.skip(...) condition.", + ); + process.exit(1); +} +if (unexpected > 0) { + // Belt and suspenders: the e2e step's own non-zero exit already fails the + // job on a real test failure, so reaching here with unexpected > 0 should + // never happen — but never let a report-parsing quirk mask a real failure. + console.error(`::error::${unexpected} unexpected result(s) in the report — treating as a failure.`); + process.exit(1); +} + +console.log("ok — the gate ran real tests, none skipped."); diff --git a/runner/scripts/dev-lib.mjs b/runner/scripts/dev-lib.mjs new file mode 100644 index 0000000000..3d9ad661b6 --- /dev/null +++ b/runner/scripts/dev-lib.mjs @@ -0,0 +1,1720 @@ +// Shared logic for `runner/scripts/dev.mjs` and `runner/scripts/o11y-dev.mjs`. +// Every function here is pure or takes its side effects (fs, exec, spawn) as +// injectable parameters, so `pipeline/dev-script.test.mjs` can exercise the +// real logic with stub binaries. Env vars read here are kept in sync with +// `runner/docs/run-and-deploy.md` by that test's own drift check. + +import { existsSync, mkdirSync, readFileSync, writeFileSync, readdirSync, statSync, rmSync } from "node:fs"; +import path from "node:path"; +import { randomBytes, createHash } from "node:crypto"; + +export const RUNNER_ROOT = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); + +// --------------------------------------------------------------------------- +// Help / arg parsing +// --------------------------------------------------------------------------- + +export const HELP_TEXT = `Usage: node scripts/dev.mjs --tier=1|2|full [options] + +Tiers: + --tier=1 Authoring app only (builds @handsontable/demo-runtime if its + dist is stale, then runs vite for apps/authoring). + --tier=2 Tier 1 + the API worker (wrangler dev, Docker containers, + local D1 migrations). + --tier=full Tier 2 + the o11y worker, docker compose (minio/clickhouse — + the box itself runs through wrangler dev's own container + orchestration), telemetry wiring, and the local Slack capture + server. + +Options: + --replay (--tier=full only) run the OTLP/Faro fixture replay once, + after the o11y worker reports ready. Without this flag, + dev.mjs just prints the replay command. + --reset-local-db (--tier=2 or --tier=full only) delete workers/api's local + D1 state (workers/api/.wrangler/state/v3/d1) and the + applied-migrations record before starting, then run every + migration fresh. Passing this flag IS the confirmation — + it prints what it deleted and does not prompt. + --fresh (--tier=full only) wipe ALL local o11y state together + before starting: docker compose ... down -v for this + project's minio/clickhouse (named volumes — logs and + runner_events) AND workers/o11y/.wrangler/state (the + InboxWriter ledger, dedupe hashes, local R2 inbox + objects). Without --fresh, both are KEPT across a + restart on purpose — see docs/run-and-deploy.md's "Run + locally" section for why they must be wiped together, + never separately (the API worker's D1 is untouched + either way; that's --reset-local-db). Prints exactly + what it removed. pnpm o11y:dev also accepts --fresh, + for just the workers/o11y/.wrangler/state half (it never + runs docker compose itself — see that command's own + startup log for the divergence risk if you've also got + a dev:full compose stack's volumes still holding data + from before). + --skip-image-check + (--tier=2 or --tier=full only) skip the pre-flight check + that every container base image (read from each + wrangler.jsonc's own containers[].image Dockerfile, + e.g. cloudflare/sandbox:0.12.3) is present locally, + pulling any that's missing before starting a worker. + Escape hatch for offline use when the images are + already built. + -h, --help Print this help and exit 0. + +Port overrides (env vars — defaults match the ones documented in +docs/run-and-deploy.md's "Run locally" section): + AUTHORING_DEV_PORT default 5173 + API_DEV_PORT default 8787 + API_DEV_INSPECTOR_PORT default 9230 + O11Y_DEV_PORT default 4200 + O11Y_DEV_INSPECTOR_PORT default 4201 + O11Y_MINIO_PORT default 9000 + O11Y_MINIO_CONSOLE_PORT default 9001 + O11Y_CLICKHOUSE_PORT default 8123 + O11Y_CLICKHOUSE_NATIVE_PORT default 9009 + O11Y_SLACK_CAPTURE_PORT default 4210 + +Other env vars read: + COMPOSE_PROJECT_NAME docker compose project name for --tier=full's + minio/clickhouse stack (default derived per + worktree — o11y-dev- — so two worktrees never collide on one + project). + WRANGLER_REGISTRY_PATH forwarded as-is to every spawned wrangler dev (see + docs/run-and-deploy.md) — set it to isolate this + run's service-binding registry from another + worktree's. +`; + +/** + * @param {string[]} argv (e.g. process.argv.slice(2)) + * @returns {{ help: boolean, tier: "1"|"2"|"full"|null, replay: boolean, resetLocalDb: boolean, fresh: boolean, skipImageCheck: boolean, errors: string[] }} + */ +export function parseArgs(argv) { + const errors = []; + let tier = null; + let replay = false; + let resetLocalDb = false; + let fresh = false; + let skipImageCheck = false; + let help = false; + for (const arg of argv) { + if (arg === "-h" || arg === "--help") { + help = true; + } else if (arg === "--replay") { + replay = true; + } else if (arg === "--reset-local-db") { + resetLocalDb = true; + } else if (arg === "--fresh") { + fresh = true; + } else if (arg === "--skip-image-check") { + skipImageCheck = true; + } else if (arg.startsWith("--tier=")) { + const value = arg.slice("--tier=".length); + if (value !== "1" && value !== "2" && value !== "full") { + errors.push(`--tier must be one of 1, 2, full (got "${value}")`); + } else { + tier = value; + } + } else { + errors.push(`unrecognized argument: ${arg}`); + } + } + if (!help && tier === null) { + errors.push("--tier=1|2|full is required"); + } + if (replay && tier !== "full" && tier !== null) { + errors.push("--replay is only valid with --tier=full"); + } + if (resetLocalDb && tier !== "2" && tier !== "full" && tier !== null) { + errors.push("--reset-local-db is only valid with --tier=2 or --tier=full"); + } + if (fresh && tier !== "full" && tier !== null) { + errors.push("--fresh is only valid with --tier=full"); + } + if (skipImageCheck && tier !== "2" && tier !== "full" && tier !== null) { + errors.push("--skip-image-check is only valid with --tier=2 or --tier=full"); + } + return { help, tier, replay, resetLocalDb, fresh, skipImageCheck, errors }; +} + +// --------------------------------------------------------------------------- +// Ports +// --------------------------------------------------------------------------- + +export const PORT_DEFAULTS = { + AUTHORING_DEV_PORT: 5173, + API_DEV_PORT: 8787, + API_DEV_INSPECTOR_PORT: 9230, + O11Y_DEV_PORT: 4200, + O11Y_DEV_INSPECTOR_PORT: 4201, + O11Y_MINIO_PORT: 9000, + O11Y_MINIO_CONSOLE_PORT: 9001, + O11Y_CLICKHOUSE_PORT: 8123, + O11Y_CLICKHOUSE_NATIVE_PORT: 9009, + O11Y_SLACK_CAPTURE_PORT: 4210, +}; + +const PORT_KEYS_BY_TIER = { + "1": ["AUTHORING_DEV_PORT"], + "2": ["AUTHORING_DEV_PORT", "API_DEV_PORT", "API_DEV_INSPECTOR_PORT"], + full: [ + "AUTHORING_DEV_PORT", + "API_DEV_PORT", + "API_DEV_INSPECTOR_PORT", + "O11Y_DEV_PORT", + "O11Y_DEV_INSPECTOR_PORT", + "O11Y_MINIO_PORT", + "O11Y_MINIO_CONSOLE_PORT", + "O11Y_CLICKHOUSE_PORT", + "O11Y_CLICKHOUSE_NATIVE_PORT", + "O11Y_SLACK_CAPTURE_PORT", + ], + // Not a CLI `--tier` value — used only by `scripts/o11y-dev.mjs` (the + // standalone `pnpm o11y:dev` entry point), which needs just these two + // ports resolved the same way `--tier=full` resolves them. + "o11y-only": ["O11Y_DEV_PORT", "O11Y_DEV_INSPECTOR_PORT"], +}; + +/** + * Resolves every port this tier needs from env overrides (falling back to + * PORT_DEFAULTS), and throws if any two resolve to the same number — this is + * what guarantees api/o11y (and their inspector ports) never collide. + * @param {"1"|"2"|"full"} tier + * @param {NodeJS.ProcessEnv} env + */ +export function resolvePorts(tier, env = process.env) { + const keys = PORT_KEYS_BY_TIER[tier]; + if (!keys) throw new Error(`resolvePorts: unknown tier "${tier}"`); + const resolved = {}; + for (const key of keys) { + const raw = env[key]; + const value = raw !== undefined && raw !== "" ? Number(raw) : PORT_DEFAULTS[key]; + if (!Number.isInteger(value) || value <= 0 || value > 65535) { + throw new Error(`invalid port for ${key}: "${raw}"`); + } + resolved[key] = value; + } + assertNoPortCollisions(resolved); + return resolved; +} + +/** Throws if any two of `resolved`'s own port values collide. Exported so a + * caller that MUTATES an already-resolved ports object after the fact (e.g. + * `dev.mjs` adopting a `.dev.vars`-pinned port — see + * `resolveDevVarsPortAdoption`) can re-run the same check `resolvePorts` + * itself runs, rather than silently allowing the adopted port to collide + * with another already-resolved one. */ +export function assertNoPortCollisions(resolved) { + const byPort = new Map(); + for (const [key, value] of Object.entries(resolved)) { + if (byPort.has(value)) { + throw new Error(`port collision: ${key} and ${byPort.get(value)} both resolve to ${value}`); + } + byPort.set(value, key); + } +} + +// --------------------------------------------------------------------------- +// .dev.vars bootstrap +// --------------------------------------------------------------------------- + +const defaultFs = { existsSync, readFileSync, writeFileSync, mkdirSync, rmSync }; + +/** + * Copies `examplePath` to `devVarsPath` ONLY when `devVarsPath` does not + * already exist — an existing file is never touched or overwritten. + * + * On a fresh copy only: `patch` fills declared-EMPTY placeholder lines + * (e.g. DEV_ADMIN) with a real non-secret local-dev value; `stripKeys` + * removes an empty `KEY=` line entirely, for a key injected per-run via + * `--var` — wrangler's `.dev.vars` always wins over `--var` even when + * empty, so a declared-but-empty key would silently swallow the override. + * + * @returns {{ created: boolean, patched: string[], stripped: string[] }} + */ +export function bootstrapDevVars({ examplePath, devVarsPath, patch = {}, stripKeys = [], fs = defaultFs }) { + if (fs.existsSync(devVarsPath)) { + return { created: false, patched: [], stripped: [] }; + } + if (!fs.existsSync(examplePath)) { + throw new Error(`missing ${examplePath} — cannot bootstrap ${devVarsPath}`); + } + let text = fs.readFileSync(examplePath, "utf8"); + const patched = []; + for (const [key, value] of Object.entries(patch)) { + const re = new RegExp(`^${key}=(?:"")?[ \\t]*$`, "m"); + if (re.test(text)) { + text = text.replace(re, `${key}=${value}`); + patched.push(key); + } + } + const stripped = []; + for (const key of stripKeys) { + const re = new RegExp(`^${key}=(?:"")?[ \\t]*\\n?`, "m"); + if (re.test(text)) { + text = text.replace(re, ""); + stripped.push(key); + } + } + const dir = path.dirname(devVarsPath); + if (!fs.existsSync(dir)) fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(devVarsPath, text); + return { created: true, patched, stripped }; +} + +/** + * `workers/o11y/.dev.vars.example`'s inert local placeholders, patched to + * a real non-secret local-dev value ONLY on first bootstrap. + * `SLACK_WEBHOOK_URL` is baked at PORT_DEFAULTS.O11Y_SLACK_CAPTURE_PORT + * (not via `--var`, which the declared-but-empty line would override). + * `O11Y_EXPORT_SECRET`/`SENTRY_HOOK_SECRET` are deliberately NOT here — + * this `patch` only fires on a FRESH bootstrap; `fillEmptyDevVarsSecrets` + * below fills those two on a pre-existing file too, persisted rather than + * injected via `--var` (which a standalone script invocation can't inherit). + */ +export function o11yDevVarsPatch(ports) { + return { + DEV_ADMIN: "dev@handsontable.com", + AE_SQL_TOKEN: "local-dev-token", + LOKI_S3_ACCESS_KEY_ID: "minioadmin", + LOKI_S3_SECRET_ACCESS_KEY: "minioadmin", + SLACK_WEBHOOK_URL: `http://localhost:${ports.O11Y_SLACK_CAPTURE_PORT}/slack`, + }; +} + +/** A key this run injects via `--var` that `bootstrapDevVars` must strip + * (if freshly created) from `workers/o11y/.dev.vars` — see that function's + * doc comment. Applies even when the key does not exist yet in the file — + * stripping a line that isn't there is a no-op. */ +export const O11Y_DEVVARS_STRIP_KEYS = ["O11Y_SESSION_SECRET"]; + +/** The two gate secrets `scripts/o11y-replay-fixtures.mjs` needs and that + * `workers/o11y/.dev.vars.example` declares empty by default — see + * `fillEmptyDevVarsSecrets`'s doc comment for why these are filled in + * place rather than stripped-and-`--var`'d like `O11Y_DEVVARS_STRIP_KEYS`. */ +export const O11Y_DEVVARS_AUTOFILL_SECRET_KEYS = ["O11Y_EXPORT_SECRET", "SENTRY_HOOK_SECRET"]; + +/** + * Fills any of `keys` that `devVarsPath` declares EMPTY (`KEY=` / `KEY=""`, + * the exact shape `bootstrapDevVars`'s own `patch` matches) with a fresh + * `generate()` value, and leaves every other line — including a key already + * holding a real value — untouched. A no-op when `devVarsPath` does not + * exist at all (the caller runs `bootstrapDevVars` first) or when none of + * `keys` are currently empty. + * + * Unlike `bootstrapDevVars`'s `patch`, this runs on EVERY invocation, not + * only a fresh bootstrap, so a `.dev.vars` that predates this function (its + * `O11Y_EXPORT_SECRET=`/`SENTRY_HOOK_SECRET=` lines still empty) still gets + * filled — `bootstrapDevVars` alone refuses to touch a file that already + * exists. + * + * @returns {{ filled: string[] }} + */ +export function fillEmptyDevVarsSecrets({ devVarsPath, keys, generate = ephemeralSecret, fs = defaultFs }) { + if (!fs.existsSync(devVarsPath)) return { filled: [] }; + let text = fs.readFileSync(devVarsPath, "utf8"); + const filled = []; + for (const key of keys) { + const re = new RegExp(`^${key}=(?:"")?[ \\t]*$`, "m"); + if (re.test(text)) { + text = text.replace(re, `${key}=${generate()}`); + filled.push(key); + } + } + if (filled.length) fs.writeFileSync(devVarsPath, text); + return { filled }; +} + +/** + * Resolves a `.dev.vars` key that pins a `host:port` value + * (`PREVIEW_HOST`, `SLACK_WEBHOOK_URL`) against the port this run resolved. + * `.dev.vars` always wins over `--var`, so a disagreeing declared port is + * the one that will actually be reached — a bare warning would leave the + * script pointed at the WRONG port. + * + * `explicit: false` (no override) ADOPTS the declared port, so every other + * piece this script controls agrees with reality; `explicit: true` WARNS + * and leaves `currentPort` alone — a deliberate override is never discarded. + * + * @returns {{ port: number, adopted: boolean, message: string|null }} + */ +export function resolveDevVarsPortAdoption({ devVarsPath, key, currentPort, explicit, fs = defaultFs }) { + const value = readDevVarsLine(devVarsPath, key, fs); + if (value === undefined) return { port: currentPort, adopted: false, message: null }; + const m = /:(\d+)(?:\/|$)/.exec(value); + if (!m) return { port: currentPort, adopted: false, message: null }; + const declaredPort = Number(m[1]); + if (declaredPort === currentPort) return { port: currentPort, adopted: false, message: null }; + if (explicit) { + return { + port: currentPort, + adopted: false, + message: + `${devVarsPath} declares ${key}=${value} (port ${declaredPort}), but this run resolved port ` + + `${currentPort} — .dev.vars always wins over this script's own port choice for a key it declares. ` + + `Edit ${devVarsPath} by hand, or delete it and re-run to get a fresh bootstrap at the new port.`, + }; + } + return { + port: declaredPort, + adopted: true, + message: + `${devVarsPath} declares ${key}=${value} (port ${declaredPort}) — adopting it for this run since no ` + + `explicit port override was set; .dev.vars always wins over this script's own port choice for a key it declares.`, + }; +} + +/** + * Warns (does not throw — this is advisory, not fatal) when a `.dev.vars` + * value baked in at bootstrap time (a `localhost:`-shaped default) + * disagrees with the port this run actually resolved AND that port was + * explicitly requested (`resolveDevVarsPortAdoption`'s `explicit: true` + * branch) — the situation where a developer set a port-override env var + * AFTER their `.dev.vars` was already bootstrapped with the old default. + * @returns {string|null} a warning line, or null if there's no drift to report + */ +export function checkDevVarsPortDrift(devVarsPath, key, expectedPort, fs = defaultFs) { + return resolveDevVarsPortAdoption({ devVarsPath, key, currentPort: expectedPort, explicit: true, fs }).message; +} + +/** + * An `workers/o11y/.dev.vars` bootstrapped before `DEV_ADMIN`, + * `O11Y_SESSION_SECRET`, `SLACK_WEBHOOK_URL` or `AE_SQL_TOKEN` existed can + * declare any of the four EMPTY — a declared-but-empty line silently wins + * over this run's own `--var`, so `/grafana/_o11y/login` 500s, Slack + * capture is off, or an AE panel 401s/renders empty. Warns for each case; + * does not fix the file (advisory, not fatal). Only fires for a STALE + * file — `o11yDevVarsPatch` fills all four on a fresh bootstrap. + * @returns {string[]} zero or more warning lines + */ +export function checkO11yDevVarsStaleness(devVarsPath, fs = defaultFs) { + const warnings = []; + const devAdmin = readDevVarsLine(devVarsPath, "DEV_ADMIN", fs); + if (devAdmin === "") { + warnings.push( + `${devVarsPath} declares DEV_ADMIN= (empty) — the local session bypass is OFF. ` + + `Edit the file to set DEV_ADMIN=dev@handsontable.com, or delete it and re-run for a fresh bootstrap.`, + ); + } + const sessionSecret = readDevVarsLine(devVarsPath, "O11Y_SESSION_SECRET", fs); + if (sessionSecret === "") { + warnings.push( + `${devVarsPath} declares O11Y_SESSION_SECRET= (empty) — this silently overrides this run's own ephemeral ` + + `--var (wrangler: .dev.vars always wins), so /grafana/_o11y/login will answer 500. Remove that line from ` + + `${devVarsPath}, or delete the file and re-run for a fresh bootstrap.`, + ); + } + const slackWebhookUrl = readDevVarsLine(devVarsPath, "SLACK_WEBHOOK_URL", fs); + if (slackWebhookUrl === "") { + warnings.push( + `${devVarsPath} declares SLACK_WEBHOOK_URL= (empty) — local Slack alert capture is OFF (an alert fires but ` + + `nothing is written for o11y-slack-capture.mjs to read). Edit the file to point it at the local capture ` + + `port, or delete it and re-run for a fresh bootstrap.`, + ); + } + const aeSqlToken = readDevVarsLine(devVarsPath, "AE_SQL_TOKEN", fs); + if (aeSqlToken === "") { + warnings.push( + `${devVarsPath} declares AE_SQL_TOKEN= (empty) — the local ClickHouse token is OFF, so every Analytics ` + + `Engine panel (Runner overview, Observability self, the Logs dashboard) will 401/render empty. Edit the ` + + `file to set a token matching compose.yml's local default, or delete it and re-run for a fresh bootstrap.`, + ); + } + return warnings; +} + +/** + * Resolves a fixture-replay gate secret + * (`O11Y_EXPORT_SECRET`/`SENTRY_HOOK_SECRET`) — the environment value + * first (covers `dev.mjs --replay`, a child process that inherits it), + * falling back to `devVarsPath` (a standalone + * `node scripts/o11y-replay-fixtures.mjs` run, with no `dev.mjs` process + * to inherit from). `""` when neither source has it. + */ +export function resolveReplaySecret(envValue, devVarsPath, key, readLine = readDevVarsLine) { + return envValue || readLine(devVarsPath, key) || ""; +} + +/** Reads one `KEY=value` line out of a `.dev.vars`-shaped file (quotes + * stripped), or `undefined` if absent/the file doesn't exist. Used for the + * O11Y_ENV=local sanity check (o11y) and the PREVIEW_HOST port-drift + * warning (api) — never for reading an actual secret value into a log. */ +export function readDevVarsLine(devVarsPath, key, fs = defaultFs) { + if (!fs.existsSync(devVarsPath)) return undefined; + const text = fs.readFileSync(devVarsPath, "utf8"); + const m = new RegExp(`^${key}=(.*)$`, "m").exec(text); + if (!m) return undefined; + return m[1].trim().replace(/^"(.*)"$/, "$1"); +} + +/** + * The origin `publicOrigin`/`grafana/login.ts` build the broker `return_to` + * against, and the `aud` every locally-minted session token is bound to. + * Must track `O11Y_DEV_PORT`, not `AUTHORING_DEV_PORT` — on a non-default + * o11y port, `publicOrigin`'s fallback would be silently wrong and every + * locally-minted token would fail its own `aud` check. + */ +export function o11yLocalPublicOrigin(ports) { + return `http://localhost:${ports.O11Y_DEV_PORT}`; +} + +/** Ephemeral, never-persisted hex secret for O11Y_SESSION_SECRET (or any + * other run-scoped local secret) — a fresh value every process start, + * injected only via `--var`/env, never written to a file. */ +export function ephemeralSecret(bytes = 32) { + return randomBytes(bytes).toString("hex"); +} + +/** `--var NAME:value` argument names whose value must never be echoed back + * to the log line `dev.mjs` prints for each spawned child — this run's + * own ephemeral `O11Y_SESSION_SECRET` is the only one + * today, but a future ephemeral local secret should be added here rather + * than growing a second ad hoc check. Does not (and cannot, from here) + * keep the value out of `ps` output — an argv is visible to any local + * process by nature — only out of dev.mjs's own terminal/log line. */ +const REDACT_VAR_NAMES = ["O11Y_SESSION_SECRET"]; + +/** Returns `args` with the value half of every `--var NAME:value` pair + * named in {@link REDACT_VAR_NAMES} replaced by ``, for + * `dev.mjs`'s own "spawning: ..." log line only — the real `args` array + * passed to `spawn()` is never touched, only a copy built for display. */ +export function redactArgsForLog(args) { + const out = []; + for (let i = 0; i < args.length; i += 1) { + const arg = args[i]; + if (args[i - 1] === "--var" && typeof arg === "string") { + const [name] = arg.split(":", 1); + if (REDACT_VAR_NAMES.includes(name)) { + out.push(`${name}:`); + continue; + } + } + out.push(arg); + } + return out; +} + +// --------------------------------------------------------------------------- +// Migrations (workers/api) +// --------------------------------------------------------------------------- + +/** Kept under the worker's own `.wrangler/` state dir (gitignored) so + * wiping local D1 state (`rm -rf workers/api/.wrangler/state`) also wipes + * the applied-migrations record — the two can never drift apart into + * "recorded applied, but the local DB is actually empty". */ +export function migrationRecordPath(workerDir) { + return path.join(workerDir, ".wrangler", "state", "dev-migrations-applied.json"); +} + +export function readAppliedMigrations(recordPath, fs = defaultFs) { + if (!fs.existsSync(recordPath)) return []; + try { + const parsed = JSON.parse(fs.readFileSync(recordPath, "utf8")); + return Array.isArray(parsed) ? parsed : []; + } catch { + return []; + } +} + +function writeAppliedMigrations(recordPath, files, fs) { + const dir = path.dirname(recordPath); + if (!fs.existsSync(dir)) fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(recordPath, JSON.stringify([...files].sort(), null, 2) + "\n"); +} + +function recordMigrationApplied(recordPath, file, fs) { + const already = readAppliedMigrations(recordPath, fs); + writeAppliedMigrations(recordPath, [...already, file], fs); +} + +const defaultFsWithReaddir = { ...defaultFs, readdirSync, statSync }; + +// --------------------------------------------------------------------------- +// Migration schema probe — adopts a local D1 already migrated by hand (or +// re-applied from 0001, where 0003_cost_ledger.sql's bare `ALTER TABLE` +// has no `IF NOT EXISTS` and dies with `duplicate column name`). +// --------------------------------------------------------------------------- + +/** Strips `--` line comments, then splits on `;` into individual statements. + * Good enough for this repo's own migrations (never a `;` inside a string + * literal or a trigger body) — not a general SQL parser. */ +function splitStatements(sql) { + return sql + .replace(/--[^\n]*/g, "") + .split(";") + .map((s) => s.trim()) + .filter(Boolean); +} + +/** + * Extracts a migration file's "checkable, additive" targets: `CREATE TABLE`, + * `CREATE [UNIQUE] INDEX`, and `ALTER TABLE ... ADD [COLUMN] ...` — the + * shapes whose effect can be checked against a schema snapshot. + * + * Deliberately conservative: any other statement shape (`DROP INDEX`, a + * bare `UPDATE`/`INSERT`, a `RENAME`) marks `checkable: false` — a + * name-only probe could otherwise see a stale schema and wrongly skip a + * migration whose whole point is to fix that schema (0002_buildkey_nonunique.sql). + * + * @returns {{ checkable: boolean, targets: Array< + * {type:'table', name:string} | {type:'index', name:string} | + * {type:'column', table:string, name:string} + * > }} + */ +export function parseMigrationTargets(sql) { + const statements = splitStatements(sql); + if (statements.length === 0) return { checkable: false, targets: [] }; + const targets = []; + for (const stmt of statements) { + let m; + if ((m = /^CREATE\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?["`[]?(\w+)["`\]]?/i.exec(stmt))) { + targets.push({ type: "table", name: m[1] }); + continue; + } + if ((m = /^CREATE\s+(?:UNIQUE\s+)?INDEX\s+(?:IF\s+NOT\s+EXISTS\s+)?["`[]?(\w+)["`\]]?/i.exec(stmt))) { + targets.push({ type: "index", name: m[1] }); + continue; + } + if ((m = /^ALTER\s+TABLE\s+["`[]?(\w+)["`\]]?\s+ADD\s+(?:COLUMN\s+)?["`[]?(\w+)["`\]]?/i.exec(stmt))) { + targets.push({ type: "column", table: m[1], name: m[2] }); + continue; + } + // Any other statement shape — this file's effect can't be probed + // generically, so it's never a candidate for "adopt without running". + return { checkable: false, targets: [] }; + } + return { checkable: true, targets }; +} + +async function queryD1Json(query, args) { + const raw = await query(args); + const parsed = JSON.parse(raw); + return parsed[0]?.results ?? []; +} + +/** + * One schema snapshot of the local D1: every table/index name in + * `sqlite_master`, plus the column list (`PRAGMA table_info`) for each table + * in `tables` — one `wrangler d1 execute --json` round trip per query, not + * per target (a `wrangler` spawn costs real wall-clock seconds, and a + * migrations dir touches only a handful of distinct tables via `ALTER TABLE` + * — `demos` is the only one today). + * @param {(args: string[]) => Promise|string} query injectable — + * real callers run `wrangler d1 execute --local --json --command=...` + * and return raw stdout; tests stub it. + */ +export async function snapshotLocalSchema({ dbName, tables, query }) { + const objects = await queryD1Json(query, [ + "d1", + "execute", + dbName, + "--local", + "--json", + "--command", + "SELECT type, name FROM sqlite_master WHERE type IN ('table','index')", + ]); + const tableNames = new Set(objects.filter((r) => r.type === "table").map((r) => r.name)); + const indexNames = new Set(objects.filter((r) => r.type === "index").map((r) => r.name)); + const columns = {}; + for (const table of tables) { + if (!tableNames.has(table)) { + columns[table] = new Set(); + continue; + } + const rows = await queryD1Json(query, ["d1", "execute", dbName, "--local", "--json", "--command", `PRAGMA table_info(${table})`]); + columns[table] = new Set(rows.map((r) => r.name)); + } + return { tableNames, indexNames, columns }; +} + +/** True when EVERY target a migration file declares (per `parseMigrationTargets`) + * already exists in `snapshot` — the condition for adopting the file as + * already-applied instead of running it. A file with zero targets (e.g. + * `checkable: false`, or a genuinely empty file) is never adopted — that + * would be vacuously "true" for a file whose effect was never checked. */ +export function isMigrationAlreadyApplied(targets, snapshot) { + if (targets.length === 0) return false; + return targets.every((t) => { + if (t.type === "table") return snapshot.tableNames.has(t.name); + if (t.type === "index") return snapshot.indexNames.has(t.name); + if (t.type === "column") return snapshot.columns[t.table]?.has(t.name) ?? false; + return false; + }); +} + +/** Typed error `applyMigrations` throws on any failure (a real `d1 execute` + * failure, or a schema-probe query failure) — carries what `dev.mjs` needs + * to print ONE clean line instead of letting a raw `execFileSync` stack + * trace reach the top-level `main().catch`. `file` is `null` for a + * probe-query failure (not tied to one specific migration file). */ +export class MigrationError extends Error { + constructor({ file, sqliteMessage, recordPath, action }) { + const where = file ? `migration ${file}` : "the local D1 schema probe"; + super(`${action} ${where} failed: ${sqliteMessage}`); + this.name = "MigrationError"; + this.file = file; + this.sqliteMessage = sqliteMessage; + this.recordPath = recordPath; + } +} + +/** Pulls the actual SQLite error text out of a failed `wrangler` invocation + * (an `execFileSync`-shaped error, with `.stderr`/`.stdout` Buffers or + * strings) — wrangler prints `✘ [ERROR] ` to stderr wrapped in ANSI + * color codes (confirmed against wrangler 4.108's own output for a real + * `duplicate column name` failure). Falls back to the last non-empty line + * of whatever output is available, then to the raw error's own `.message`, + * so this never throws trying to format another error. */ +function extractSqliteMessage(err) { + const chunk = (v) => (v === undefined || v === null ? "" : v.toString("utf8")); + const raw = chunk(err.stderr) + chunk(err.stdout); + const clean = raw.replace(/\x1b\[[0-9;]*m/g, ""); + const m = /✘\s*\[ERROR\]\s*(.+)/.exec(clean); + if (m) return m[1].trim(); + const lastLine = clean + .split("\n") + .map((l) => l.trim()) + .filter(Boolean) + .pop(); + return lastLine || err.message || String(err); +} + +function toMigrationError({ file, recordPath, cause, action }) { + if (cause instanceof MigrationError) return cause; + return new MigrationError({ file, sqliteMessage: extractSqliteMessage(cause), recordPath, action }); +} + +/** Every `NNNN_*.sql` file on disk, sorted, split into already-applied + * (per the record) and pending. Pure — takes no action. */ +export function planMigrations({ migrationsDir, recordPath, fs = defaultFsWithReaddir }) { + const applied = new Set(readAppliedMigrations(recordPath, fs)); + const files = fs + .readdirSync(migrationsDir) + .filter((f) => /^\d{4}_.*\.sql$/.test(f)) + .sort(); + const pending = files.filter((f) => !applied.has(f)); + return { files, applied: [...applied], pending }; +} + +/** + * Applies every pending migration, one `wrangler d1 execute --local` call + * per file (never `migrations apply --local` — see run-and-deploy.md for + * why 0003_cost_ledger.sql's bare `ALTER TABLE` makes that unsafe). Records + * each file as applied immediately after its own call succeeds, so a + * failure partway through never re-applies a landed file. + * + * If `query` is given, adopts (records as applied, without running) any + * pending file whose targets already exist — absorbs a local D1 migrated + * by hand with no applied-migrations record, without the raw + * `duplicate column name` error re-running would otherwise hit. + * + * @param {object} opts + * @param {string} opts.migrationsDir + * @param {string} opts.recordPath + * @param {string} opts.dbName + * @param {(args: string[]) => Promise|void} opts.run injectable wrangler runner + * @param {(args: string[]) => Promise|string} [opts.query] injectable schema probe; omit to skip it + * @param {(line: string) => void} [opts.log] + * @returns {Promise<{ applied: string[], adopted: string[] }>} + */ +export async function applyMigrations({ migrationsDir, recordPath, dbName, run, query, fs = defaultFsWithReaddir, log = () => {} }) { + const { pending } = planMigrations({ migrationsDir, recordPath, fs }); + if (pending.length === 0) { + log("migrations: nothing to apply (all recorded as already applied)"); + return { applied: [], adopted: [] }; + } + + const parsed = pending.map((file) => { + const sql = fs.readFileSync(path.join(migrationsDir, file), "utf8"); + return { file, ...parseMigrationTargets(sql) }; + }); + const alterTables = [...new Set(parsed.flatMap((p) => p.targets.filter((t) => t.type === "column").map((t) => t.table)))]; + + async function takeSnapshot() { + try { + return await snapshotLocalSchema({ dbName, tables: alterTables, query }); + } catch (cause) { + throw toMigrationError({ file: null, recordPath, cause, action: "probing" }); + } + } + + let snapshot = query ? await takeSnapshot() : null; + const appliedThisRun = []; + const adoptedThisRun = []; + for (const { file, checkable, targets } of parsed) { + if (snapshot && checkable && isMigrationAlreadyApplied(targets, snapshot)) { + log(`migration ${file}: every target already exists in the local D1 — adopting it as already applied (not running it)`); + adoptedThisRun.push(file); + recordMigrationApplied(recordPath, file, fs); + continue; + } + log(`applying migration ${file}`); + try { + await run(["d1", "execute", dbName, "--local", `--file=migrations/${file}`, "-y"]); + } catch (cause) { + throw toMigrationError({ file, recordPath, cause, action: "applying" }); + } + appliedThisRun.push(file); + recordMigrationApplied(recordPath, file, fs); + if (query && alterTables.length > 0) snapshot = await takeSnapshot(); + } + return { applied: appliedThisRun, adopted: adoptedThisRun }; +} + +/** Formats a caught {@link MigrationError} (or any other error) as ONE clean + * line for `dev.mjs`'s own top-level catch — never a raw stack trace — plus + * a recovery line naming the applied-migrations record and, if the caller + * passed `mentionReset`, the `--reset-local-db` flag. */ +export function formatMigrationError(err, { mentionReset = true } = {}) { + if (err instanceof MigrationError) { + const resetHint = mentionReset ? ", or wipe local D1 state and start over with `--reset-local-db`" : ""; + return ( + `error: ${err.message}\n` + + ` Recovery: inspect/edit the applied-migrations record at ${err.recordPath}${resetHint}.` + ); + } + return `error: migrations failed: ${err.message ?? err}`; +} + +/** Deletes workers/api's local D1 state (`.wrangler/state/v3/d1`) and the + * applied-migrations record (`migrationRecordPath`) — the two `dev-lib.mjs` + * otherwise keeps in lockstep (see `migrationRecordPath`'s own doc comment). + * Passing `--reset-local-db` on the CLI IS the confirmation (no interactive + * prompt from a script that's meant to run unattended); this function just + * logs exactly what it found and deleted, so the action is never silent. */ +export function resetLocalD1(apiDir, fs = defaultFs, log = () => {}) { + const stateDir = path.join(apiDir, ".wrangler", "state", "v3", "d1"); + const recordPath = migrationRecordPath(apiDir); + const hadState = fs.existsSync(stateDir); + const hadRecord = fs.existsSync(recordPath); + if (hadState) fs.rmSync(stateDir, { recursive: true, force: true }); + if (hadRecord) fs.rmSync(recordPath, { force: true }); + if (hadState || hadRecord) { + log(`--reset-local-db: deleted ${hadState ? stateDir : ""}${hadState && hadRecord ? " and " : ""}${hadRecord ? recordPath : ""}`); + } else { + log("--reset-local-db: no local D1 state or applied-migrations record found — nothing to delete"); + } +} + +// --------------------------------------------------------------------------- +// Docker +// --------------------------------------------------------------------------- + +export const DOCKER_NOT_RUNNING_MESSAGE = + "Docker does not appear to be running (`docker info` failed). This tier needs Docker " + + "for the Tier-2 session containers" + + " (and, for --tier=full, the o11y box's own container orchestration under wrangler dev). " + + "Start Docker Desktop (or your Docker daemon) and try again."; + +/** @param {(cmd: string, args: string[]) => void} execFileSyncImpl throws on + * a non-zero exit, like node:child_process's execFileSync with no + * `stdio: "ignore"` swallow — callers pass that in for real use, and a stub + * that throws/doesn't in tests. */ +export function isDockerAvailable(execFileSyncImpl) { + try { + execFileSyncImpl("docker", ["info"]); + return true; + } catch { + return false; + } +} + +/** + * Ctrl-C does NOT make wrangler's Sandbox-container orchestration tear + * down synchronously — containers can still be `Up` seconds after the + * wrapper process exits. + * + * This module never `docker stop`s a container it cannot prove it started: + * a name-pattern match on containers "new since this run started" is NOT a + * safe ownership proof — several worktrees routinely run `wrangler dev` at + * once, and "new since my snapshot" races this run's ENTIRE session, so a + * concurrent worktree's own container can look "new" here too and get + * silently killed. Neither wrangler nor the Sandbox SDK stamps a + * per-worktree Docker label to tell them apart. + * + * `listRunningContainers`/`possiblyLeftoverContainers` below only PRINT a + * report and a manual cleanup command — see `dev.mjs`'s teardown step. + */ +export function listRunningContainers(execFileSyncImpl) { + const out = execFileSyncImpl("docker", ["ps", "--format", "{{.ID}}\t{{.Names}}"]).toString(); + return out + .split("\n") + .map((line) => line.trim()) + .filter(Boolean) + .map((line) => { + const [id, ...rest] = line.split("\t"); + return { id, name: rest.join("\t") }; + }); +} + +const LEFTOVER_CONTAINER_NAME_RE = /handsontable-demos-(api|o11y)/; + +/** `before`: a Set of container ids running when this run started (from + * `listRunningContainers` at that point, ids only). `after`: + * `listRunningContainers()`'s result now. Returns the ones that are both + * new since `before` AND look like a `handsontable-demos-*` worker + * container — a REPORTING signal only (see this file's module-level doc + * comment above): `dev.mjs` prints these, it never stops them, because + * "new since my own snapshot" cannot distinguish this run's own container + * from one a concurrent, unrelated `wrangler dev` session (another + * worktree) started in the same window. */ +export function possiblyLeftoverContainers(before, after) { + return after.filter((c) => !before.has(c.id) && LEFTOVER_CONTAINER_NAME_RE.test(c.name)); +} + +/** The exact `docker ps` filter and manual `docker stop` command to print + * for the containers `possiblyLeftoverContainers` found — so a developer + * who recognizes them as genuinely this run's own can clean them up by + * hand, after confirming (e.g. `docker inspect` the ports/mounts) that + * they are not another worktree's session. */ +function describeLeftoverContainersForOperator(candidates) { + const ids = candidates.map((c) => c.id).join(" "); + const names = candidates.map((c) => c.name).join(", "); + return ( + `${candidates.length} container(s) look like this run's own worker containers ` + + `(new since startup, name matches ${LEFTOVER_CONTAINER_NAME_RE}) but this cannot be proven — ` + + `another worktree's concurrent \`wrangler dev\` session can produce the exact same signal. ` + + `NOT stopping them automatically. Names: ${names}. ` + + `Inspect first (e.g. \`docker inspect ${candidates[0]?.id ?? ""}\` for its ports/mounts), ` + + `list candidates with \`docker ps --filter "name=handsontable-demos-"\`, ` + + `and if you're sure they're yours: \`docker stop ${ids}\`.` + ); +} + +/** The whole leftover-container REPORT step `dev.mjs`'s teardown runs — + * factored out here (rather than left inline in `dev.mjs`) so it is + * directly unit-testable with a stubbed `execFileSyncImpl`, the same way + * every other side-effecting piece of this module is. This function calls + * `docker` only to LIST containers (`docker ps`, via + * `listRunningContainers`) — it never calls `docker stop`, no matter what + * it finds. Returns the candidates found (possibly empty) so a caller can + * assert on them without re-parsing the log line. */ +export function reportLeftoverContainers(before, execFileSyncImpl, logImpl) { + const after = listRunningContainers(execFileSyncImpl); + const candidates = possiblyLeftoverContainers(before, after); + if (candidates.length > 0) logImpl(describeLeftoverContainersForOperator(candidates)); + return candidates; +} + +/** Signals `dev.mjs` treats as "shut everything down cleanly". SIGHUP is + * included because every child is spawned `detached: true` (its own + * process group/session) — closing the terminal `dev.mjs` runs in sends + * SIGHUP to `dev.mjs` itself but not to those detached children, so + * without a handler here, the default SIGHUP action (immediate exit, no + * cleanup) would leave every child running. */ +export const SHUTDOWN_SIGNALS = ["SIGINT", "SIGTERM", "SIGHUP"]; + +// --------------------------------------------------------------------------- +// Container base-image pre-pull +// +// `wrangler dev`'s own local container build silently races a missing base +// image against Docker Hub, failing opaquely at container-start time. This +// section checks every base image is present BEFORE any worker is spawned, +// pulling what's missing with a bounded retry, and fails fast instead. +// --------------------------------------------------------------------------- + +/** + * Minimal string-aware JSONC comment stripper (line comments and block + * comments, respecting quoted strings/escapes) — same zero-dependency + * approach `pipeline/o11y-box-config.test.mjs` already uses, since + * `wrangler.jsonc` is JSONC, not plain JSON. Kept as this section's own + * private copy rather than a shared export, so this section stays + * self-contained. + */ +function stripJsonCommentsForContainerConfig(text) { + let result = ""; + let inString = false; + let inLineComment = false; + let inBlockComment = false; + for (let i = 0; i < text.length; i++) { + const c = text[i]; + const next = text[i + 1]; + if (inLineComment) { + if (c === "\n") { + inLineComment = false; + result += c; + } + continue; + } + if (inBlockComment) { + if (c === "*" && next === "/") { + inBlockComment = false; + i++; + } + continue; + } + if (inString) { + result += c; + if (c === "\\") { + result += next; + i++; + continue; + } + if (c === '"') inString = false; + continue; + } + if (c === '"') { + inString = true; + result += c; + continue; + } + if (c === "/" && next === "/") { + inLineComment = true; + i++; + continue; + } + if (c === "/" && next === "*") { + inBlockComment = true; + i++; + continue; + } + result += c; + } + return result; +} + +/** + * Reads a worker's `wrangler.jsonc` `containers[].image` paths — Dockerfile + * paths relative to `wranglerJsoncPath`'s own directory (this repo's own + * config already names them; this never hardcodes a second copy) — + * resolved to absolute paths. Returns `[]` when the config has no + * `containers` block. + * @param {string} wranglerJsoncPath + * @param {typeof defaultFs} [fs] + * @returns {string[]} + */ +export function readContainerDockerfilePaths(wranglerJsoncPath, fs = defaultFs) { + const raw = fs.readFileSync(wranglerJsoncPath, "utf8"); + const config = JSON.parse(stripJsonCommentsForContainerConfig(raw)); + const containers = config.containers ?? []; + const dir = path.dirname(wranglerJsoncPath); + return containers.map((c) => path.resolve(dir, c.image)); +} + +/** + * Extracts every base image a Dockerfile's `FROM` instructions need pulled + * from a registry (what `docker build` needs present locally first). + * Excludes a stage alias reference (`FROM ` for an earlier + * `AS `) and `FROM scratch`. Resolves `ARG`-declared build args from + * the Dockerfile's own default; an ARG with no default is left as the + * literal placeholder, which `docker pull` will visibly fail on. + * @param {string} dockerfileText + * @returns {string[]} base image refs, in FROM order, deduped + */ +export function parseDockerfileBaseImages(dockerfileText) { + const globalArgs = new Map(); + const stageNames = new Set(); + const seen = new Set(); + const images = []; + let sawFrom = false; + for (const rawLine of dockerfileText.split("\n")) { + const line = rawLine.trim(); + if (!line || line.startsWith("#")) continue; + let m; + if (!sawFrom && (m = /^ARG\s+([A-Za-z_][A-Za-z0-9_]*)(?:=(.*))?$/.exec(line))) { + let value = m[2]; + if (value !== undefined) { + value = value.trim(); + const q = /^"(.*)"$|^'(.*)'$/.exec(value); + if (q) value = q[1] ?? q[2]; + } + globalArgs.set(m[1], value); + continue; + } + if ((m = /^FROM\s+(.+)$/i.exec(line))) { + sawFrom = true; + const parts = m[1].trim().split(/\s+/); + let idx = 0; + while (parts[idx]?.startsWith("--")) idx++; + let ref = parts[idx]; + let alias; + const asIdx = parts.findIndex((p, i) => i > idx && /^as$/i.test(p)); + if (asIdx !== -1) alias = parts[asIdx + 1]; + ref = ref.replace(/\$\{?([A-Za-z_][A-Za-z0-9_]*)\}?/g, (whole, name) => { + const resolved = globalArgs.get(name); + return globalArgs.has(name) && resolved !== undefined ? resolved : whole; + }); + const referencesEarlierStage = stageNames.has(ref); + if (alias) stageNames.add(alias); + if (ref.toLowerCase() === "scratch") continue; + if (referencesEarlierStage) continue; + if (!seen.has(ref)) { + seen.add(ref); + images.push(ref); + } + } + } + return images; +} + +/** + * Maps a `dev.mjs` tier to the `wrangler.jsonc`(s) whose `containers[].image` + * Dockerfiles that tier's workers actually start. Tier "1" needs none — no + * worker with a container starts. Read from this repo's own config + * (requirement: derive from `containers[].image`, never hardcode the + * Dockerfile paths a second time). + * @param {"1"|"2"|"full"} tier + * @param {string} runnerRoot + * @returns {string[]} + */ +export function containerWranglerConfigsForTier(tier, runnerRoot) { + if (tier === "1") return []; + const apiConfig = path.join(runnerRoot, "workers", "api", "wrangler.jsonc"); + if (tier === "2") return [apiConfig]; + if (tier === "full") return [apiConfig, path.join(runnerRoot, "workers", "o11y", "wrangler.jsonc")]; + throw new Error(`containerWranglerConfigsForTier: unknown tier "${tier}"`); +} + +/** + * Every distinct base image ref this tier's Dockerfiles declare, across + * every `wrangler.jsonc` `containers[].image` Dockerfile the tier needs — + * deduped, first-seen order. Throws a clear error (not a raw `ENOENT`) if a + * `containers[].image` path doesn't exist on disk. + * @param {"1"|"2"|"full"} tier + * @param {string} runnerRoot + * @param {typeof defaultFs} [fs] + * @returns {string[]} + */ +export function collectTierBaseImages(tier, runnerRoot, fs = defaultFs) { + const refs = []; + const seen = new Set(); + for (const wranglerJsoncPath of containerWranglerConfigsForTier(tier, runnerRoot)) { + for (const dockerfilePath of readContainerDockerfilePaths(wranglerJsoncPath, fs)) { + if (!fs.existsSync(dockerfilePath)) { + throw new Error(`containers[].image path not found: ${dockerfilePath} (declared in ${wranglerJsoncPath})`); + } + const text = fs.readFileSync(dockerfilePath, "utf8"); + for (const ref of parseDockerfileBaseImages(text)) { + if (!seen.has(ref)) { + seen.add(ref); + refs.push(ref); + } + } + } + } + return refs; +} + +/** True when `--skip-image-check` was not passed and this tier actually + * needs container images checked (tier "1" never does). Factored out as + * its own pure function so the CLI wiring is directly unit-testable + * without spawning `dev.mjs` for every tier/flag combination. */ +export function shouldCheckContainerImages(tier, skipImageCheck) { + return (tier === "2" || tier === "full") && !skipImageCheck; +} + +/** True if `docker image inspect ` succeeds — the image already exists + * locally. Read-only, no network; never itself triggers a pull. + * @param {string} ref + * @param {(cmd: string, args: string[]) => void} execFileSyncImpl throws on + * a non-zero exit (a real `execFileSync` for real use; a stub for tests). + */ +export function isImagePresent(ref, execFileSyncImpl) { + try { + execFileSyncImpl("docker", ["image", "inspect", ref]); + return true; + } catch { + return false; + } +} + +/** Pulls the last non-empty line of a failed `execFileSync`-shaped error's + * OWN error output (ANSI stripped) — stderr first, falling back to stdout, + * then to `err.message`. stderr-first matters: `docker pull` writes + * per-layer progress to STDOUT and the real failure to STDERR, so + * concatenating the two and taking the last line could report a harmless + * progress line instead. */ +function lastErrorLine(err) { + const chunk = (v) => (v === undefined || v === null ? "" : v.toString("utf8")); + const lastNonEmptyLine = (text) => + text + .replace(/\x1b\[[0-9;]*m/g, "") + .split("\n") + .map((l) => l.trim()) + .filter(Boolean) + .pop(); + const stderrLine = lastNonEmptyLine(chunk(err?.stderr)); + if (stderrLine) return stderrLine; + const stdoutLine = lastNonEmptyLine(chunk(err?.stdout)); + if (stdoutLine) return stdoutLine; + return err?.message || String(err); +} + +/** + * Pulls `ref` with up to `maxAttempts` tries (default 3) and a short + * backoff between attempts, printing one plain progress line per attempt + * via `log` (the caller — `dev.mjs` — prefixes it `[images]`, matching this + * repo's own per-subsystem log convention). Never throws: returns + * `{ ok: true }` on the first successful pull, or + * `{ ok: false, lastErrorLine }` (the failing pull's last output line) once + * every attempt is exhausted. + * @param {object} opts + * @param {string} opts.ref + * @param {(cmd: string, args: string[]) => void} opts.execFileSyncImpl + * @param {number} [opts.maxAttempts] + * @param {number} [opts.backoffMs] base backoff; attempt N waits `backoffMs * N` + * @param {(line: string) => void} [opts.log] + * @param {(ms: number) => Promise} [opts.sleep] injectable so tests + * run instantly instead of waiting out a real backoff + * @returns {Promise<{ ok: true } | { ok: false, lastErrorLine: string }>} + */ +export async function pullImageWithRetry({ + ref, + execFileSyncImpl, + maxAttempts = 3, + backoffMs = 500, + log = () => {}, + sleep = (ms) => new Promise((r) => setTimeout(r, ms)), +}) { + let lastErr = ""; + for (let attempt = 1; attempt <= maxAttempts; attempt++) { + log(`pulling ${ref} (attempt ${attempt}/${maxAttempts})...`); + try { + execFileSyncImpl("docker", ["pull", ref]); + log(`pulled ${ref}`); + return { ok: true }; + } catch (err) { + lastErr = lastErrorLine(err); + log(`pull failed for ${ref} (attempt ${attempt}/${maxAttempts}): ${lastErr}`); + if (attempt < maxAttempts) await sleep(backoffMs * attempt); + } + } + return { ok: false, lastErrorLine: lastErr }; +} + +/** + * The whole pre-pull gate: for each ref in `refs` (in order), checks + * `isImagePresent` and, if missing, pulls it (`pullImageWithRetry`). + * Stops at the FIRST ref that cannot be pulled after every retry — no later + * ref is even checked — and returns which one failed, so the caller can + * print one clear message and exit before starting any worker. Never + * throws. + * @param {object} opts + * @param {string[]} opts.refs + * @param {(cmd: string, args: string[]) => void} opts.execFileSyncImpl + * @param {(line: string) => void} [opts.log] + * @param {number} [opts.maxAttempts] + * @param {number} [opts.backoffMs] + * @param {(ms: number) => Promise} [opts.sleep] + * @returns {Promise<{ ok: true } | { ok: false, ref: string, lastErrorLine: string }>} + */ +export async function ensureContainerImagesPresent({ refs, execFileSyncImpl, log = () => {}, maxAttempts = 3, backoffMs = 500, sleep }) { + for (const ref of refs) { + if (isImagePresent(ref, execFileSyncImpl)) { + log(`${ref} already present`); + continue; + } + log(`${ref} missing locally`); + const result = await pullImageWithRetry({ ref, execFileSyncImpl, maxAttempts, backoffMs, log, sleep }); + if (!result.ok) { + return { ok: false, ref, lastErrorLine: result.lastErrorLine }; + } + } + return { ok: true }; +} + +/** The one clean, actionable message `dev.mjs` prints (never a raw + * `execFileSync` stack trace) when {@link ensureContainerImagesPresent} + * stops on a ref it could not pull: which image, the Docker error's last + * line, and the exact `docker pull ...` command to retry by hand — plus + * the `--skip-image-check` escape hatch, for offline use when the images + * are already built. */ +export function formatImagePullFailure({ ref, lastErrorLine: line }) { + return ( + `error: could not pull required container base image ${ref}: ${line}\n` + + ` Retry by hand: docker pull ${ref}\n` + + ` Or skip this check entirely (e.g. offline, images already built): pass --skip-image-check.` + ); +} + +// --------------------------------------------------------------------------- +// Runtime staleness (packages/runtime dist vs src) +// --------------------------------------------------------------------------- + +/** True when `dist/` is missing, or any file under `src/` is newer than the + * newest file under `dist/` — the same "rebuild if stale" rule `pnpm dev` + * documents. Injectable fs for tests (a real run uses node:fs). */ +export function isRuntimeDistStale(runtimeDir, fs = defaultFsWithReaddir) { + const distDir = path.join(runtimeDir, "dist"); + if (!fs.existsSync(distDir)) return true; + const srcDir = path.join(runtimeDir, "src"); + const newestUnder = (dir) => { + let newest = 0; + const walk = (d) => { + for (const entry of fs.readdirSync(d, { withFileTypes: true })) { + const full = path.join(d, entry.name); + if (entry.isDirectory()) walk(full); + else { + const mtime = fs.statSync(full).mtimeMs; + if (mtime > newest) newest = mtime; + } + } + }; + walk(dir); + return newest; + }; + return newestUnder(srcDir) > newestUnder(distDir); +} + +// --------------------------------------------------------------------------- +// pnpm install staleness ("stale dependencies after a pull" dev-stack note) +// --------------------------------------------------------------------------- + +/** + * True when `pnpm-lock.yaml` is newer than `node_modules/.modules.yaml` — + * the file `pnpm install` itself rewrites on every run (verified: even a + * genuine no-op rewrites its mtime), so this self-heals once someone runs + * the command it recommends. A real `git pull` only touches a tracked + * file's mtime when its content changed, so an ordinary pull that never + * touches the lockfile never trips this. `node_modules` missing outright + * also counts as needing an install. Injectable fs/runnerRoot so tests + * exercise a temp directory, never this worktree's own `node_modules`. + */ +export function isPnpmInstallNeeded(runnerRoot = RUNNER_ROOT, fs = defaultFsWithReaddir) { + const lockfilePath = path.join(runnerRoot, "pnpm-lock.yaml"); + const modulesYamlPath = path.join(runnerRoot, "node_modules", ".modules.yaml"); + if (!fs.existsSync(lockfilePath)) return false; + if (!fs.existsSync(modulesYamlPath)) return true; + return fs.statSync(lockfilePath).mtimeMs > fs.statSync(modulesYamlPath).mtimeMs; +} + +/** The one clean, actionable message `dev.mjs` prints (never a silent 120s + * timeout followed by a buried wrangler build error) when + * {@link isPnpmInstallNeeded} is true. */ +export const PNPM_INSTALL_NEEDED_MESSAGE = + "pnpm-lock.yaml is newer than node_modules/.modules.yaml — dependencies look out of date for this checkout.\n" + + " Run: pnpm install --frozen-lockfile"; + +// --------------------------------------------------------------------------- +// Surfacing a wrangler build failure instead of waiting out the full +// readiness timeout ("stale dependencies after a pull" dev-stack note) +// --------------------------------------------------------------------------- + +/** + * True when `line` is wrangler/esbuild's own build-failure marker + * (`✘ [ERROR] `, ANSI stripped first) — EXCLUDING wrangler's + * runtime uncaught-exception logging, which reuses the same prefix for a + * request handler throwing at RUNTIME rather than esbuild failing to + * bundle. Real esbuild failures never start "Uncaught " (their own + * vocabulary is "Could not resolve", "Transform failed with N errors", + * "Unexpected token"). Returns the matched message (trimmed), or `null`. + */ +export function wranglerBuildErrorLine(line) { + const clean = line.replace(/\x1b\[[0-9;]*m/g, ""); + const m = /✘\s*\[ERROR\]\s*(.+)/.exec(clean); + if (!m) return null; + const message = m[1].trim(); + if (/^Uncaught\b/.test(message)) return null; + return message; +} + +/** + * Polls `fetchImpl(url)` every `pollMs` until it resolves, throwing once + * `timeoutMs` elapses. On every failed attempt — before ever sleeping again, + * however large `timeoutMs` is — also calls `getEarlyFailure()`; a truthy + * return (a {@link wranglerBuildErrorLine}) throws IMMEDIATELY with that + * line instead of waiting out the rest of `timeoutMs`. Built for `dev.mjs`'s + * two `wrangler dev` readiness waits: a + * build failure means wrangler either exits or hangs without ever binding + * its port, so without this the generic "never came up within 120000ms" + * timeout is the only signal for the full 2 minutes, burying wrangler's own + * much more specific error. Injectable fetch/sleep so a test never waits out a + * real network timeout or a real `setTimeout`. + */ +export async function waitForServer( + url, + timeoutMs, + label, + { + fetchImpl = fetch, + getEarlyFailure = () => undefined, + pollMs = 250, + sleepImpl = (ms) => new Promise((r) => setTimeout(r, ms)), + } = {}, +) { + const deadline = Date.now() + timeoutMs; + for (;;) { + try { + await fetchImpl(url); + return; + } catch (err) { + const earlyFailure = getEarlyFailure(); + if (earlyFailure) { + throw new Error(`${label} on ${url} failed to build: ${earlyFailure}`); + } + if (Date.now() > deadline) { + throw new Error(`${label} on ${url} never came up within ${timeoutMs}ms: ${err}`); + } + await sleepImpl(pollMs); + } + } +} + +// --------------------------------------------------------------------------- +// Spawn plan +// --------------------------------------------------------------------------- + +/** + * The set of long-running, log-prefixed, SIGINT-forwarded child processes + * for a tier — NOT one-shot setup steps (Docker check, migrations, docker + * compose up/down, the runtime build), which run before/after this plan. + * + * @param {"1"|"2"|"full"} tier + * @param {ReturnType} ports + * @param {{ replay?: boolean, sessionSecret?: string }} [opts] + * @returns {{ name: string, bin: string, args: string[], cwd: string, env: Record }[]} + */ +export function buildPlan(tier, ports, opts = {}) { + const plan = []; + // Env injection, not a written `.env.local` (research doc §5, and the + // exact pattern `e2e/o11y-local.spec.ts` already proves for a build): + // nothing lands on disk, so nothing can leak the dev-login bypass into a + // later "real" build the way a forgotten `.env.local` can. `VITE_API_BASE` + // points at THIS dev server (not directly at the API worker) so + // `vite.config.ts`'s own `/api`/`/d`/`/embed` proxy is what actually talks + // to the API worker — required for `?mode=full`'s single-origin framing + // rule (AGENTS.md). + const appEnv = {}; + if (tier === "2" || tier === "full") { + appEnv.VITE_API_BASE = `http://localhost:${ports.AUTHORING_DEV_PORT}`; + appEnv.VITE_DEV_USER = "dev@handsontable.com"; + appEnv.API_DEV_PORT = String(ports.API_DEV_PORT); + } + if (tier === "full") { + appEnv.VITE_TELEMETRY_LOCAL = "1"; + appEnv.O11Y_DEV_PORT = String(ports.O11Y_DEV_PORT); + // Admin.tsx's "Open Grafana" link: same origin as the login redirect + // (`o11yLocalPublicOrigin`/O11Y_LOCAL_PUBLIC_ORIGIN below) so the + // DEV_ADMIN bypass's session cookie lands on the host the browser is + // actually asked to open — NOT O11Y_GRAFANA_PORT (compose.yml's + // container-internal port), which the o11y worker proxies to, not the + // browser reaches directly. + appEnv.VITE_GRAFANA_URL = `${o11yLocalPublicOrigin(ports)}/grafana/`; + } + plan.push({ + name: "app", + bin: "node_modules/.bin/vite", + args: ["--port", String(ports.AUTHORING_DEV_PORT), "--strictPort"], + cwd: "apps/authoring", + env: appEnv, + }); + if (tier === "2" || tier === "full") { + const apiVars = []; + if (tier === "full") { + apiVars.push( + "--var", + `RUNNER_EVENTS_CLICKHOUSE_URL:http://localhost:${ports.O11Y_CLICKHOUSE_PORT}`, + "--var", + "AE_SQL_TOKEN:local-dev-token", + ); + } + plan.push({ + name: "api", + bin: "node_modules/.bin/wrangler", + args: ["dev", "--port", String(ports.API_DEV_PORT), "--inspector-port", String(ports.API_DEV_INSPECTOR_PORT), ...apiVars], + cwd: "workers/api", + env: {}, + }); + } + if (tier === "full") { + const sessionSecret = opts.sessionSecret ?? ephemeralSecret(); + plan.push({ + name: "o11y", + bin: "node_modules/.bin/wrangler", + args: [ + "dev", + "--port", + String(ports.O11Y_DEV_PORT), + "--inspector-port", + String(ports.O11Y_DEV_INSPECTOR_PORT), + "--var", + `O11Y_SESSION_SECRET:${sessionSecret}`, + "--var", + `O11Y_LOCAL_MINIO_PORT:${ports.O11Y_MINIO_PORT}`, + "--var", + `O11Y_LOCAL_CLICKHOUSE_PORT:${ports.O11Y_CLICKHOUSE_PORT}`, + "--var", + `RUNNER_EVENTS_CLICKHOUSE_URL:http://localhost:${ports.O11Y_CLICKHOUSE_PORT}`, + "--var", + `O11Y_LOCAL_PUBLIC_ORIGIN:${o11yLocalPublicOrigin(ports)}`, + ], + cwd: "workers/o11y", + env: {}, + }); + plan.push({ + name: "slack", + bin: process.execPath, + args: ["scripts/o11y-slack-capture.mjs", "--port", String(ports.O11Y_SLACK_CAPTURE_PORT)], + cwd: ".", + env: {}, + }); + } + return plan; +} + +// --------------------------------------------------------------------------- +// --fresh: compose.yml's minio/clickhouse use named volumes, so a plain +// restart KEEPS logs/metrics; `workers/o11y/.wrangler/state` persists too. +// `--fresh` wipes both together, so the ledger and R2 data can never +// diverge. See `resetO11yLocalState`/`detectO11yStateDivergence` below. +// --------------------------------------------------------------------------- + +// --------------------------------------------------------------------------- +// Per-worktree compose project name: a fixed default meant every worktree's +// `--tier=full` resolved to the SAME docker compose project, so `--fresh` +// or even Ctrl-C in one worktree could silently affect another's stack. +// `dev.mjs`, `stop-roundtrip.mjs` and the docs all derive their default +// from the ONE helper below, so they can never drift apart. +// --------------------------------------------------------------------------- + +/** + * This worktree's own default `COMPOSE_PROJECT_NAME` — a pure, stable + * function of `runnerRoot`'s absolute path (not the git remote/branch, so + * two worktrees of the same repo get distinct names; not random, so the + * SAME worktree gets the SAME name across restarts). `sha256` truncated to + * 8 hex chars avoids leaking the worktree's directory/username into a + * project name a developer might paste elsewhere. + * + * @param {string} [runnerRoot] + * @returns {string} + */ +export function defaultComposeProjectName(runnerRoot = RUNNER_ROOT) { + const hash = createHash("sha256").update(runnerRoot).digest("hex").slice(0, 8); + return `o11y-dev-${hash}`; +} + +/** + * Resolves the `COMPOSE_PROJECT_NAME` a `--tier=full` run actually uses: an + * explicit env override always wins (unchanged behaviour — a developer who + * deliberately shares one project across worktrees, or picks their own name, + * is never overridden), and {@link defaultComposeProjectName} is the + * fallback instead of the old fixed `"o11y-dev"` literal. + * + * @param {NodeJS.ProcessEnv} [env] + * @param {string} [runnerRoot] + * @returns {string} + */ +export function resolveComposeProjectName(env = process.env, runnerRoot = RUNNER_ROOT) { + return env.COMPOSE_PROJECT_NAME || defaultComposeProjectName(runnerRoot); +} + +/** The exact `docker compose ... down` argv, with `-v` appended only when + * `fresh` — factored out so `dev.mjs`'s normal (kept-data) Ctrl-C teardown + * and `resetO11yLocalState`'s `--fresh` wipe are provably running the same + * command shape with only the one intentional difference, instead of two + * independently-typed argv literals that could silently drift apart. */ +export function composeDownArgs(composeFile, { fresh = false } = {}) { + const args = ["compose", "-f", composeFile, "down"]; + if (fresh) args.push("-v"); + return args; +} + +/** One line, printed once at startup for `--tier=full` (`dev.mjs`) — the + * point-3 "startup mode line" the task/report needs to be able to point at + * verbatim. */ +export function o11yDevDataModeLine(fresh) { + return fresh ? "o11y local data: fresh" : "o11y local data: kept (MinIO/ClickHouse volumes + o11y worker state)"; +} + +/** + * `--fresh`'s whole job: wipe compose's named volumes (minio/clickhouse) + * AND `workers/o11y/.wrangler/state` TOGETHER, so the two never diverge. + * Leaves the API worker's local D1 alone — that's `--reset-local-db`'s job. + * + * `composeFile`/`composeEnv` optional: `o11y-dev.mjs` never runs compose, + * so its own `--fresh` wipes only the o11y worker state. Passing them + * scopes `down -v` to `composeEnv.COMPOSE_PROJECT_NAME` — never another + * worktree's, unless the caller shares a project name on purpose. + * + * @param {object} opts + * @param {string} opts.o11yDir + * @param {string} [opts.composeFile] + * @param {NodeJS.ProcessEnv} [opts.composeEnv] + * @param {(cmd: string, args: string[], opts?: object) => void} opts.execFileSyncImpl + * real callers pass `(cmd, args, o) => execFileSync(cmd, args, { cwd: RUNNER_ROOT, env: composeEnv, stdio: "inherit", ...o })` + * @param {typeof defaultFs} [opts.fs] + * @param {(line: string) => void} [opts.log] + * @returns {{ composeDownRan: boolean, stateDirRemoved: boolean, stateDir: string }} + */ +export function resetO11yLocalState({ o11yDir, composeFile, composeEnv, execFileSyncImpl, fs = defaultFs, log = () => {} }) { + let composeDownRan = false; + if (composeFile) { + log(`--fresh: docker compose down -v (project ${composeEnv?.COMPOSE_PROJECT_NAME ?? "?"})`); + execFileSyncImpl("docker", composeDownArgs(composeFile, { fresh: true }), { env: composeEnv }); + composeDownRan = true; + } + const stateDir = path.join(o11yDir, ".wrangler", "state"); + const stateDirRemoved = fs.existsSync(stateDir); + if (stateDirRemoved) { + fs.rmSync(stateDir, { recursive: true, force: true }); + log(`--fresh: deleted ${stateDir} (InboxWriter ledger, dedupe hashes, local R2 inbox objects)`); + } else { + log(`--fresh: no ${stateDir} found — nothing to delete there`); + } + return { composeDownRan, stateDirRemoved, stateDir }; +} + +/** + * Brings up `minio`/`clickhouse` via `docker compose ... up -d --wait`, and + * — if that call itself throws — tears the SAME project back down (never + * `-v`: a startup FAILURE, not `--fresh`'s wipe) before rethrowing, instead + * of leaving whichever service DID start orphaned with no teardown. + * + * A throw from `up` happens BEFORE `dev.mjs`'s own teardown step is ever + * pushed, so it would otherwise propagate straight past cleanup to + * `main().catch`, which only logs and exits. Injectable `execFileSyncImpl` + * makes this unit-testable with a stub instead of a real `docker compose`. + * + * @param {object} opts + * @param {string} opts.composeFile + * @param {NodeJS.ProcessEnv} opts.composeEnv + * @param {(cmd: string, args: string[], opts?: object) => void} opts.execFileSyncImpl + * real callers pass `(cmd, args, o) => execFileSync(cmd, args, { cwd: RUNNER_ROOT, stdio: "inherit", ...o })` + * @param {(line: string) => void} [opts.log] + */ +export function bringUpO11yCompose({ composeFile, composeEnv, execFileSyncImpl, log = () => {} }) { + try { + execFileSyncImpl("docker", ["compose", "-f", composeFile, "up", "-d", "--wait", "minio", "clickhouse"], { env: composeEnv }); + } catch (err) { + log(`startup failed (${err.message}) — tearing down minio + clickhouse (data kept)`); + try { + execFileSyncImpl("docker", composeDownArgs(composeFile), { env: composeEnv }); + } catch (downErr) { + log(`teardown after startup failure also failed: ${downErr.message}`); + } + throw err; + } +} + +/** + * Reads the committed-key count straight out of the InboxWriter DO's local + * SQLite storage (wrangler's local dev backing store). A `done:` entry + * is a key the ledger considers drained and will never look at again on + * its own — if the data it points at (Loki chunks in MinIO) is gone, this + * count is what makes that silent, since nothing else re-checks it. + * + * Best-effort: returns 0 (never throws) if `node:sqlite` isn't available, + * the state dir doesn't exist, or a `.sqlite` file can't be opened — the + * safe failure direction for a warning-only check. + * + * @param {string} o11yDir + * @param {typeof defaultFsWithReaddir} [fs] + * @returns {Promise} + */ +export async function o11yLedgerCommittedKeyCount(o11yDir, fs = defaultFsWithReaddir) { + const doDir = path.join(o11yDir, ".wrangler", "state", "v3", "do"); + if (!fs.existsSync(doDir)) return 0; + let DatabaseSync; + try { + ({ DatabaseSync } = await import("node:sqlite")); + } catch { + return 0; + } + let entries; + try { + entries = fs.readdirSync(doDir); + } catch { + return 0; + } + let total = 0; + for (const entry of entries.filter((name) => name.includes("InboxWriter"))) { + const dir = path.join(doDir, entry); + let files; + try { + files = fs.readdirSync(dir); + } catch { + continue; + } + for (const file of files) { + if (!file.endsWith(".sqlite") || file === "metadata.sqlite") continue; + let db; + try { + db = new DatabaseSync(path.join(dir, file), { readOnly: true }); + const row = db.prepare(`SELECT count(*) as c FROM _cf_KV WHERE key LIKE '${DONE_PREFIX_SQL_LIKE}'`).get(); + total += Number(row?.c ?? 0); + } catch { + // Not this DO's storage shape, or the file is locked/corrupt — + // best-effort, skip it. + } finally { + try { + db?.close(); + } catch { + // already closed/never opened + } + } + } + } + return total; +} + +/** `ledger.ts`'s `DONE_PREFIX` ("done:"), as a SQL `LIKE` pattern — kept as + * its own named constant (rather than string-building `"done:" + "%"` + * inline) so it reads as the same contract value that file documents, not + * an ad hoc string. */ +const DONE_PREFIX_SQL_LIKE = "done:%"; + +/** Finds the real docker volume name compose created for `volumeKey` via + * compose's own `com.docker.compose.project`/`.volume` labels, never by + * guessing compose's project-name sanitization rule. Returns `null` if no + * such volume exists — every caller treats that the same as "no data". + * @param {(cmd: string, args: string[]) => Buffer|string} execFileSyncImpl + */ +export function findComposeVolume({ composeProjectName, volumeKey, execFileSyncImpl }) { + const out = execFileSyncImpl("docker", [ + "volume", + "ls", + "-q", + "--filter", + `label=com.docker.compose.project=${composeProjectName}`, + "--filter", + `label=com.docker.compose.volume=${volumeKey}`, + ]) + .toString() + .trim(); + if (!out) return null; + return out.split("\n")[0].trim(); +} + +/** + * The divergent case: named volumes empty/gone while the o11y worker's + * ledger still has `done:` keys pointing at data that no longer exists. + * Checks MinIO only (not ClickHouse, written directly, outside the ledger) + * — existence, not emptiness, since MinIO creates the bucket on every `up`. + * + * Warns and points at `--fresh` rather than auto-reopening: this runs + * before the o11y worker even starts, so an automatic reopen would need + * its own post-startup step. + * + * @param {object} opts + * @param {string} opts.composeProjectName + * @param {string} opts.o11yDir + * @param {(cmd: string, args: string[]) => Buffer|string} opts.execFileSyncImpl + * @param {typeof defaultFsWithReaddir} [opts.fs] + * @param {(o11yDir: string, fs: typeof defaultFsWithReaddir) => Promise} [opts.countCommittedLedgerKeys] + * @returns {Promise<{ divergent: boolean, committedCount: number }>} + */ +export async function detectO11yStateDivergence({ + composeProjectName, + o11yDir, + execFileSyncImpl, + fs = defaultFsWithReaddir, + countCommittedLedgerKeys = o11yLedgerCommittedKeyCount, +}) { + const committedCount = await countCommittedLedgerKeys(o11yDir, fs); + if (committedCount === 0) return { divergent: false, committedCount: 0 }; + const minioVolume = findComposeVolume({ composeProjectName, volumeKey: "minio-data", execFileSyncImpl }); + return { divergent: minioVolume === null, committedCount }; +} + +/** The warning line `dev.mjs` prints when {@link detectO11yStateDivergence} + * finds the divergent case. */ +export function formatO11yDivergenceWarning(committedCount) { + return ( + `warning: workers/o11y's local ledger has ${committedCount} committed key(s) marking data as already drained, ` + + `but this project's MinIO volume doesn't exist (removed by hand, e.g. \`docker volume rm\`?) — that data is gone ` + + `and these keys will NEVER be re-drained on their own. Run \`node scripts/dev.mjs --tier=full --fresh\` to wipe ` + + `the o11y worker state too so both stores agree again, or — to keep what R2 still has (7-day retention) instead ` + + `of starting over — once the worker is up: POST /grafana/_o11y/reopen for the affected time window.` + ); +} diff --git a/runner/scripts/dev.mjs b/runner/scripts/dev.mjs new file mode 100644 index 0000000000..2382411e1e --- /dev/null +++ b/runner/scripts/dev.mjs @@ -0,0 +1,554 @@ +#!/usr/bin/env node +// One-command local dev for the runner (`pnpm dev`/`dev:live`/`dev:full`). +// See docs/run-and-deploy.md's "Run locally" section for the walkthrough. +// wrangler's `.dev.vars` always wins over `--var`, which is why +// `dev-lib.mjs`'s bootstrap/patch writes non-secret defaults into +// `.dev.vars` up front. `pnpm o11y:dev` (scripts/o11y-dev.mjs) shares this +// file's dev-lib.mjs helpers for the o11y-only entry point. + +import { spawn, execFileSync } from "node:child_process"; +import { existsSync, mkdirSync } from "node:fs"; +import path from "node:path"; +import { + RUNNER_ROOT, + HELP_TEXT, + parseArgs, + resolvePorts, + assertNoPortCollisions, + bootstrapDevVars, + o11yDevVarsPatch, + O11Y_DEVVARS_STRIP_KEYS, + fillEmptyDevVarsSecrets, + O11Y_DEVVARS_AUTOFILL_SECRET_KEYS, + resolveDevVarsPortAdoption, + checkO11yDevVarsStaleness, + readDevVarsLine, + ephemeralSecret, + migrationRecordPath, + applyMigrations, + formatMigrationError, + resetLocalD1, + DOCKER_NOT_RUNNING_MESSAGE, + isDockerAvailable, + isRuntimeDistStale, + isPnpmInstallNeeded, + PNPM_INSTALL_NEEDED_MESSAGE, + wranglerBuildErrorLine, + waitForServer, + buildPlan, + listRunningContainers, + reportLeftoverContainers, + SHUTDOWN_SIGNALS, + redactArgsForLog, + PORT_DEFAULTS, +} from "./dev-lib.mjs"; +import { shouldCheckContainerImages, collectTierBaseImages, ensureContainerImagesPresent, formatImagePullFailure } from "./dev-lib.mjs"; +import { + composeDownArgs, + o11yDevDataModeLine, + resetO11yLocalState, + detectO11yStateDivergence, + formatO11yDivergenceWarning, + bringUpO11yCompose, + resolveComposeProjectName, +} from "./dev-lib.mjs"; + +const COLORS = { + app: "\x1b[36m", // cyan + api: "\x1b[33m", // yellow + o11y: "\x1b[35m", // magenta + compose: "\x1b[34m", // blue + slack: "\x1b[32m", // green + build: "\x1b[90m", // grey + dev: "\x1b[97m", // bright white + images: "\x1b[96m", // bright cyan +}; +const RESET = "\x1b[0m"; + +function prefixed(name) { + const color = COLORS[name] ?? ""; + return (line) => `${color}[${name}]${RESET} ${line}`; +} + +function log(name, line) { + console.log(prefixed(name)(line)); +} + +// `onLine`: lets the caller watch each raw line (before the `[name]` +// prefix) for a wrangler build-error marker, reported immediately instead +// of waiting out the full readiness timeout. Optional. +function pipeLines(stream, name, sink = console.log, onLine = () => {}) { + let buf = ""; + stream.on("data", (chunk) => { + buf += chunk.toString(); + const lines = buf.split("\n"); + buf = lines.pop() ?? ""; + for (const line of lines) { + sink(prefixed(name)(line)); + onLine(line); + } + }); + stream.on("end", () => { + if (buf) { + sink(prefixed(name)(buf)); + onLine(buf); + } + }); +} + +function runWrangler(cwd, args) { + execFileSync(path.join(cwd, "node_modules", ".bin", "wrangler"), args, { + cwd, + stdio: "inherit", + }); +} + +/** Same binary, but stdio is captured (not inherited) and returned as a + * string — used only for the migration schema probe's read-only + * `d1 execute ... --json` queries, whose JSON output on stdout must be + * parsed rather than printed. A real apply (`runWrangler`, above) keeps + * inheriting stdio so its output stays visible live. */ +function runWranglerCapture(cwd, args) { + return execFileSync(path.join(cwd, "node_modules", ".bin", "wrangler"), args, { + cwd, + encoding: "utf8", + }); +} + +async function main() { + const { help, tier, replay, resetLocalDb, fresh, skipImageCheck, errors } = parseArgs(process.argv.slice(2)); + if (help) { + console.log(HELP_TEXT); + process.exit(0); + } + if (errors.length > 0) { + console.error(errors.map((e) => `error: ${e}`).join("\n")); + console.error(""); + console.error(HELP_TEXT); + process.exit(1); + } + + // A pull that adds a dependency with node_modules never reinstalled + // would otherwise run to the full 120s readiness timeout before a + // generic "worker never came up", burying the real wrangler error. + if (isPnpmInstallNeeded()) { + console.error(`error: ${PNPM_INSTALL_NEEDED_MESSAGE}`); + process.exit(1); + } + + let ports; + try { + ports = resolvePorts(tier, process.env); + } catch (err) { + console.error(`error: ${err.message}`); + process.exit(1); + } + + // Checked BEFORE --reset-local-db: reversed, a run with Docker not + // running would delete workers/api's local D1 state and then + // immediately exit on the Docker error — a surprising side effect for a + // run that otherwise did nothing. + let containersBefore = new Set(); + if (tier === "2" || tier === "full") { + if (!isDockerAvailable((cmd, args) => execFileSync(cmd, args, { stdio: "ignore" }))) { + console.error(`error: ${DOCKER_NOT_RUNNING_MESSAGE}`); + process.exit(1); + } + } + + if (resetLocalDb) { + resetLocalD1(path.join(RUNNER_ROOT, "workers", "api"), undefined, (line) => log("dev", line)); + } + + if (tier === "2" || tier === "full") { + // Pre-pull gate: every base image this tier's Dockerfiles need must be + // present BEFORE any worker starts, or `wrangler dev`'s container build + // fails silently and only surfaces later, opaquely, at session start. + if (shouldCheckContainerImages(tier, skipImageCheck)) { + const refs = collectTierBaseImages(tier, RUNNER_ROOT); + if (refs.length > 0) { + log("images", `checking ${refs.length} base image(s) needed for --tier=${tier}`); + const dockerExec = (cmd, args) => execFileSync(cmd, args, { stdio: ["ignore", "pipe", "pipe"] }); + const result = await ensureContainerImagesPresent({ + refs, + execFileSyncImpl: dockerExec, + log: (line) => log("images", line), + }); + if (!result.ok) { + console.error(formatImagePullFailure(result)); + process.exit(1); + } + } + } else if (skipImageCheck) { + log("images", "--skip-image-check: skipping the container base-image pre-pull check"); + } + + // Baseline for the leftover-container REPORT on shutdown (this run + // never stops a container it cannot prove it started — see + // dev-lib.mjs's module-level doc comment on `possiblyLeftoverContainers` + // for why). + containersBefore = new Set(listRunningContainers((cmd, args) => execFileSync(cmd, args)).map((c) => c.id)); + } + + // ---- runtime build (blocking, all tiers) -------------------------------- + const runtimeDir = path.join(RUNNER_ROOT, "packages", "runtime"); + if (isRuntimeDistStale(runtimeDir)) { + log("build", "packages/runtime/dist is stale (or missing) — building @handsontable/demo-runtime first"); + execFileSync("pnpm", ["--filter", "@handsontable/demo-runtime", "build"], { + cwd: RUNNER_ROOT, + stdio: "inherit", + }); + } else { + log("build", "packages/runtime/dist is up to date — skipping rebuild"); + } + + // ---- workers/api setup (tier 2 + full) ---------------------------------- + const apiDir = path.join(RUNNER_ROOT, "workers", "api"); + if (tier === "2" || tier === "full") { + const devVarsPath = path.join(apiDir, ".dev.vars"); + const examplePath = path.join(apiDir, ".dev.vars.example"); + const { created } = bootstrapDevVars({ examplePath, devVarsPath }); + if (created) log("api", `created ${path.relative(RUNNER_ROOT, devVarsPath)} from .dev.vars.example`); + + // PREVIEW_HOST port adoption: `.dev.vars` always wins over `--var`, so + // adopt a pre-existing file's declared port instead of starting + // pointed at a port `.dev.vars` will silently override anyway. + const apiPortExplicit = process.env.API_DEV_PORT !== undefined && process.env.API_DEV_PORT !== ""; + const apiPortAdoption = resolveDevVarsPortAdoption({ + devVarsPath, + key: "PREVIEW_HOST", + currentPort: ports.API_DEV_PORT, + explicit: apiPortExplicit, + }); + if (apiPortAdoption.message) log("api", `${apiPortAdoption.adopted ? "info" : "warning"}: ${apiPortAdoption.message}`); + if (apiPortAdoption.adopted) { + ports.API_DEV_PORT = apiPortAdoption.port; + try { + assertNoPortCollisions(ports); + } catch (err) { + console.error(`error: ${err.message}`); + process.exit(1); + } + } + + const recordPath = migrationRecordPath(apiDir); + let migrations; + try { + migrations = await applyMigrations({ + migrationsDir: path.join(apiDir, "migrations"), + recordPath, + dbName: "handsontable-demos", + run: (args) => runWrangler(apiDir, args), + query: (args) => runWranglerCapture(apiDir, args), + log: (line) => log("api", line), + }); + } catch (err) { + // Clean, single-line failure — never a raw execFileSync stack trace. + // Nothing has been spawned yet, so exiting here leaves nothing to clean up. + console.error(formatMigrationError(err)); + process.exit(1); + } + if (migrations.applied.length > 0) log("api", `applied ${migrations.applied.length} migration(s): ${migrations.applied.join(", ")}`); + if (migrations.adopted.length > 0) { + log( + "api", + `adopted ${migrations.adopted.length} pre-existing migration(s) without running them (local D1 already matched): ${migrations.adopted.join(", ")}`, + ); + } + } + + // ---- workers/o11y + compose setup (full only) --------------------------- + const o11yDir = path.join(RUNNER_ROOT, "workers", "o11y"); + const teardownSteps = []; + if (tier === "2" || tier === "full") { + // Ctrl-C does not make wrangler's own Tier-2 Sandbox-container + // orchestration tear itself down synchronously — a session's + // containers can still be `Up` several seconds after this wrapper has + // already exited. This step never `docker stop`s a container it cannot + // prove it started (see dev-lib.mjs's `possiblyLeftoverContainers` doc + // comment: several worktrees running `wrangler dev` at once is the + // NORMAL case) — it only PRINTS a report and the exact manual + // `docker ps`/`docker stop` commands, so a developer can decide by hand + // after confirming (e.g. `docker inspect`) what a container actually is. + teardownSteps.push(() => { + reportLeftoverContainers(containersBefore, (cmd, args) => execFileSync(cmd, args), (msg) => log("dev", msg)); + }); + } + if (tier === "full") { + const devVarsPath = path.join(o11yDir, ".dev.vars"); + const examplePath = path.join(o11yDir, ".dev.vars.example"); + const { created, patched, stripped } = bootstrapDevVars({ + examplePath, + devVarsPath, + patch: o11yDevVarsPatch(ports), + stripKeys: O11Y_DEVVARS_STRIP_KEYS, + }); + if (created) { + log("o11y", `created ${path.relative(RUNNER_ROOT, devVarsPath)} from .dev.vars.example`); + if (patched.length) log("o11y", `filled in local-dev defaults for: ${patched.join(", ")}`); + if (stripped.length) log("o11y", `left ${stripped.join(", ")} undeclared so this run's own ephemeral --var takes effect`); + } + // Runs on EVERY invocation, not only a fresh bootstrap — a + // pre-existing `.dev.vars` needs these two filled too. Only the key + // NAME is logged, never the generated value. + const { filled } = fillEmptyDevVarsSecrets({ devVarsPath, keys: O11Y_DEVVARS_AUTOFILL_SECRET_KEYS }); + if (filled.length) { + log("o11y", `filled in ephemeral local-dev values for: ${filled.join(", ")} (values never logged)`); + } + const envLine = readDevVarsLine(devVarsPath, "O11Y_ENV"); + if (envLine !== "local") { + console.error(`error: ${devVarsPath} must set O11Y_ENV=local — refusing to start against a non-local config`); + process.exit(1); + } + // Same PREVIEW_HOST-style port adoption, for SLACK_WEBHOOK_URL — only + // out of sync on a pre-existing file with a since-changed port env var. + const slackPortExplicit = process.env.O11Y_SLACK_CAPTURE_PORT !== undefined && process.env.O11Y_SLACK_CAPTURE_PORT !== ""; + const slackAdoption = resolveDevVarsPortAdoption({ + devVarsPath, + key: "SLACK_WEBHOOK_URL", + currentPort: ports.O11Y_SLACK_CAPTURE_PORT, + explicit: slackPortExplicit, + }); + if (slackAdoption.message) log("o11y", `${slackAdoption.adopted ? "info" : "warning"}: ${slackAdoption.message}`); + if (slackAdoption.adopted) { + ports.O11Y_SLACK_CAPTURE_PORT = slackAdoption.port; + try { + assertNoPortCollisions(ports); + } catch (err) { + console.error(`error: ${err.message}`); + process.exit(1); + } + } + // Only fires for an EXISTING .dev.vars (a fresh one just got DEV_ADMIN + // patched in and O11Y_SESSION_SECRET stripped, above) — a stale file + // otherwise fails closed silently. + for (const warning of checkO11yDevVarsStaleness(devVarsPath)) log("o11y", `warning: ${warning}`); + + const composeFile = path.join(RUNNER_ROOT, "containers", "o11y", "compose.yml"); + // Per-worktree default (an explicit COMPOSE_PROJECT_NAME still wins) — + // see resolveComposeProjectName's own doc comment in + // dev-lib.mjs for why a single fixed default collided across worktrees. + const composeProjectName = resolveComposeProjectName(process.env); + const composeEnv = { + ...process.env, + COMPOSE_PROJECT_NAME: composeProjectName, + O11Y_MINIO_PORT: String(ports.O11Y_MINIO_PORT), + O11Y_MINIO_CONSOLE_PORT: String(ports.O11Y_MINIO_CONSOLE_PORT), + O11Y_CLICKHOUSE_PORT: String(ports.O11Y_CLICKHOUSE_PORT), + O11Y_CLICKHOUSE_NATIVE_PORT: String(ports.O11Y_CLICKHOUSE_NATIVE_PORT), + AE_SQL_TOKEN: "local-dev-token", + }; + + // execFileSync wrapper for `resetO11yLocalState`'s injectable — runs + // from RUNNER_ROOT with the compose stack's own env, stdio inherited. + const runDocker = (cmd, args, opts = {}) => execFileSync(cmd, args, { cwd: RUNNER_ROOT, stdio: "inherit", ...opts }); + + if (fresh) { + resetO11yLocalState({ + o11yDir, + composeFile, + composeEnv, + execFileSyncImpl: runDocker, + log: (line) => log("dev", line), + }); + } else { + // Cheap (local file read) unless there's actually something to warn + // about — see detectO11yStateDivergence's own doc comment for why + // MinIO (not ClickHouse) is the volume this checks. + const divergence = await detectO11yStateDivergence({ + composeProjectName, + o11yDir, + execFileSyncImpl: (cmd, args) => execFileSync(cmd, args), + }); + if (divergence.divergent) log("dev", formatO11yDivergenceWarning(divergence.committedCount)); + } + log("dev", o11yDevDataModeLine(fresh)); + + log("compose", `starting minio + clickhouse (project ${composeProjectName})`); + // `minio-init` is not used — quay.io/minio/mc is not pullable. + // compose.yml's `minio` creates its bucket via MINIO_DEFAULT_BUCKETS + // before its healthcheck goes green, so `--wait` is sufficient. + // + // If `up` itself throws, tears the SAME project back down (no `-v`, + // data kept) before rethrowing — see `bringUpO11yCompose`'s own doc + // comment in dev-lib.mjs for why this had to be pulled out of `main()`. + bringUpO11yCompose({ + composeFile, + composeEnv, + execFileSyncImpl: (cmd, args, opts = {}) => execFileSync(cmd, args, { cwd: RUNNER_ROOT, stdio: "inherit", ...opts }), + log: (line) => log("compose", line), + }); + teardownSteps.push(() => { + log("compose", "tearing down minio + clickhouse (data kept — named volumes; use --fresh next run to wipe)"); + try { + // Never `-v` here: Ctrl-C is the KEEP path. --fresh's own wipe + // already ran, if at all, before this compose stack even started. + execFileSync("docker", composeDownArgs(composeFile), { + cwd: RUNNER_ROOT, + env: composeEnv, + stdio: "inherit", + }); + } catch (err) { + log("compose", `teardown failed: ${err.message}`); + } + }); + } + + // ---- spawn the long-running processes ----------------------------------- + const sessionSecret = tier === "full" ? ephemeralSecret() : undefined; + const plan = buildPlan(tier, ports, { sessionSecret }); + const wranglerRegistryPath = process.env.WRANGLER_REGISTRY_PATH; + const children = []; + // The first wrangler build-error line seen from each `wrangler dev` + // child, fed to that worker's readiness wait so a build failure is + // reported immediately instead of after the full timeout. + const buildErrorLines = new Map(); + + function killAll(signal) { + for (const child of children) { + if (child.exited) continue; + try { + // `detached: true` puts each child in its own process group — + // signal the whole group so wrangler's own child processes are reached too. + process.kill(-child.pid, signal); + } catch { + // already gone + } + } + } + + // `cleanup()` (kill children + teardown) is split from `shutdown()` + // (cleanup, then exit 0) so a STARTUP failure can also run cleanup + // without lying about the exit code. Without this split, a readiness + // timeout would throw straight past every teardown step, leaving live + // children and any compose stack running. + let shuttingDown = false; + async function cleanup(reason) { + if (shuttingDown) return; + shuttingDown = true; + log("dev", `${reason} — shutting down`); + killAll("SIGINT"); + const deadline = Date.now() + 8000; + // `child.killed` only reflects whether `.kill()` was called, not + // whether the process exited, and we signal the process GROUP, which + // never touches that flag. Track real exits via `child.exited` instead. + while (children.some((c) => !c.exited) && Date.now() < deadline) { + await new Promise((r) => setTimeout(r, 200)); + } + if (children.some((c) => !c.exited)) { + log("dev", "escalating to SIGKILL for any process still up after 8s"); + killAll("SIGKILL"); + } + for (const step of teardownSteps) { + try { + await step(); + } catch (err) { + log("dev", `teardown step failed: ${err.message}`); + } + } + } + async function shutdown(signal) { + if (shuttingDown) return; + await cleanup(`${signal} received`); + process.exit(0); + } + // SIGHUP: every child is `detached: true`, so closing the terminal + // sends SIGHUP to dev.mjs but not the children — handle it or they'd leak. + for (const sig of SHUTDOWN_SIGNALS) { + process.on(sig, () => shutdown(sig)); + } + + for (const proc of plan) { + const cwd = path.join(RUNNER_ROOT, proc.cwd); + const env = { ...process.env, ...proc.env }; + if (wranglerRegistryPath && (proc.name === "api" || proc.name === "o11y")) { + env.WRANGLER_REGISTRY_PATH = wranglerRegistryPath; + } + // The real args (below, `spawn`) still carry the ephemeral + // O11Y_SESSION_SECRET value in full — this only keeps it out of + // dev.mjs's own printed log line. + log(proc.name, `spawning: ${proc.bin} ${redactArgsForLog(proc.args).join(" ")}`); + const child = spawn(proc.bin, proc.args, { + cwd, + env, + stdio: ["ignore", "pipe", "pipe"], + detached: process.platform !== "win32", + }); + // Only the two `wrangler dev` children ever get a `waitForServer` call + // below — no need to scan vite's or the Slack capture server's own + // output for a marker nothing ever reads. + const onLine = + proc.name === "api" || proc.name === "o11y" + ? (line) => { + if (buildErrorLines.has(proc.name)) return; // first one wins + const errLine = wranglerBuildErrorLine(line); + if (errLine) buildErrorLines.set(proc.name, errLine); + } + : undefined; + pipeLines(child.stdout, proc.name, console.log, onLine); + pipeLines(child.stderr, proc.name, console.log, onLine); + child.exited = false; + child.on("exit", (code, signal) => { + child.exited = true; + if (!shuttingDown) { + log(proc.name, `exited unexpectedly (code=${code} signal=${signal}) — tearing everything else down`); + shutdown("SIGTERM"); + } + }); + children.push(child); + } + + // ---- readiness ------------------------------------------------------ + // Wrapped so a readiness TIMEOUT also runs `cleanup()` before reaching + // `main().catch` — see `cleanup`/`shutdown`'s own doc comment above. + try { + if (tier === "full") { + log("dev", `waiting for o11y on http://localhost:${ports.O11Y_DEV_PORT} ...`); + await waitForServer(`http://localhost:${ports.O11Y_DEV_PORT}`, 120_000, "o11y worker", { + getEarlyFailure: () => buildErrorLines.get("o11y"), + }); + log("dev", "o11y is up"); + } + if (tier === "2" || tier === "full") { + log("dev", `waiting for api on http://localhost:${ports.API_DEV_PORT} ...`); + await waitForServer(`http://localhost:${ports.API_DEV_PORT}`, 120_000, "api worker", { + getEarlyFailure: () => buildErrorLines.get("api"), + }); + log("dev", "api is up"); + } + } catch (err) { + await cleanup(`startup failed: ${err.message}`); + throw err; + } + + if (tier === "full") { + const replayCmd = `node scripts/o11y-replay-fixtures.mjs --base http://localhost:${ports.O11Y_DEV_PORT}`; + if (replay) { + log("dev", `--replay: running fixture replay now (${replayCmd})`); + try { + execFileSync("node", ["scripts/o11y-replay-fixtures.mjs", "--base", `http://localhost:${ports.O11Y_DEV_PORT}`], { + cwd: RUNNER_ROOT, + stdio: "inherit", + }); + } catch (err) { + log("dev", `fixture replay exited non-zero: ${err.message}`); + } + } else { + log("dev", `once you want fixture data, run: ${replayCmd}`); + } + log( + "dev", + `Grafana (local DEV_ADMIN bypass): http://localhost:${ports.O11Y_DEV_PORT}/grafana/ — Slack alerts land at ` + + `http://localhost:${ports.O11Y_SLACK_CAPTURE_PORT}/_captured`, + ); + log("dev", `to trigger the */10 alert cron by hand: curl "http://localhost:${ports.O11Y_DEV_PORT}/cdn-cgi/local/scheduled"`); + } + + log("dev", `authoring app: http://localhost:${ports.AUTHORING_DEV_PORT}`); + log("dev", "ready. Press Ctrl-C to stop everything."); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/runner/scripts/o11y-dev.mjs b/runner/scripts/o11y-dev.mjs new file mode 100644 index 0000000000..7648bd4e1c --- /dev/null +++ b/runner/scripts/o11y-dev.mjs @@ -0,0 +1,154 @@ +#!/usr/bin/env node +// `pnpm o11y:dev` (ADR-0041 §I) — the standalone o11y-only entry point, +// kept separate from `pnpm dev:full` for someone who only wants the o11y +// worker running (e.g. working a pure o11y bug, without the API worker, +// Docker compose, or the Slack capture server). Shares its bootstrap and +// port-resolution logic with `dev.mjs`/`dev-lib.mjs` (this task's dev-stack +// work) instead of duplicating it. +// +// This does NOT also start `containers/o11y/compose.yml` — `wrangler dev` +// manages its OWN container instance via the SAME Dockerfile, and running +// both would fight over the same image/ports for no benefit. `compose.yml` +// remains the right tool for a standalone Loki+Grafana+MinIO+ClickHouse +// stack; use `pnpm dev:full` to get both the o11y worker AND compose's +// minio/clickhouse wired together correctly (RUNNER_EVENTS_CLICKHOUSE_URL, +// the local Slack capture server, etc.). +// +// Usage: `pnpm o11y:dev` from `runner/`, or `node scripts/o11y-dev.mjs`. +// Env overrides: O11Y_DEV_PORT (default 4200), O11Y_DEV_INSPECTOR_PORT +// (default 4201) — same names/defaults `dev.mjs --tier=full` reads. + +import { spawn } from "node:child_process"; +import path from "node:path"; +import { + RUNNER_ROOT, + resolvePorts, + bootstrapDevVars, + o11yDevVarsPatch, + O11Y_DEVVARS_STRIP_KEYS, + readDevVarsLine, + ephemeralSecret, + o11yLocalPublicOrigin, + PORT_DEFAULTS, + resetO11yLocalState, + o11yDevDataModeLine, +} from "./dev-lib.mjs"; + +const o11yDir = path.join(RUNNER_ROOT, "workers", "o11y"); + +// dev-persist task: shares dev-lib.mjs's `resetO11yLocalState` with +// `dev.mjs --tier=full --fresh` (see this file's own module doc comment on +// why the two commands share bootstrap/port logic rather than duplicating +// it), but this standalone command never runs `docker compose` itself — so +// its own `--fresh` wipes ONLY workers/o11y/.wrangler/state, never any +// compose volume. If you've ALSO been running `pnpm dev:full`'s compose +// stack (minio/clickhouse, now persisted by default — see +// containers/o11y/compose.yml), wiping just the worker state here can +// create the exact divergence `dev.mjs`'s own `--fresh` exists to prevent +// (the ledger says a key was drained; the data it points at is still +// sitting in that other stack's MinIO volume, or vice versa). Run +// `docker compose -f containers/o11y/compose.yml down -v` yourself first if +// you want both wiped together. +const fresh = process.argv.slice(2).includes("--fresh"); +if (fresh) { + resetO11yLocalState({ o11yDir, log: (line) => console.log(`[o11y:dev] ${line}`) }); +} +console.log(`[o11y:dev] ${o11yDevDataModeLine(fresh)}`); + +let ports; +try { + ports = resolvePorts("o11y-only", process.env); +} catch (err) { + console.error(`[o11y:dev] ${err.message}`); + process.exit(1); +} + +// o11yDevVarsPatch also wants O11Y_SLACK_CAPTURE_PORT, for the +// SLACK_WEBHOOK_URL default it bakes in on a fresh bootstrap. This +// standalone command doesn't start that server, so fall back to the +// documented default rather than requiring an unrelated port override. +const patchPorts = { + ...ports, + O11Y_SLACK_CAPTURE_PORT: Number(process.env.O11Y_SLACK_CAPTURE_PORT) || PORT_DEFAULTS.O11Y_SLACK_CAPTURE_PORT, +}; + +const devVarsPath = path.join(o11yDir, ".dev.vars"); +const examplePath = path.join(o11yDir, ".dev.vars.example"); + +let bootstrap; +try { + bootstrap = bootstrapDevVars({ + examplePath, + devVarsPath, + patch: o11yDevVarsPatch(patchPorts), + stripKeys: O11Y_DEVVARS_STRIP_KEYS, + }); +} catch (err) { + console.error(`[o11y:dev] ${err.message}`); + process.exit(1); +} +if (bootstrap.created) { + console.log(`[o11y:dev] created ${devVarsPath} from .dev.vars.example — edit it if you need real secret values`); + if (bootstrap.patched.length) console.log(`[o11y:dev] filled in local-dev defaults for: ${bootstrap.patched.join(", ")}`); +} + +// O11Y_ENV=local and DEV_ADMIN must both be set for the local session bypass +// (K1: env.ts, gates/session.ts#verifySession — replaces the old Access +// gate) and the local jurisdiction-skip paths (inbox/accessor.ts, box.ts) to +// engage. Fail loudly rather than silently running against an unusable +// config. +const envLine = readDevVarsLine(devVarsPath, "O11Y_ENV"); +if (envLine !== "local") { + console.error(`[o11y:dev] ${devVarsPath} must set O11Y_ENV=local — refusing to start against a non-local config`); + process.exit(1); +} + +console.log(`[o11y:dev] starting wrangler dev on port ${ports.O11Y_DEV_PORT} (inspector ${ports.O11Y_DEV_INSPECTOR_PORT})`); +console.log( + `[o11y:dev] once ready, replay the fixtures in another shell: node scripts/o11y-replay-fixtures.mjs --base http://localhost:${ports.O11Y_DEV_PORT}`, +); +console.log( + "[o11y:dev] a local Slack-webhook capture server is NOT started by this command — use `pnpm dev:full` for that, or point SLACK_WEBHOOK_URL in .dev.vars at your own `node scripts/o11y-slack-capture.mjs --port `.", +); +console.log( + "[o11y:dev] the box's local container image build + first `wake()` can take anywhere from a few seconds to well over a minute. This is a real, environment-dependent Container-platform characteristic, not a hang.", +); +console.log( + '[o11y:dev] to trigger the */10 cron by hand (wrangler no longer wires --test-scheduled/__scheduled locally): curl "http://localhost:' + + ports.O11Y_DEV_PORT + + '/cdn-cgi/local/scheduled" (wrangler 4.136.3)', +); + +// Spawn `node_modules/.bin/wrangler` directly, not via `npx` — `npx` is a +// wrapper process, and killing it does not reliably kill the real +// `wrangler`/`workerd` grandchild it spawns (the exact orphan-container risk +// this task's research flagged; `detached: true` + signalling the whole +// process group below is what actually reaches workerd's own children too). +const sessionSecret = ephemeralSecret(); +const child = spawn( + path.join("node_modules", ".bin", "wrangler"), + [ + "dev", + "--port", + String(ports.O11Y_DEV_PORT), + "--inspector-port", + String(ports.O11Y_DEV_INSPECTOR_PORT), + "--var", + `O11Y_SESSION_SECRET:${sessionSecret}`, + "--var", + `O11Y_LOCAL_PUBLIC_ORIGIN:${o11yLocalPublicOrigin(ports)}`, + ], + { cwd: o11yDir, stdio: "inherit", detached: process.platform !== "win32" }, +); + +child.on("exit", (code) => process.exit(code ?? 0)); + +for (const sig of ["SIGINT", "SIGTERM"]) { + process.on(sig, () => { + try { + process.kill(-child.pid, sig); + } catch { + child.kill(sig); + } + }); +} diff --git a/runner/scripts/o11y-replay-fixtures.mjs b/runner/scripts/o11y-replay-fixtures.mjs new file mode 100644 index 0000000000..bce26f1d84 --- /dev/null +++ b/runner/scripts/o11y-replay-fixtures.mjs @@ -0,0 +1,138 @@ +#!/usr/bin/env node +// Replays every `pipeline/fixtures/{otlp,faro}/**` fixture against a +// running o11y worker: `( cd workers/o11y && npx wrangler dev )`, then +// `node scripts/o11y-replay-fixtures.mjs --base http://localhost:4300`. +// +// Faro fixtures get a fresh Origin + timestamp; OTLP fixtures get the +// `x-o11y-secret` header, read from env or `workers/o11y/.dev.vars`. +// Exits non-zero if any fixture does not answer 2xx. + +import { readFileSync, readdirSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { createHmac } from "node:crypto"; +import { resolveReplaySecret } from "./dev-lib.mjs"; + +const args = process.argv.slice(2); +const baseIndex = args.indexOf("--base"); +const base = baseIndex !== -1 ? args[baseIndex + 1] : "http://localhost:4300"; + +const ROOT = fileURLToPath(new URL("../pipeline/fixtures/", import.meta.url)); +const O11Y_DEV_VARS_PATH = fileURLToPath(new URL("../workers/o11y/.dev.vars", import.meta.url)); +const EXPORT_SECRET = resolveReplaySecret(process.env.O11Y_EXPORT_SECRET, O11Y_DEV_VARS_PATH, "O11Y_EXPORT_SECRET"); +const SENTRY_SECRET = resolveReplaySecret(process.env.SENTRY_HOOK_SECRET, O11Y_DEV_VARS_PATH, "SENTRY_HOOK_SECRET"); + +if (!EXPORT_SECRET) { + console.warn( + "O11Y_EXPORT_SECRET not set (checked env and workers/o11y/.dev.vars) — v1/logs and deploy fixtures will 401.", + ); +} +if (!SENTRY_SECRET) { + console.warn( + "SENTRY_HOOK_SECRET not set (checked env and workers/o11y/.dev.vars) — the Sentry hook fixture will 401.", + ); +} + +let failures = 0; + +async function post(path, body, headers = {}) { + const res = await fetch(`${base}${path}`, { method: "POST", headers, body }); + const ok = res.status >= 200 && res.status < 300; + console.log(`${ok ? "OK " : "FAIL"} ${res.status} POST ${path} (${body.length} bytes)`); + if (!ok) { + failures++; + console.log(` ${(await res.text()).slice(0, 300)}`); + } + return res; +} + +/** Faro fixtures embed a JSON `timestamp` field per item — stamp it fresh + * (see the file header) and inject one item with a timestamp far outside + * the ±5 min clamp window, so the clamp-to-`received_at` fallback is + * actually exercised by at least one item per replay, not only the + * in-window happy path. */ +function freshenFaro(text) { + const body = JSON.parse(text); + let stampedOne = false; + for (const key of ["exceptions", "logs", "measurements", "events"]) { + for (const item of body[key] ?? []) { + if (!stampedOne) { + item.timestamp = new Date(Date.now() - 60 * 60 * 1000).toISOString(); // 1h out of window + stampedOne = true; + } else { + item.timestamp = new Date().toISOString(); + } + } + } + return JSON.stringify(body); +} + +async function replayFaro() { + for (const name of readdirSync(`${ROOT}faro`)) { + const text = readFileSync(`${ROOT}faro/${name}`, "utf8"); + await post("/telemetry/collect", freshenFaro(text), { + Origin: base.startsWith("http://localhost") ? "http://localhost:4300" : "https://demos.handsontable.com", + "content-type": "application/json", + }); + } +} + +async function replayOtlpJson() { + for (const name of readdirSync(`${ROOT}otlp/json`)) { + const text = readFileSync(`${ROOT}otlp/json/${name}`, "utf8"); + await post("/telemetry/v1/logs", text, { + "x-o11y-secret": EXPORT_SECRET, + "content-type": "application/json", + }); + } +} + +async function replayOtlpProtobuf() { + for (const name of readdirSync(`${ROOT}otlp/protobuf`)) { + const bytes = readFileSync(`${ROOT}otlp/protobuf/${name}`); + await post("/telemetry/v1/logs", bytes, { + "x-o11y-secret": EXPORT_SECRET, + "content-type": "application/x-protobuf", + }); + } +} + +async function replayDeploy() { + const text = readFileSync(`${ROOT}otlp/deploy-event.json`, "utf8"); + await post("/telemetry/deploy", text, { + "x-o11y-secret": EXPORT_SECRET, + "content-type": "application/json", + }); +} + +async function replaySentry() { + const text = readFileSync(`${ROOT}otlp/sentry-issue.json`, "utf8"); + const sig = createHmac("sha256", SENTRY_SECRET).update(text).digest("hex"); + await post("/telemetry/hooks/sentry", text, { + "sentry-hook-signature": sig, + "content-type": "application/json", + }); +} + +async function replayDuplicate() { + // Exit criterion 4: the same body twice, seconds apart. Both must + // answer 2xx; dedupe itself is asserted by the pipeline tests. + const text = readFileSync(`${ROOT}otlp/json/zero-timestamp.json`, "utf8"); + const headers = { "x-o11y-secret": EXPORT_SECRET, "content-type": "application/json" }; + await post("/telemetry/v1/logs", text, headers); + await new Promise((r) => setTimeout(r, 2000)); + await post("/telemetry/v1/logs", text, headers); +} + +console.log(`Replaying fixtures against ${base} …`); +await replayFaro(); +await replayOtlpJson(); +await replayOtlpProtobuf(); +await replayDeploy(); +await replaySentry(); +await replayDuplicate(); + +if (failures > 0) { + console.error(`\n${failures} fixture(s) failed.`); + process.exit(1); +} +console.log("\nAll fixtures replayed successfully."); diff --git a/runner/scripts/o11y-slack-capture.mjs b/runner/scripts/o11y-slack-capture.mjs new file mode 100644 index 0000000000..30d375a794 --- /dev/null +++ b/runner/scripts/o11y-slack-capture.mjs @@ -0,0 +1,81 @@ +#!/usr/bin/env node +// A tiny local-only stand-in for a Slack incoming webhook (ADR-0041 §F.3). +// `pnpm dev:full` points the o11y worker's `SLACK_WEBHOOK_URL` at this +// server instead of a real Slack webhook, so a fired alert is visible +// locally without touching a real Slack channel. +// +// Usage: node scripts/o11y-slack-capture.mjs --port + +import { createServer } from "node:http"; + +const MAX_CAPTURED = 50; + +/** @param {number} port + * @returns {{ server: import("node:http").Server, captured: {at: string, path: string, body: unknown}[], close: () => Promise }} */ +export function createSlackCaptureServer(port) { + const captured = []; + const server = createServer((req, res) => { + if (req.method === "GET" && req.url === "/_captured") { + res.writeHead(200, { "Content-Type": "application/json" }); + res.end(JSON.stringify(captured)); + return; + } + if (req.method !== "POST") { + res.writeHead(405, { "Content-Type": "text/plain" }); + res.end("method not allowed"); + return; + } + let raw = ""; + req.on("data", (chunk) => { + raw += chunk; + }); + req.on("end", () => { + let body; + try { + body = JSON.parse(raw); + } catch { + body = raw; + } + const entry = { at: new Date().toISOString(), path: req.url ?? "/", body }; + captured.push(entry); + while (captured.length > MAX_CAPTURED) captured.shift(); + const text = typeof body === "object" && body && "text" in body ? String(body.text) : raw; + console.log(`[slack] captured alert post: ${text}`); + res.writeHead(200, { "Content-Type": "text/plain" }); + res.end("ok"); + }); + }); + // Loopback only: without an explicit host, `listen(port)` binds every + // interface, exposing unauthenticated alert text to the LAN. + server.listen(port, "127.0.0.1"); + return { + server, + captured, + close: () => + new Promise((resolve) => { + server.close(() => resolve()); + }), + }; +} + +function main() { + const args = process.argv.slice(2); + const portIndex = args.indexOf("--port"); + const port = portIndex !== -1 ? Number(args[portIndex + 1]) : 4210; + if (!Number.isInteger(port) || port <= 0) { + console.error(`[slack] invalid --port: ${args[portIndex + 1]}`); + process.exit(1); + } + const { close } = createSlackCaptureServer(port); + console.log(`[slack] local Slack capture server listening on http://localhost:${port} (POST any path; GET /_captured to inspect)`); + for (const sig of ["SIGINT", "SIGTERM"]) { + process.on(sig, async () => { + await close(); + process.exit(0); + }); + } +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main(); +} diff --git a/runner/workers/api/.dev.vars.example b/runner/workers/api/.dev.vars.example new file mode 100644 index 0000000000..fcfe67e84a --- /dev/null +++ b/runner/workers/api/.dev.vars.example @@ -0,0 +1,25 @@ +# Copy to .dev.vars (gitignored) for `wrangler dev`. Never commit real values. +# `runner/scripts/dev.mjs` (pnpm dev:live / dev:full) bootstraps this +# automatically on first run, the same way workers/o11y/.dev.vars.example is +# bootstrapped — see docs/run-and-deploy.md's "Run locally" section. +# +# Both values below are non-secret dev-only bypasses (auth.ts, index.ts) and +# already match this script's own documented default ports (API_DEV_PORT +# 8787). If you edit PREVIEW_HOST here to a different port WITHOUT also +# setting the API_DEV_PORT env var, dev.mjs ADOPTS this file's port for the +# run (prints an info line) instead of starting pointed at the wrong one — +# wrangler's `.dev.vars` always wins over this script's own `--var`, so this +# file's value was always going to be the one actually reached. Only an +# EXPLICIT API_DEV_PORT env var that disagrees with this file gets a warning +# instead (your explicit choice is never silently overridden). + +# Local login bypass — an internal @handsontable.com address, checked in +# auth.ts AFTER the persistent-API-token check. See AGENTS.md. +DEV_AUTH_EMAIL="dev@handsontable.com" + +# Overrides wrangler.jsonc's real default (demos.handsontable.com) so Tier-2 +# container preview URLs come out as *.localhost:8787 — browsers treat that +# as 127.0.0.1 (RFC 6761) and reach your local `wrangler dev`. Must stay a +# non-production host: it also selects the local (vs. production) telemetry +# sink (telemetry/resource.ts). +PREVIEW_HOST="localhost:8787" diff --git a/runner/workers/api/migrations/0008_example_daily.sql b/runner/workers/api/migrations/0008_example_daily.sql new file mode 100644 index 0000000000..8ab14dec93 --- /dev/null +++ b/runner/workers/api/migrations/0008_example_daily.sql @@ -0,0 +1,37 @@ +-- ADR-0042 (example analytics) — the permanent daily record. Analytics +-- Engine keeps ~3 months; this table is what a longer-range "which guides +-- get opened" view reads from, recomputed nightly by +-- `reconcile.ts#rollupExampleDaily` for the previous full UTC day. +-- +-- `area` is a function of `ref` (a docs guide's breadcrumb never changes +-- which area it is in) but is stored anyway, not joined at read time — this +-- table has no other source of the docs-example taxonomy to join against +-- (the taxonomy lives in the docs-examples JSON, not in D1), and ADR-0042 §5 +-- names it as a column explicitly. +-- +-- Primary key is (day, kind, ref, framework, ht_major), NOT including area +-- (ADR-0042 §5: "area is a function of ref") — re-running the rollup for a +-- day replaces that day's rows (`rollupExampleDaily` does a real DELETE + +-- INSERT, not a bare INSERT OR REPLACE: a group with zero events on a re-run +-- must disappear, not linger from the previous run). +-- +-- No `downloaded` column — ADR-0042 §5 names five counters (opens, engaged, +-- forked, saved, shared) for six `example.*` metrics; `example.downloaded` +-- has no column here by the ADR's own spec (T12 Outcome, flagged as a +-- concern: the raw AE point still exists, just not rolled into this table). +CREATE TABLE IF NOT EXISTS example_daily ( + day TEXT NOT NULL, -- YYYY-MM-DD, UTC + kind TEXT NOT NULL, -- 'docs' | 'starter' | 'saved' | 'import' | 'payload' + ref TEXT NOT NULL, -- guide path or starter id + area TEXT NOT NULL, -- first breadcrumb element (docs only; '' otherwise) + framework TEXT NOT NULL, + ht_major TEXT NOT NULL, + opens INTEGER NOT NULL DEFAULT 0, + engaged INTEGER NOT NULL DEFAULT 0, + forked INTEGER NOT NULL DEFAULT 0, + saved INTEGER NOT NULL DEFAULT 0, + shared INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (day, kind, ref, framework, ht_major) +); + +CREATE INDEX IF NOT EXISTS idx_example_daily_day ON example_daily (day); diff --git a/runner/workers/api/migrations/0009_example_daily_downloaded.sql b/runner/workers/api/migrations/0009_example_daily_downloaded.sql new file mode 100644 index 0000000000..e2d7e24745 --- /dev/null +++ b/runner/workers/api/migrations/0009_example_daily_downloaded.sql @@ -0,0 +1,17 @@ +-- ADR-0042 (example analytics), follow-up to 0008_example_daily.sql. +-- +-- ADR-0042 §2 names six `example.*` metrics (open, engaged, forked, saved, +-- shared, downloaded); 0008's `example_daily` table only ever stored five +-- counters — `example.downloaded` existed as an Analytics Engine point but +-- was never rolled into D1 (0008's own header comment, and ADR-0042 §5, +-- flagged this explicitly as a known gap). This migration closes it. +-- +-- Additive and safe on production data: a bare `ADD COLUMN ... DEFAULT 0` +-- backfills every existing row with `downloaded = 0` (correct — those rows +-- were computed before this column existed, so their true downloaded count +-- for that day is unknown, and 0 is the least misleading value: it never +-- overcounts, and a nightly re-roll from Analytics Engine's own retention +-- window will fill in real numbers for any day still inside it). 0008 itself +-- is left untouched, as ever — a migration already shipped is never edited, +-- only superseded by a new one. +ALTER TABLE example_daily ADD COLUMN downloaded INTEGER NOT NULL DEFAULT 0; diff --git a/runner/workers/api/package.json b/runner/workers/api/package.json index e446149d00..0f5f0a73ac 100644 --- a/runner/workers/api/package.json +++ b/runner/workers/api/package.json @@ -5,7 +5,7 @@ "type": "module", "scripts": { "dev": "wrangler dev", - "deploy": "wrangler deploy --routes '*.demos.handsontable.com/*' --routes 'demos.handsontable.com/api/*' --routes 'demos.handsontable.com/d/*' --routes 'demos.handsontable.com/embed/*' --var SENTRY_ENVIRONMENT:api-production", + "deploy": "wrangler deploy --routes '*.demos.handsontable.com/*' --routes 'demos.handsontable.com/api/*' --routes 'demos.handsontable.com/d/*' --routes 'demos.handsontable.com/embed/*' --var SENTRY_ENVIRONMENT:api-production --var SERVICE_VERSION:$GITHUB_SHA", "typecheck": "tsc --noEmit" }, "dependencies": { diff --git a/runner/workers/api/src/admin.ts b/runner/workers/api/src/admin.ts index ea34915d56..4de752fd26 100644 --- a/runner/workers/api/src/admin.ts +++ b/runner/workers/api/src/admin.ts @@ -12,6 +12,7 @@ import type { Env } from "./env.js"; import { computeBudgetState, + computeO11ySpend, containerUsdPerSecond, KV_METER_PREFIX, SESSION_INSTANCE_TYPE, @@ -101,7 +102,7 @@ const MAX_LEGACY_READS = 200; * hundreds of concurrent subrequests is its own failure mode. */ const LEGACY_READ_BATCH = 20; -interface MeterRecord { +export interface MeterRecord { sessionId: string; startedAt: number; meteredThrough: number; @@ -117,8 +118,12 @@ interface MeterRecord { * ids begin with the framework slug, that cap meant the table could only ever * show `angular` (the alphabetically first Tier-2 slug). That artifact is the * whole reason DEV-2567 read as an Angular-specific leak. + * + * Exported so `telemetry/cron.ts#countLiveSessionMeters` can share this + * exact scan — and, with it, `classifyMeter`'s awake/slept split — instead of a + * second, narrower KV walk that could only ever count keys, never sessions. */ -async function readMeters(env: Env): Promise<{ meters: MeterRecord[]; truncated: boolean }> { +export async function readMeters(env: Env): Promise<{ meters: MeterRecord[]; truncated: boolean }> { const meters: MeterRecord[] = []; const legacyKeys: string[] = []; let cursor: string | undefined; @@ -222,7 +227,7 @@ export async function adminUsage(env: Env, days: number) { const since = dayAgo(days); const monthPrefix = new Date().toISOString().slice(0, 7); - const [ledger, usage, demoTotals, demosByFramework, topDemos, budget, sessions, audience] = await Promise.all([ + const [ledger, usage, demoTotals, demosByFramework, topDemos, budget, o11ySpend, sessions, audience] = await Promise.all([ env.DB.prepare( `SELECT day, sku, source, units, usd FROM cost_ledger WHERE day >= ?1 ORDER BY day DESC, sku`, ).bind(since).all(), @@ -257,6 +262,12 @@ export async function adminUsage(env: Env, days: number) { // not a five-minute-old copy of it. computeBudgetState(env), + // ADR-0041 §G: the observability-only slice of the same ledger, + // plus its own (smaller) cap — see `budget.ts#computeO11ySpend`'s own + // doc comment for why this is additive to `computeBudgetState` above, + // not a replacement. + computeO11ySpend(env), + // The default view (awake only, first page). Paging and the "show the 24h // tail" toggle go to `GET /api/admin/sessions` instead, so neither re-runs // the D1 aggregates above. @@ -285,6 +296,16 @@ export async function adminUsage(env: Env, days: number) { reconciled: budget.reconciled, enforced: budget.enforced, }, + // ADR-0041 §G: "`/admin` shows app, observability and total." + // `total` equals `budget.spendUsd` above (every sku, o11y included — + // §G: "product tiers keep acting on the total"); `app` is the + // remainder, never a second D1 read. + o11y: { + spendUsd: o11ySpend.spendUsd, + capUsd: o11ySpend.capUsd, + appSpendUsd: Math.max(0, budget.spendUsd - o11ySpend.spendUsd), + totalSpendUsd: budget.spendUsd, + }, // The editable thresholds, so the panel's form starts from what is // actually in force rather than from a copy of the defaults. settings: budget.settings, diff --git a/runner/workers/api/src/analytics.ts b/runner/workers/api/src/analytics.ts index 1d5a0962c4..3958816210 100644 --- a/runner/workers/api/src/analytics.ts +++ b/runner/workers/api/src/analytics.ts @@ -21,6 +21,7 @@ // everything else so the numbers mean something. import type { Env } from "./env.js"; +import { BOT_RE, isBot, deviceOf, browserOf, osOf } from "@handsontable/demo-runtime/telemetry"; const SALT_TTL_SECONDS = 60 * 60 * 48; const FLUSH_AT_EVENTS = 100; @@ -40,35 +41,10 @@ const utcDay = (): string => new Date().toISOString().slice(0, 10); // ---- Bucketing --------------------------------------------------------------- -const BOT_RE = /bot|crawler|spider|crawling|slurp|bingpreview|headlesschrome|lighthouse|curl\/|wget\/|python-requests|node-fetch|axios\/|monitoring|uptime|pingdom|semrush|ahrefs|facebookexternalhit|whatsapp|telegrambot|preview/i; - -export const isBot = (userAgent: string): boolean => BOT_RE.test(userAgent); - -/** Coarse device class. Deliberately three buckets — anything finer starts to - * look like a fingerprint. */ -function deviceOf(ua: string): string { - if (/ipad|tablet|playbook|silk/i.test(ua)) return "tablet"; - if (/mobi|iphone|ipod|android.*mobile|windows phone/i.test(ua)) return "mobile"; - return "desktop"; -} - -function browserOf(ua: string): string { - if (/edg\//i.test(ua)) return "edge"; - if (/opr\/|opera/i.test(ua)) return "opera"; - if (/chrome|crios|chromium/i.test(ua)) return "chrome"; - if (/firefox|fxios/i.test(ua)) return "firefox"; - if (/safari/i.test(ua)) return "safari"; - return "other"; -} - -function osOf(ua: string): string { - if (/windows/i.test(ua)) return "windows"; - if (/iphone|ipad|ipod|ios/i.test(ua)) return "ios"; - if (/mac os x|macintosh/i.test(ua)) return "macos"; - if (/android/i.test(ua)) return "android"; - if (/linux|x11|cros/i.test(ua)) return "linux"; - return "other"; -} +// `BOT_RE` and the UA classifiers moved to the telemetry contract module, +// which the o11y ingest gates (ADR §B.5) share the same definitions with. Byte- +// identical regexes, re-exported here so no other importer's path changes. +export { BOT_RE, isBot, deviceOf, browserOf, osOf }; /** Referring *hostname* only. A full referrer URL can carry a search query or * a private path, so the path and query never leave this function. */ diff --git a/runner/workers/api/src/budget.ts b/runner/workers/api/src/budget.ts index b8e157eb15..a82ab82a0c 100644 --- a/runner/workers/api/src/budget.ts +++ b/runner/workers/api/src/budget.ts @@ -192,17 +192,42 @@ function upsertEstimate(env: Env, day: string, sku: string, units: number, usd: /** * Meter a closed container awake window. Additive per (day, sku) so concurrent * writers never clobber each other. + * + * `sku` (ADR-0041 §G) defaults to `"container"` — every existing call site + * is unaffected. `o11y-usage.ts` calls this with `sku: "o11y_container"`, + * a distinct SKU so its upsert never overwrites the app's `container` rows. */ export async function recordContainerUsage( env: Env, - opts: { instanceType: InstanceType; awakeSeconds: number }, + opts: { instanceType: InstanceType; awakeSeconds: number; sku?: string }, ): Promise { if (!(opts.awakeSeconds > 0)) return; const usd = opts.awakeSeconds * containerUsdPerSecond(opts.instanceType); - await upsertEstimate(env, utcDay(), "container", opts.awakeSeconds, usd).run(); + await upsertEstimate(env, utcDay(), opts.sku ?? "container", opts.awakeSeconds, usd).run(); await invalidateBudgetState(env); } +/** + * Month-to-date observability spend (`o11y_container` + `o11y_workers` + * SKUs only), plus the ADR-0041 §G cap it is checked against. Deliberately + * separate from {@link computeBudgetState}, whose `limitUsd`/tier machinery + * keeps summing every sku unchanged (ADR §G: "Product tiers keep acting on + * the total"). + */ +export async function computeO11ySpend(env: Env): Promise<{ spendUsd: number; capUsd: number }> { + const settings = await loadSettings(env); + const monthPrefix = new Date().toISOString().slice(0, 7); + const { results } = await env.DB.prepare( + `SELECT COALESCE(MAX(CASE WHEN source = 'billing' THEN usd END), + MAX(CASE WHEN source = 'estimate' THEN usd END), 0) AS usd + FROM cost_ledger + WHERE day LIKE ?1 AND sku IN ('o11y_container', 'o11y_workers') + GROUP BY day, sku`, + ).bind(`${monthPrefix}%`).all<{ usd: number }>(); + const spendUsd = (results ?? []).reduce((sum, r) => sum + (r.usd ?? 0), 0); + return { spendUsd, capUsd: settings.o11yBudgetUsd }; +} + /** * Meter one assistant answer (DEV-2047). * @@ -244,6 +269,10 @@ export interface SessionMeter { startedAt: number; meteredThrough: number; instanceType: InstanceType; + /** The session's framework (`session.end`, contract §5). Optional: a meter + * written before this field existed round-trips with it absent. Carried + * here because teardown call sites only ever have a `sessionId`. */ + framework?: string; } const meterKey = (sessionId: string) => `${KV_METER_PREFIX}${sessionId}`; @@ -292,9 +321,10 @@ export async function startSessionMeter( env: Env, sessionId: string, instanceType: InstanceType = SESSION_INSTANCE_TYPE, + framework?: string, ): Promise { const now = Date.now(); - const meter: SessionMeter = { startedAt: now, meteredThrough: now, instanceType }; + const meter: SessionMeter = { startedAt: now, meteredThrough: now, instanceType, ...(framework ? { framework } : {}) }; await env.CACHE.put(meterKey(sessionId), JSON.stringify(meter), { expirationTtl: KV_METER_TTL_SECONDS, metadata: meterMetadata(meter), @@ -304,20 +334,25 @@ export async function startSessionMeter( /** * Book the slice of awake time since the last flush. * `final` (teardown) always books and then drops the meter. + * + * Returns the meter's `framework` (contract §5's `session.end` blob) when + * on record — `undefined` on a KV miss. `void`-safe: every existing caller + * already ignores the return value. */ export async function meterSession( env: Env, sessionId: string, opts: { final?: boolean } = {}, -): Promise { +): Promise { // Metering is telemetry, and telemetry must never be the reason a request // fails. The teardown path in particular: a throw here would skip the // `sandbox.destroy()` that follows it and leave a container billing until // its idle window lapses — the exact cost this file exists to prevent. try { - await meterSessionUnsafe(env, sessionId, opts); + return await meterSessionUnsafe(env, sessionId, opts); } catch (err) { console.warn("[budget] session metering failed:", err instanceof Error ? err.message : String(err)); + return undefined; } } @@ -325,14 +360,14 @@ async function meterSessionUnsafe( env: Env, sessionId: string, opts: { final?: boolean }, -): Promise { +): Promise { const key = meterKey(sessionId); const meter = (await env.CACHE.get(key, "json").catch(() => null)) as SessionMeter | null; - if (!meter) return; + if (!meter) return undefined; const now = Date.now(); const elapsedSeconds = Math.max(0, (now - meter.meteredThrough) / 1000); - if (!opts.final && elapsedSeconds < METER_FLUSH_SECONDS) return; + if (!opts.final && elapsedSeconds < METER_FLUSH_SECONDS) return meter.framework; const awakeSeconds = Math.min(elapsedSeconds, MAX_UNSEEN_AWAKE_SECONDS); if (opts.final) { @@ -347,6 +382,7 @@ async function meterSessionUnsafe( }).catch(() => { /* next ping re-books the same slice; capped above */ }); } await recordContainerUsage(env, { instanceType: meter.instanceType, awakeSeconds }); + return meter.framework; } // ---- Traffic accumulator ----------------------------------------------------- diff --git a/runner/workers/api/src/chat.ts b/runner/workers/api/src/chat.ts index 2a27d535a9..37fcd1ffcc 100644 --- a/runner/workers/api/src/chat.ts +++ b/runner/workers/api/src/chat.ts @@ -114,6 +114,10 @@ export interface ChatAnswer { references: string[]; /** What the answer cost, when the gateway tells us. */ usd: number; + /** `chat.answer`'s tokens_in/tokens_out (contract §5), when the gateway's + * OpenAI-compatible `usage` object reports them. 0 when it does not. */ + tokensIn: number; + tokensOut: number; } // ---- Input validation -------------------------------------------------------- @@ -482,13 +486,23 @@ export async function requestAnswer(env: Env, req: ChatRequest, pages: DocPage[] // problem, not a user one, so it must be loud in the logs and vague to // the caller. console.error(`[chat] gateway ${res.status} (request id: ${requestId})`); + // ADR-0041 §E.1 diagnostic (an upstream failure reported with tags) — + // `status`/`requestId` ride on the thrown error rather than a + // `reportDiagnostic` call here: this module is copied and imported + // standalone by `pipeline/chat-sanitise.test.mjs`/`decode-entities.test.mjs` + // (its own header comment explains why), which cannot resolve a sibling + // `./telemetry/*.js` import. `index.ts`'s `ChatUnavailableError` catch is + // where the diagnostic capture actually happens (never the gateway body, + // same rule the console.error above already follows). throw new ChatUnavailableError( res.status === 401 || res.status === 403 ? "chat is not configured" : "the assistant is unavailable", + { status: res.status, requestId }, ); } const payload = (await res.json()) as { choices?: { message?: { tool_calls?: { function?: { name?: string; arguments?: string } }[] } }[]; + usage?: { prompt_tokens?: number; completion_tokens?: number }; }; const call = payload.choices?.[0]?.message?.tool_calls?.find((c) => c.function?.name === "answer"); if (typeof call?.function?.arguments !== "string") { @@ -505,10 +519,27 @@ export async function requestAnswer(env: Env, req: ChatRequest, pages: DocPage[] // LiteLLM reports what the call cost; when it does, the ledger gets a real // number instead of an estimate (see DEV-2030). const usd = Number(res.headers.get("x-litellm-response-cost") ?? 0); - return sanitiseAnswer(parsed, req, Number.isFinite(usd) ? usd : 0); + const tokensIn = Number(payload.usage?.prompt_tokens ?? 0); + const tokensOut = Number(payload.usage?.completion_tokens ?? 0); + return sanitiseAnswer(parsed, req, Number.isFinite(usd) ? usd : 0, { + tokensIn: Number.isFinite(tokensIn) ? tokensIn : 0, + tokensOut: Number.isFinite(tokensOut) ? tokensOut : 0, + }); } -export class ChatUnavailableError extends Error {} +export class ChatUnavailableError extends Error { + /** Set only for a gateway (LiteLLM) failure — `index.ts`'s diagnostic + * capture reads these; every other throw site in this file leaves them + * undefined (a config/parse problem, not an upstream one). */ + readonly status?: number; + readonly requestId?: string; + + constructor(message: string, upstream?: { status: number; requestId: string }) { + super(message); + this.status = upstream?.status; + this.requestId = upstream?.requestId; + } +} /** * Whitelist everything on the way out. @@ -525,7 +556,12 @@ export class ChatUnavailableError extends Error {} * Escaping belongs at the sink, and the sink already does it — so ` + + +`; + + return new Response(html, { + status: 200, + // The broker JWT sits in this page's own URL fragment until the + // script strips it; `no-store`/`no-referrer` stop it leaking through + // history-adjacent caches. + headers: await staticPageHeaders(CALLBACK_SCRIPT), + }); +}; + +// ---- POST /grafana/_o11y/session ----------------------------------------- + +interface SessionBody { + token: string; + n: string; +} + +function isSessionBody(value: unknown): value is SessionBody { + const v = value as Partial | null; + return typeof v === "object" && v !== null && typeof v.token === "string" && v.token.length > 0 && + typeof v.n === "string" && v.n.length > 0; +} + +export const handleSession: RouteHandler = async (req, env) => { + if (!isSessionSecretValid(env)) return jsonResponse({ error: "not_configured" }, 500); + if (!isSameOrigin(req)) return jsonResponse({ error: "bad_origin" }, 403); + if (!contentTypeIsJson(req)) return jsonResponse({ error: "expected content-type: application/json" }, 415); + + const retryAfter = await rateLimited(req, env, "o11y-session"); + if (retryAfter !== null) { + return jsonResponse({ error: "rate_limited" }, 429, { "retry-after": String(retryAfter) }); + } + + let body: unknown; + try { + body = await req.json(); + } catch { + return jsonResponse({ error: "invalid_json" }, 400); + } + if (!isSessionBody(body)) return jsonResponse({ error: "expected { token: string, n: string }" }, 400); + + // Login-CSRF binding: the state minted at `/login` must still be + // present (a missing `o11y_login` cookie is refused too) and must name + // the SAME nonce the callback's `?n=` carried, so an attacker cannot + // complete the exchange without forging the victim's signed cookie. + const state = await verifyLoginCookie(req, env); + if (!state || state.nonce !== body.n) { + return jsonResponse({ error: "nonce_mismatch" }, 401); + } + + // One live verification against the broker, never cached: the JWT is + // discarded the moment this returns, never stored, logged, or cookied. + const identity = await resolveBrokerIdentity(env, body.token); + if (!identity) return jsonResponse({ error: "not_authorized" }, 401); + + // Capped at the broker token's own `exp`, not a flat 12h — see + // `computeSessionTtlSeconds`'s own doc comment for why, and ADR-0041 §M's + // DEV-3088 blast-radius reasoning. + const ttlSeconds = computeSessionTtlSeconds(identity.exp); + const sessionToken = await signSessionCookie(env, identity.email, ttlSeconds); + const headers = new Headers(); + headers.append("Set-Cookie", sessionSetCookieHeader(sessionToken, ttlSeconds)); + headers.append("Set-Cookie", loginClearCookieHeader()); + return jsonResponse({ next: sanitizeNext(state.next, env) }, 200, headers); +}; + +// ---- GET /grafana/_o11y/logout (a same-origin sign-out page) ------------ + +/** Fires the actual, CSRF-protected `POST /grafana/_o11y/logout` from a + * same-origin script — a bare ``/GET would be forgeable under + * `SameSite=Lax`. Grafana's own sign-out menu item is disabled. */ +const LOGOUT_SCRIPT = `(function(){ + var msg = document.getElementById("m"); + fetch("/grafana/_o11y/logout", { + method: "POST", + headers: { "content-type": "application/json" }, + body: "{}" + }).then(function () { + location.replace("/grafana/"); + }).catch(function () { + msg.textContent = "Sign-out failed. Try again."; + }); +})();`; + +export const handleLogoutPage: RouteHandler = async () => { + const html = ` + + + +Signing out — Handsontable observability + + +

Signing out…

+ + + +`; + return new Response(html, { status: 200, headers: await staticPageHeaders(LOGOUT_SCRIPT) }); +}; + +// ---- POST /grafana/_o11y/logout ------------------------------------------- + +export const handleLogout: RouteHandler = async (req) => { + if (!isSameOrigin(req)) return jsonResponse({ error: "bad_origin" }, 403); + if (!contentTypeIsJson(req)) return jsonResponse({ error: "expected content-type: application/json" }, 415); + + // Clears BOTH cookies — leaving `o11y_login` behind would be harmless on + // its own short TTL, but "logout clears state" should mean all of it. + const headers = new Headers({ Location: "/" }); + headers.append("Set-Cookie", sessionClearCookieHeader()); + headers.append("Set-Cookie", loginClearCookieHeader()); + return new Response(null, { status: 302, headers }); +}; diff --git a/runner/workers/o11y/src/grafana/proxy.ts b/runner/workers/o11y/src/grafana/proxy.ts new file mode 100644 index 0000000000..b15b3592be --- /dev/null +++ b/runner/workers/o11y/src/grafana/proxy.ts @@ -0,0 +1,159 @@ +// `/grafana/*` (ADR §B.5/§H — see gates/session.ts's own header): verify +// the Worker's own session cookie, wake (idempotent), strip the cookie and +// any client-supplied auth headers, set `x-o11y-grafana-user`, proxy to +// port 3000, renew activity only once the request actually reaches this +// far — never for a request served the waking page. + +import { isBrowserNavigation, sanitizeNext, verifySession } from "../gates/session.js"; +import { GRAFANA_PROXY_MAX_BYTES, contentLengthExceeds } from "../gates/limits.js"; +import { BodyTooLargeError } from "../normalise/read-body.js"; +import { getGrafanaBoxStub } from "../box.js"; +import { wakingPageResponse } from "./waking-page.js"; +import type { RouteHandler } from "../router.js"; + +/** Never forwarded to the container: `cookie` carries our own session + * cookies; `x-o11y-grafana-user` is set BY this Worker; a client-supplied + * value of either must never reach Grafana. `cf-container-target-port` + * is stripped so a client can never steer a proxied request at Loki. */ +const STRIPPED_HEADERS = ["cookie", "x-o11y-grafana-user", "cf-container-target-port"]; + +/** A dashboard's own auto-refresh must not call `wake("visit")` like a + * real page load — that would re-start a stopped box on every refresh, at + * the ADR §A cost model's full awake-hour rate. Fetch Metadata tells them + * apart: real navigation sends `sec-fetch-dest: document`, background + * fetch sends `empty`. Fails OPEN when absent, since the real waste + * vector always carries the header. Only gates STARTING a stopped box. */ +function isTopLevelNavigation(req: Request): boolean { + const dest = req.headers.get("sec-fetch-dest"); + if (dest === null) return true; + return dest === "document"; +} + +function payloadTooLargeResponse(): Response { + return new Response(JSON.stringify({ error: "payload too large" }), { + status: 413, + headers: { "content-type": "application/json" }, + }); +} + +/** Reads `req`'s body verbatim, buffered rather than piped (see the note + * below on why), refusing once the byte count crosses `maxBytes`. Cancels + * the reader (not merely releasing its lock) the moment it detects the + * overflow, so nothing keeps pumping past the cap — an absent or wrong + * `Content-Length` must not bypass this. */ +async function readCappedArrayBuffer(req: Request, maxBytes: number): Promise { + if (!req.body) return new ArrayBuffer(0); + const reader = req.body.getReader(); + const chunks: Uint8Array[] = []; + let total = 0; + try { + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + total += value.byteLength; + if (total > maxBytes) { + await reader.cancel().catch(() => {}); + throw new BodyTooLargeError(maxBytes); + } + chunks.push(value); + } + } finally { + reader.releaseLock(); + } + const out = new Uint8Array(total); + let offset = 0; + for (const chunk of chunks) { + out.set(chunk, offset); + offset += chunk.byteLength; + } + return out.buffer; +} + +export const handleGrafana: RouteHandler = async (req, env) => { + const identity = await verifySession(req, env); + if (!identity) { + // An unauthenticated request must NEVER wake the box. A top-level + // navigation gets a real sign-in redirect; everything else gets 401 + // JSON, which also recovers a session that expired mid-use on a + // background panel-refresh call without navigating away. + if (isBrowserNavigation(req)) { + const url = new URL(req.url); + const next = sanitizeNext(url.pathname + url.search, env); + return new Response(null, { + status: 302, + headers: { Location: `/grafana/_o11y/login?next=${encodeURIComponent(next)}` }, + }); + } + return new Response(JSON.stringify({ error: "unauthorized" }), { + status: 401, + headers: { "content-type": "application/json" }, + }); + } + + // Checked before the box is ever touched — an oversized request must + // not wake a stopped box. `Content-Length` is only a pre-check; + // `readCappedArrayBuffer` below is the real enforcement. + if (contentLengthExceeds(req, GRAFANA_PROXY_MAX_BYTES)) return payloadTooLargeResponse(); + + const box = getGrafanaBoxStub(env); + + if (!isTopLevelNavigation(req) && !(await box.isAwake())) { + // A background request reaching a STOPPED box: never start it. Serve + // the waking page instead of a bare error, so a stale open tab + // degrades quietly. + return wakingPageResponse(); + } + + try { + // Idempotent: an already-running box returns its existing wake + // record; `wake()` refuses only while the container is stopping. + await box.wake("visit"); + } catch { + return wakingPageResponse(); + } + + if (!(await box.isReady())) { + // A request that only ever saw the waking page still counts as + // visitor activity — otherwise a visit wake with an empty backlog + // would SIGTERM itself before anyone "visited". Called AFTER + // `wake()`: `#doWake` resets this same key at wake-start. + await box.noteVisitorActivity(); + return wakingPageResponse(); + } + + // Renews activity again now that a real request is about to reach + // Grafana — keeps the "renew only on HTTP requests" rule honest even + // after a stale waking-page hit from before boot. + await box.noteVisitorActivity(); + + // Reuse the ORIGINAL request's URL verbatim: Grafana behind a sub-path + // needs `Host` and path preserved exactly; `containerFetch` only + // rewrites the scheme. + // + // The body is read in full HERE, buffered rather than piped: when the + // box answers without reading a piped body (the live-path/Loki-allowlist + // refusals, the not-running 503s), this Worker would otherwise send that + // response while the runtime is still pumping the incoming body into the + // DO subrequest, throwing a stream error. + let body: ArrayBuffer | null = null; + if (req.body) { + try { + // Enforced again here, not just against the `Content-Length` hint + // above — an absent or wrong header must not let an oversized body + // reach the buffer at all. + body = await readCappedArrayBuffer(req, GRAFANA_PROXY_MAX_BYTES); + } catch (err) { + if (err instanceof BodyTooLargeError) return payloadTooLargeResponse(); + // The client went away mid-upload: nothing is left to answer. + return new Response(null, { status: 400 }); + } + } + const upstream = new Request(req.url, { method: req.method, headers: req.headers, body, redirect: req.redirect }); + for (const h of STRIPPED_HEADERS) upstream.headers.delete(h); + upstream.headers.set("x-o11y-grafana-user", identity.email); + + // The DO's `fetch()` handler, never the `containerFetch` RPC method: a + // body-bearing `Request` sent via RPC serialises as a stream, which + // throws inside the box DO. See `GrafanaBox.fetch`'s doc comment. + return box.fetch(upstream); +}; diff --git a/runner/workers/o11y/src/grafana/reopen.ts b/runner/workers/o11y/src/grafana/reopen.ts new file mode 100644 index 0000000000..205978d83b --- /dev/null +++ b/runner/workers/o11y/src/grafana/reopen.ts @@ -0,0 +1,72 @@ +// `POST /grafana/_o11y/reopen` (ADR §B.3/§J, contract §1): manual ledger +// re-open for a time window. Session-gated exactly like `/grafana/*`. +// Requires exact `application/json` content-type: without it, this route +// would be reachable via a cross-site "simple" request with no preflight. +// An exact `Origin` check sits on top, since `o11y_session` is +// `SameSite=Lax`. The window is capped to the 7-day retention. + +import { isSameOrigin, verifySession } from "../gates/session.js"; +import { inboxWriter } from "../inbox/accessor.js"; +import { reopenWindowExceedsRetention } from "../inbox/ledger.js"; +import type { RouteHandler } from "../router.js"; + +interface ReopenBody { + fromMs: number; + toMs: number; +} + +function isReopenBody(value: unknown): value is ReopenBody { + const v = value as Partial | null; + return ( + typeof v === "object" && + v !== null && + typeof v.fromMs === "number" && + typeof v.toMs === "number" && + Number.isFinite(v.fromMs) && + Number.isFinite(v.toMs) && + v.fromMs < v.toMs + ); +} + +/** Only `application/json` is accepted — this is the actual CSRF defense + * (see this file's header): it forces a preflight for any cross-origin + * caller. */ +function hasJsonContentType(req: Request): boolean { + const raw = req.headers.get("content-type"); + if (!raw) return false; + const mediaType = raw.split(";")[0]?.trim().toLowerCase(); + return mediaType === "application/json"; +} + +export const handleReopen: RouteHandler = async (req, env) => { + const identity = await verifySession(req, env); + if (!identity) return new Response("Forbidden", { status: 403 }); + + if (!isSameOrigin(req)) { + return new Response(JSON.stringify({ error: "bad_origin" }), { status: 403 }); + } + + if (!hasJsonContentType(req)) { + return new Response(JSON.stringify({ error: "expected content-type: application/json" }), { status: 415 }); + } + + let body: unknown; + try { + body = await req.json(); + } catch { + return new Response(JSON.stringify({ error: "invalid_json" }), { status: 400 }); + } + if (!isReopenBody(body)) { + return new Response(JSON.stringify({ error: "expected { fromMs: number, toMs: number }, fromMs < toMs" }), { + status: 400, + }); + } + if (reopenWindowExceedsRetention(body.fromMs, body.toMs)) { + return new Response(JSON.stringify({ error: "window exceeds the 7-day retention cap" }), { status: 400 }); + } + + const result = await inboxWriter(env).reopenWindow(body.fromMs, body.toMs); + console.log(JSON.stringify({ event: "o11y.reopen", by: identity.email, ...body, reopened: result.reopened })); + + return new Response(JSON.stringify(result), { status: 200, headers: { "content-type": "application/json" } }); +}; diff --git a/runner/workers/o11y/src/grafana/waking-page.ts b/runner/workers/o11y/src/grafana/waking-page.ts new file mode 100644 index 0000000000..bfcb5742fe --- /dev/null +++ b/runner/workers/o11y/src/grafana/waking-page.ts @@ -0,0 +1,57 @@ +// ADR §A: "The Handsontable logo, one line of text, ``, no script, served by the Worker while the box is +// not ready." Served for every `/grafana/*` request while `isReady()` is +// false — nothing here ever proxies to the container. The logo markup is +// `packages/editor-shell/src/logo.svg`, copied rather than imported: a +// Worker has no filesystem at request time. + +const LOGO_SVG = + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '' + + "" + + '' + + ""; + +/** No script (ADR §A) — the browser's own `meta refresh` does the polling. */ +export function wakingPageHtml(): string { + return ` + + + + +Waking up — Handsontable observability + + + +
+${LOGO_SVG} +

Waking up the observability box…

+
+ + +`; +} + +export function wakingPageResponse(): Response { + return new Response(wakingPageHtml(), { + status: 200, + headers: { "content-type": "text/html; charset=utf-8", "cache-control": "no-store" }, + }); +} diff --git a/runner/workers/o11y/src/heartbeat.ts b/runner/workers/o11y/src/heartbeat.ts new file mode 100644 index 0000000000..5507921b01 --- /dev/null +++ b/runner/workers/o11y/src/heartbeat.ts @@ -0,0 +1,26 @@ +// ADR-0041 §F.3 watchdog: `heartbeat()` returns lastCron, lastIngest and the +// backlog age. RPC-only: callers bind to it by name +// (`entrypoint: "O11yHeartbeat"`); the default export does not serve it, so no +// public request can reach it. + +import { WorkerEntrypoint } from "cloudflare:workers"; +import type { Env } from "./env.js"; +import { inboxWriter } from "./inbox/accessor.js"; + +export interface HeartbeatReport { + lastCron: number; + lastIngest: number; + backlogOldestAgeMs: number | null; +} + +export class O11yHeartbeat extends WorkerEntrypoint { + async heartbeat(): Promise { + return readHeartbeatReport(this.env); + } +} + +export async function readHeartbeatReport(env: Env): Promise { + const writer = inboxWriter(env); + const [heartbeat, backlogOldestAgeMs] = await Promise.all([writer.heartbeat(), writer.backlogOldestAgeMs()]); + return { lastCron: heartbeat.lastCron, lastIngest: heartbeat.lastIngest, backlogOldestAgeMs }; +} diff --git a/runner/workers/o11y/src/inbox/accessor.ts b/runner/workers/o11y/src/inbox/accessor.ts new file mode 100644 index 0000000000..067a3ff62c --- /dev/null +++ b/runner/workers/o11y/src/inbox/accessor.ts @@ -0,0 +1,17 @@ +// The one InboxWriter accessor. `.jurisdiction("eu")` and the plain +// namespace mint different ids for the same name, so every caller must use +// this rather than repeating the two-call chain. `InboxWriter`'s one +// instance is EU-pinned (ADR §H). + +import type { Env } from "../env.js"; + +// No explicit `DurableObjectStub` return-type annotation: +// spelling it out sends `tsc` into `TS2589` through the `Env` ↔ +// `DurableObjectNamespace` ↔ `InboxWriter` type-only cycle; +// plain inference avoids the eager expansion. Locally, workerd throws +// `Jurisdiction restrictions are not implemented`, so `O11Y_ENV === +// "local"` skips it; production always calls `.jurisdiction("eu")`. +export function inboxWriter(env: Env) { + const ns = env.O11Y_ENV === "local" ? env.INBOX_WRITER : env.INBOX_WRITER.jurisdiction("eu"); + return ns.get(ns.idFromName("main")); +} diff --git a/runner/workers/o11y/src/inbox/dedupe.ts b/runner/workers/o11y/src/inbox/dedupe.ts new file mode 100644 index 0000000000..7c8f80e3fb --- /dev/null +++ b/runner/workers/o11y/src/inbox/dedupe.ts @@ -0,0 +1,117 @@ +// ADR §B.2 step 4: "Deduplicate in `InboxWriter`: each hash is checked +// against a 24-hour set in DO storage." Pure over {@link StorageLike}, +// unit-testable without a real DO. `hash:` as a flat, permanent +// set heads toward the 10 GB SQLite-DO limit within days under sustained +// traffic; keys are day-bucketed instead (`hash::`), and +// `pruneHashBuckets` sweeps stale buckets with a bounded range read. + +import { DEDUPE_WINDOW_MS } from "@handsontable/demo-runtime/telemetry"; +import { deleteChunked, getManyChunked, type StorageLike } from "./storage.js"; + +const HASH_PREFIX = "hash:"; +const DAY_MS = 24 * 60 * 60 * 1000; +/** How many stale `hash:` rows one `pruneHashBuckets` call may delete — + * bounds the cost of a call after a quiet period let a backlog build up. + * Against a 10-minute cron, 500/tick tops out at 72,000/day, below ADR + * §D's 10× headroom projection of ~220,000/day. 5,000/tick gives + * 720,000/day, over 3× that headroom. */ +const HASH_PRUNE_BATCH_LIMIT = 5000; + +/** `yyyymmdd`, UTC — sorts lexicographically in chronological order, which + * is what makes `pruneHashBuckets`'s range delete correct without + * inspecting each row's value. */ +function dayBucket(ms: number): string { + return new Date(ms).toISOString().slice(0, 10).replace(/-/g, ""); +} + +function bucketedHashKey(bucket: string, sha256Hex: string): string { + return `${HASH_PREFIX}${bucket}:${sha256Hex}`; +} + +export interface DedupeResult { + /** One entry per input hash, index-aligned with `hashes`: `true` when + * THAT occurrence is a duplicate — already seen, or a later copy of a + * hash earlier in this same batch. The first in-batch occurrence of an + * unseen hash is `false`: it is the copy that gets stored. Per + * occurrence, not per hash: a plain `Set` of duplicate hashes would + * drop the first copy of an in-batch repeat along with the later ones, + * losing the record for good. */ + isDuplicate: readonly boolean[]; + /** `hash::` entries to write for every non-duplicate + * hash — the caller commits these in the same transaction as the + * rows/fingerprints. */ + writes: Readonly>; +} + +/** Checks `hashes` (in order — a within-batch repeat keeps only its first + * occurrence as non-duplicate) against storage's 24 h dedupe window. The + * 24h window can only ever straddle two calendar-day buckets, so + * checking exactly those two per unique hash is correct and bounded. + * Read-only: a caller that decides not to commit never leaves a hash + * marked seen for a record that was never stored. */ +export async function checkDuplicates( + storage: StorageLike, + hashes: readonly string[], + nowMs: number, +): Promise { + const uniqueHashes = [...new Set(hashes)]; + const today = dayBucket(nowMs); + const yesterday = dayBucket(nowMs - DAY_MS); + + // `today`/`yesterday` are UTC calendar days (no DST discontinuity), so + // they are always exactly one day apart and never equal — both are always + // worth a lookup. + const lookupKeys: string[] = []; + for (const hash of uniqueHashes) { + lookupKeys.push(bucketedHashKey(today, hash)); + lookupKeys.push(bucketedHashKey(yesterday, hash)); + } + // A batch of 65+ unique records (still under the 200-item ingest cap) + // already needs 130+ lookup keys here (2 buckets each) — over the real + // DO storage limit (`storage.ts#DO_STORAGE_MAX_KEYS_PER_CALL`). + const existing = await getManyChunked(storage, lookupKeys); + + const isDuplicate: boolean[] = []; + const writes: Record = {}; + const acceptedThisBatch = new Set(); + + for (const hash of hashes) { + if (acceptedThisBatch.has(hash)) { + isDuplicate.push(true); + continue; + } + const firstSeen = existing.get(bucketedHashKey(today, hash)) ?? existing.get(bucketedHashKey(yesterday, hash)); + const withinWindow = firstSeen !== undefined && nowMs - firstSeen < DEDUPE_WINDOW_MS; + if (withinWindow) { + isDuplicate.push(true); + } else { + writes[bucketedHashKey(today, hash)] = nowMs; + acceptedThisBatch.add(hash); + isDuplicate.push(false); + } + } + + return { isDuplicate, writes }; +} + +export interface HashPruneResult { + hashDeleted: number; +} + +/** Deletes `hash:` rows whose bucket is strictly older than `keepDays` + * full days ago (default 2 — today+yesterday are the only buckets + * `checkDuplicates` reads). Bounded per call via a `start`/`end` range + * read, never a full-prefix scan; a backlog clears incrementally across + * successive calls, oldest rows first. Called from `writer.ts#backlog()`, + * wrapped in `try/catch` there. */ +export async function pruneHashBuckets(storage: StorageLike, nowMs: number, keepDays = 2): Promise { + const cutoffBucket = dayBucket(nowMs - keepDays * DAY_MS); + const stale = await storage.list({ + start: HASH_PREFIX, + end: `${HASH_PREFIX}${cutoffBucket}:`, + limit: HASH_PRUNE_BATCH_LIMIT, + }); + const toDelete = [...stale.keys()]; + if (toDelete.length > 0) await deleteChunked(storage, toDelete); + return { hashDeleted: toDelete.length }; +} diff --git a/runner/workers/o11y/src/inbox/ledger.ts b/runner/workers/o11y/src/inbox/ledger.ts new file mode 100644 index 0000000000..7a74c7c35e --- /dev/null +++ b/runner/workers/o11y/src/inbox/ledger.ts @@ -0,0 +1,557 @@ +// The ledger (ADR §B.3): `written` → `provisional(wakeId)` → `committed` | +// `rejected` transitions, resolved at every cron tick and at the start of +// each wake, plus `backlog()` and the manual `reopen` window. Pure over +// `StorageLike` (unit-testable via `memoryStorage()`, no real DO needed). +// Bounded throughout: `key:`/`wake:` scans and `done:`/`hash:` pruning stay +// O(live rows), not O(all-time history) — see `pipeline/o11y-ledger-scale.test.mjs`. + +import { + doneKeyStorageKey, + inboxKeyStorageKey, + parseInboxKey, + wakeStorageKey, + type InboxKeyState, + type Tenant, + type WakeState, +} from "@handsontable/demo-runtime/telemetry"; +import { deleteChunked, getManyChunked, putChunked, type StorageLike } from "./storage.js"; + +const WAKE_PREFIX = "wake:"; +const KEY_PREFIX = "key:"; +const DONE_PREFIX = "done:"; +const PROVISIONAL_PREFIX = "provisional:"; +// `nextWrittenKeys`'s sort order alone gives reopened keys drain priority, +// so `reopenWindow` also drops a one-shot `reopenmark:` per key it moves to +// `written`, consumed by `takeReopenedFlag` so the drain's `o11y.drain` +// point can report `reason: "reopen"`. +const REOPEN_MARK_PREFIX = "reopenmark:"; +function reopenMarkStorageKey(inboxKey: string): string { + return `${REOPEN_MARK_PREFIX}${inboxKey}`; +} + +/** Contract §8: inbox objects live 7 days (R2 lifecycle) — `done:` + * pruning and the manual-reopen window cap both use this same window. */ +export const KEY_RETENTION_MS = 7 * 24 * 60 * 60 * 1000; + +/** How many `done:` rows one `pruneLedger` call may delete. Raised from 500 + * — a 10-min cron at 500/tick tops out at 72,000/day, below ADR §D's 10× + * headroom projection of ~220,000/day. */ +const PRUNE_BATCH_LIMIT = 5000; + +function wakeIdOf(storageKey: string): string { + return storageKey.slice(WAKE_PREFIX.length); +} +function inboxKeyOf(storageKey: string): string { + return storageKey.slice(KEY_PREFIX.length); +} + +export interface LedgerDeps { + /** Best-available "is the box still running the wake the ledger thinks + * is current" signal (`GrafanaBox.isAwake()`), backed by `getState()` + * rather than the container's live flag. Called with no wakeId: at most + * one wake is ever not-over at a time. An outbound RPC briefly opens + * this DO's input gate; `resolveOverWakes` is written to be correct + * across that reopening. */ + isBoxRunning(): Promise; + /** `state/wakes//clean` exists in the Loki bucket (ADR §A/§B.3). + * Also a real R2 `head()` call — same input-gate note as + * {@link isBoxRunning} applies. */ + markerExists(wakeId: string): Promise; +} + +export interface WakeResolution { + wakeId: string; + reason: WakeState["reason"]; + clean: boolean; + keysAffected: number; + /** The wake's `readyMs` as stored at resolution time (read inside the + * resolving transaction, so the freshest value) — `undefined` when the + * box never became ready during this wake. */ + readyMs?: number; +} + +export interface ResolveResult { + /** wakeIds newly marked `over: true` this call (either because a newer + * wake already superseded them via `recordWake`, or because this call + * observed the box not running). */ + newlyOver: string[]; + /** Every wake this call actually resolved (deleted `wake:` for) — + * `keysAffected` can be 0 and still counts, since the `o11y.wake` point + * is about the wake's outcome, not whether it had keys left. */ + resolved: WakeResolution[]; +} + +/** Resolves a single wake's provisional keys against the marker, moving + * each to `done:`/`written` as appropriate and deleting `wake:` — all + * inside one `storage.transaction()`, so a crash mid-way leaves the + * pre-transaction state rather than an orphaned partial write (chunked + * put/delete would otherwise leave keys stuck in `provisional:` + * forever). Re-reads each key's CURRENT state inside the transaction, + * rather than trusting a possibly-stale caller snapshot — a concurrent + * manual reopen can move a key back to `written` between the read and + * this point, and that must not be silently undone. */ +async function finalizeWakeResolution( + storage: StorageLike, + wake: WakeState, + wakeId: string, + provisionalStorageKeys: readonly string[], + clean: boolean, +): Promise { + return storage.transaction(async (txn) => { + const stillThere = await txn.get(wakeStorageKey(wakeId)); + if (!stillThere) return null; // a concurrent call already resolved this wake + + const marker = `${PROVISIONAL_PREFIX}${wakeId}`; + const currentStates = + provisionalStorageKeys.length > 0 ? await getManyChunked(txn, provisionalStorageKeys) : new Map(); + + const writes: Record = {}; + const doneWrites: Record = {}; + const toDelete: string[] = []; + let keysAffected = 0; + for (const storageKey of provisionalStorageKeys) { + if (currentStates.get(storageKey) !== marker) continue; // no longer this wake's — a concurrent reopen won + keysAffected++; + if (clean) { + // Move OUT of `key:` into `done:` on commit, so `key:` never + // accumulates committed history (see this file's header, point 1). + toDelete.push(storageKey); + doneWrites[doneKeyStorageKey(inboxKeyOf(storageKey))] = 1; + } else { + writes[storageKey] = "written"; + } + } + toDelete.push(wakeStorageKey(wakeId)); // deleted last — see this function's own doc comment + + await putChunked(txn, { ...writes, ...doneWrites }); + await deleteChunked(txn, toDelete); + + return { wakeId, reason: wake.reason, clean, keysAffected, readyMs: stillThere.readyMs }; + }); +} + +/** + * ADR §B.3: "at each cron tick and at the start of each wake, InboxWriter + * resolves every wake that still owns provisional keys and is over." A + * wake is over when a newer wake started, or the box is observed not + * running (this call's own job for the CURRENT wake). + */ +export async function resolveOverWakes(storage: StorageLike, deps: LedgerDeps): Promise { + const wakes = await storage.list({ prefix: WAKE_PREFIX }); + const newlyOver: string[] = []; + const resolved: WakeResolution[] = []; + + const alreadyOver: [string, WakeState][] = []; + let active: [string, WakeState] | null = null; + for (const [storageKey, wake] of wakes) { + if (wake.over) alreadyOver.push([wakeIdOf(storageKey), wake]); + else active = [storageKey, wake]; // at most one, by construction (recordWake's invariant) + } + + // Wakes already `over`: their key sets cannot grow further + // (`markKeysProvisional` refuses once `over` is true), so one SHARED + // `key:` scan grouped by wakeId resolves all of them: O(wakes+keys). + if (alreadyOver.length > 0) { + const keys = await storage.list({ prefix: KEY_PREFIX }); + const byWake = new Map(); + for (const [storageKey, state] of keys) { + if (typeof state !== "string" || !state.startsWith(PROVISIONAL_PREFIX)) continue; + const owner = state.slice(PROVISIONAL_PREFIX.length); + const list = byWake.get(owner); + if (list) list.push(storageKey); + else byWake.set(owner, [storageKey]); + } + for (const [wakeId, wake] of alreadyOver) { + const provisionalKeys = byWake.get(wakeId) ?? []; + // Zero provisional keys ever recorded for this wake is trivially + // clean — no marker was ever going to exist. + const clean = provisionalKeys.length > 0 ? await deps.markerExists(wakeId) : true; + const outcome = await finalizeWakeResolution(storage, wake, wakeId, provisionalKeys, clean); + if (outcome) resolved.push(outcome); + } + } + + // The one wake that may become `over` THIS call. + if (active) { + const [storageKey, wake] = active; + const wakeId = wakeIdOf(storageKey); + const stillRunning = await deps.isBoxRunning(); // input-gate-opening await + if (!stillRunning) { + // Mark over FIRST, before reading provisional keys, so a concurrent + // `markKeysProvisional` either lands before the fresh read or is + // refused (it checks `over` itself). Re-read `wake` here too — a + // `recordWakeReady` landed during `isBoxRunning()` must not be + // overwritten by the stale snapshot. + const fresh = (await storage.get(storageKey)) ?? wake; + await storage.put({ [storageKey]: { ...fresh, over: true } satisfies WakeState }); + newlyOver.push(wakeId); + + const freshKeys = await storage.list({ prefix: KEY_PREFIX }); + const provisionalKeys: string[] = []; + const marker = `${PROVISIONAL_PREFIX}${wakeId}`; + for (const [sk, state] of freshKeys) if (state === marker) provisionalKeys.push(sk); + + const clean = provisionalKeys.length > 0 ? await deps.markerExists(wakeId) : true; + const outcome = await finalizeWakeResolution(storage, { ...fresh, over: true }, wakeId, provisionalKeys, clean); + if (outcome) resolved.push(outcome); + } + } + + return { newlyOver, resolved }; +} + +/** A key becomes `provisional(wakeId)` only after all its requests + * returned `2xx` (ADR §B.3). Refuses — inside one transaction — when + * `wake:` is over or missing: a key marked provisional under a + * gone wake would sit there forever, and a stale-snapshot race could + * commit it without ever passing the marker check. Read-then-write inside + * `storage.transaction()` makes the check and the write atomic. */ +export async function markKeysProvisional(storage: StorageLike, wakeId: string, keys: readonly string[]): Promise { + if (keys.length === 0) return; + await storage.transaction(async (txn) => { + const wake = await txn.get(wakeStorageKey(wakeId)); + if (!wake || wake.over) return; // refuse: unknown or already-over wake + const writes: Record = {}; + for (const key of keys) writes[inboxKeyStorageKey(key)] = `provisional:${wakeId}`; + // A drain batch can carry more than 128 keys (DRAIN_BATCH_SIZE, box.ts). + await putChunked(txn, writes); + }); +} + +/** A key whose drain pushed ZERO bytes (already deduped/too old) has + * nothing an unclean stop could lose, so it skips the wake-marker + * durability check entirely. Routing it through `markKeysProvisional` + * instead would leave a wake with only zero-byte keys looking unclean + * forever (no Loki marker was ever written for it), re-waking and + * re-draining the same empty keys on a loop. This commits straight + * `written` → `done:`, safe since nothing durable rides on it. */ +export async function commitKeys(storage: StorageLike, keys: readonly string[]): Promise { + if (keys.length === 0) return; + await storage.transaction(async (txn) => { + const doneWrites: Record = {}; + for (const key of keys) doneWrites[doneKeyStorageKey(key)] = 1; + await putChunked(txn, doneWrites); + // Chunked to the real 128-key DO limit, same as `finalizeWakeResolution`. + await deleteChunked( + txn, + keys.map((key) => inboxKeyStorageKey(key)), + ); + }); +} + +// ---- backlog() --------------------------------------------------------------- + +export interface InboxObjectInfo { + key: string; + size: number; + /** R2's own `uploaded` timestamp — exact, unlike deriving age from the + * key's hour bucket (which understates age by up to 59 minutes and + * wakes the box early). No extra request cost from `list()`. */ + uploaded: Date; +} + +export interface BacklogInfo { + /** 0 when there is no backlog (nothing `written`). */ + oldestWrittenAgeMs: number; + totalBytes: number; + writtenCount: number; + drainsPaused: boolean; +} + +/** + * `backlog()` "over `written` keys only, computed after that resolution" + * (ADR §B.3) — callers must run {@link resolveOverWakes} first in the same + * call (`InboxWriter.backlog()`, writer.ts, does this). + */ +export async function computeBacklog( + storage: StorageLike, + listInboxObjects: () => Promise, + drainsPaused: boolean, +): Promise { + const keys = await storage.list({ prefix: KEY_PREFIX }); + const written = new Set(); + for (const [storageKey, state] of keys) { + if (state === "written") written.add(inboxKeyOf(storageKey)); + } + if (written.size === 0) { + return { oldestWrittenAgeMs: 0, totalBytes: 0, writtenCount: 0, drainsPaused }; + } + + const objects = await listInboxObjects(); + let totalBytes = 0; + let oldestUploadedMs: number | null = null; + for (const obj of objects) { + if (!written.has(obj.key)) continue; + totalBytes += obj.size; + const t = obj.uploaded.getTime(); + if (oldestUploadedMs === null || t < oldestUploadedMs) oldestUploadedMs = t; + } + + return { + oldestWrittenAgeMs: oldestUploadedMs === null ? 0 : Math.max(0, Date.now() - oldestUploadedMs), + totalBytes, + writtenCount: written.size, + drainsPaused, + }; +} + +// ---- Drain support: ordering, rejection --------------------------------------- + +/** + * `written` keys, in key order. The inbox key format sorts chronologically + * within a tenant by construction, so this satisfies "re-opened keys + * first" without a separate flag — a re-opened key is always older than + * any key from the current wake. `excludeTenants` (tenants the drain found + * stream-limited this wake) are skipped before the limit applies, so the + * batch fills with the other tenant's keys. + */ +export async function nextWrittenKeys( + storage: StorageLike, + limit: number, + excludeTenants: readonly string[] = [], +): Promise { + const keys = await storage.list({ prefix: KEY_PREFIX }); + const excludedPrefixes = excludeTenants.map((t) => `inbox/${t}/`); + const written: string[] = []; + for (const [storageKey, state] of keys) { + if (state !== "written") continue; + const inboxKey = inboxKeyOf(storageKey); + if (excludedPrefixes.some((prefix) => inboxKey.startsWith(prefix))) continue; + written.push(inboxKey); + } + written.sort(); + return written.slice(0, limit); +} + +// ---- rejectedEvent: audit log ---------------------------------------------- +// `rejected:` `key:` entries are never pruned, so a plain count-based rule +// would fire forever; this chronological, independently-prunable log gives +// the alert a REJECTION TIME to fire-once/resolve-once on instead. +const REJECTED_EVENT_PREFIX = "rejectedEvent:"; +const REJECTED_EVENT_TIMESTAMP_DIGITS = 15; +/** Same window contract §8 already uses for `done:`/`hash:`/reopen (the + * underlying inbox object's own 7-day retention) — past this, nothing + * about the rejection is diagnosable any more anyway. */ +const REJECTED_EVENT_RETENTION_MS = KEY_RETENTION_MS; + +function rejectedEventStorageKey(ms: number, inboxKeyStr: string): string { + return `${REJECTED_EVENT_PREFIX}${Math.max(0, Math.trunc(ms)).toString().padStart(REJECTED_EVENT_TIMESTAMP_DIGITS, "0")}:${inboxKeyStr}`; +} + +/** A `400` marks the key `rejected`, logged with Loki's message (ADR + * §B.3) — never retried by a later wake. Stays under `key:` past + * retention only via `pruneLedger`'s value-filtered sweep (not `done:`'s + * blind range delete), since an operator diagnosing the alert needs to + * still find it. Also logs a `rejectedEvent:` entry so the alert can + * tell "rejected, ever" from "rejected, recently." */ +export async function rejectKey(storage: StorageLike, key: string, reason: string, nowMs = Date.now()): Promise { + await storage.put({ + [inboxKeyStorageKey(key)]: `rejected:${reason}` satisfies InboxKeyState, + [rejectedEventStorageKey(nowMs, key)]: reason, + }); +} + +/** A key with at least one 2xx chunk AND one permanently-400'd chunk + * stays `provisional` (its accepted content is durable — see + * `drain.ts#drainKey`), so it never becomes `rejected:`. Logs the + * SAME `rejectedEvent:` entry `rejectKey` would, so the loss is still + * operator-visible even though the key itself durably resolves. */ +export async function recordPartialReject(storage: StorageLike, key: string, reason: string, nowMs = Date.now()): Promise { + await storage.put({ [rejectedEventStorageKey(nowMs, key)]: reason }); +} + +/** Count of `rejectedEvent:` entries strictly newer than `sinceMs` — bounded + * `start`/`end` range read (chronologically keyed by construction, same + * pattern as `fpts:`), never a full-prefix scan. */ +export async function recentRejectionCount(storage: StorageLike, sinceMs: number): Promise { + const start = rejectedEventStorageKey(sinceMs + 1, ""); + const end = `${REJECTED_EVENT_PREFIX}￿`; + const page = await storage.list({ start, end, limit: PRUNE_BATCH_LIMIT }); + return page.size; +} + +// ---- Manual reopen (POST /grafana/_o11y/reopen) ------------------------------- + +export interface ReopenResult { + reopened: number; +} + +/** The manual reopen window is capped to {@link KEY_RETENTION_MS} — + * nothing past it can still exist, so a wider request is refused up + * front rather than silently reopening nothing. */ +export function reopenWindowExceedsRetention(fromMs: number, toMs: number): boolean { + return toMs - fromMs > KEY_RETENTION_MS; +} + +/** + * Re-opens every key whose inbox-key hour bucket overlaps `[fromMs, toMs)`: + * live `key:` entries (except the active wake's own in-flight + * `provisional:` keys) AND `done:` entries (committed keys live there, not + * under `key:`). Both scans are bounded by {@link KEY_RETENTION_MS}, + * enforced by the caller via `reopenWindowExceedsRetention`. + */ +export async function reopenWindow( + storage: StorageLike, + fromMs: number, + toMs: number, + activeWakeId: string | null, +): Promise { + const overlaps = (inboxKey: string): boolean => { + const parsed = parseInboxKey(inboxKey); + if (!parsed) return false; + const hourStart = Date.UTC( + Number(parsed.date.slice(0, 4)), + Number(parsed.date.slice(5, 7)) - 1, + Number(parsed.date.slice(8, 10)), + Number(parsed.hour), + ); + const hourEnd = hourStart + 60 * 60 * 1000; + return !(hourEnd <= fromMs || hourStart >= toMs); + }; + + const writes: Record = {}; + const toDelete: string[] = []; + + const keys = await storage.list({ prefix: KEY_PREFIX }); + const protectedState = activeWakeId ? `${PROVISIONAL_PREFIX}${activeWakeId}` : null; + for (const [storageKey, state] of keys) { + if (state === "written") continue; // already open + if (protectedState && state === protectedState) continue; // in-flight — protected + if (!overlaps(inboxKeyOf(storageKey))) continue; + writes[storageKey] = "written"; + } + + const done = await storage.list<1>({ prefix: DONE_PREFIX }); + for (const [storageKey] of done) { + const inboxKeyStr = storageKey.slice(DONE_PREFIX.length); + if (!overlaps(inboxKeyStr)) continue; + writes[inboxKeyStorageKey(inboxKeyStr)] = "written"; + toDelete.push(storageKey); + } + + const reopened = Object.keys(writes).length; + // One transient `reopenmark:` per key this call moves to `written`, in + // the SAME transaction as the `written` write itself — see this file's + // `REOPEN_MARK_PREFIX` doc comment. + const marks: Record = {}; + for (const storageKey of Object.keys(writes)) { + marks[reopenMarkStorageKey(inboxKeyOf(storageKey))] = 1; + } + // Wrapped in one transaction so a crash mid-chunk never leaves a `done:` + // entry deleted without its `key: = written` twin written, or the + // reverse — same atomicity approach as `finalizeWakeResolution`. + await storage.transaction(async (txn) => { + if (reopened > 0) { + await putChunked(txn, writes); + await putChunked(txn, marks); + } + if (toDelete.length > 0) await deleteChunked(txn, toDelete); + }); + return { reopened }; +} + +/** + * Consumes (reads AND clears) the reopen markers `reopenWindow` left for + * `inboxKeys`, one-shot, and reports whether ANY were found. Called once + * per drain batch so its `o11y.drain` point can emit `reason: "reopen"`. + */ +export async function takeReopenedFlag(storage: StorageLike, inboxKeys: readonly string[]): Promise { + if (inboxKeys.length === 0) return false; + const markKeys = inboxKeys.map(reopenMarkStorageKey); + const found = await getManyChunked(storage, markKeys); + if (found.size === 0) return false; + await deleteChunked(storage, [...found.keys()]); + return true; +} + +/** Records a wake's wake-to-ready time on `wake:`, once — first + * call wins, and an already-resolved wake is left alone. */ +export async function recordWakeReady(storage: StorageLike, wakeId: string, readyMs: number): Promise { + await storage.transaction(async (txn) => { + const key = wakeStorageKey(wakeId); + const wake = await txn.get(key); + if (!wake || wake.readyMs !== undefined) return; + await txn.put({ [key]: { ...wake, readyMs } satisfies WakeState }); + }); +} + +/** The current not-over wake's id, or `null` (fully stopped). Used by the + * reopen route to protect an in-flight drain (see {@link reopenWindow}) and + * by drain/wake orchestration to know "which wakeId am I." */ +export async function currentWakeId(storage: StorageLike): Promise { + const wakes = await storage.list({ prefix: WAKE_PREFIX }); + for (const [storageKey, wake] of wakes) { + if (!wake.over) return wakeIdOf(storageKey); + } + return null; +} + +// ---- Pruning: bounded, retention-based housekeeping ------------------------ + +export interface PruneResult { + doneDeleted: number; + /** Stale `key: = rejected:` entries deleted this call — see + * `pruneLedger`'s own doc comment. */ + rejectedDeleted: number; +} + +/** Deletes `done:` entries older than {@link KEY_RETENTION_MS} — a + * `done:` entry for an object R2 has already deleted is worthless. + * Bounded per call via a `start`/`end` range delete: `done:inbox//` + * sorts chronologically, so `[start, cutoffDate)` names exactly the + * oldest stale rows. Called from `writer.ts#backlog()`, wrapped in + * `try/catch` there so a pruning failure never fails the backlog read. */ +export async function pruneLedger(storage: StorageLike, nowMs: number): Promise { + const cutoff = new Date(nowMs - KEY_RETENTION_MS); + const cutoffDate = cutoff.toISOString().slice(0, 10); // yyyy-mm-dd, UTC + + let doneDeleted = 0; + for (const tenant of ["browser", "worker"] satisfies Tenant[]) { + const stale = await storage.list<1>({ + start: `${DONE_PREFIX}inbox/${tenant}/`, + end: `${DONE_PREFIX}inbox/${tenant}/${cutoffDate}/`, + limit: PRUNE_BATCH_LIMIT, + }); + const toDelete = [...stale.keys()]; + if (toDelete.length > 0) { + await deleteChunked(storage, toDelete); + doneDeleted += toDelete.length; + } + } + + // `rejected:` entries need pruning too, but `key:inbox//` + // mixes live and rejected entries chronologically, so this range read + // must filter by VALUE, not blind-delete. A batch whose oldest rows are + // all non-rejected makes no progress this tick; it converges over later + // ticks. + let rejectedDeleted = 0; + for (const tenant of ["browser", "worker"] satisfies Tenant[]) { + const stale = await storage.list({ + start: `${KEY_PREFIX}inbox/${tenant}/`, + end: `${KEY_PREFIX}inbox/${tenant}/${cutoffDate}/`, + limit: PRUNE_BATCH_LIMIT, + }); + const toDelete: string[] = []; + for (const [key, state] of stale) { + if (typeof state === "string" && state.startsWith("rejected:")) toDelete.push(key); + } + if (toDelete.length > 0) { + await deleteChunked(storage, toDelete); + rejectedDeleted += toDelete.length; + } + } + + // The `rejectedEvent:` audit log is itself chronologically keyed, so a + // plain bounded range delete (no value filtering) prunes it too. + const rejectedEventStale = await storage.list({ + start: REJECTED_EVENT_PREFIX, + end: rejectedEventStorageKey(nowMs - REJECTED_EVENT_RETENTION_MS, ""), + limit: PRUNE_BATCH_LIMIT, + }); + const rejectedEventToDelete = [...rejectedEventStale.keys()]; + if (rejectedEventToDelete.length > 0) await deleteChunked(storage, rejectedEventToDelete); + + return { doneDeleted, rejectedDeleted }; +} + +export { wakeStorageKey }; diff --git a/runner/workers/o11y/src/inbox/pack.ts b/runner/workers/o11y/src/inbox/pack.ts new file mode 100644 index 0000000000..9f91845ea8 --- /dev/null +++ b/runner/workers/o11y/src/inbox/pack.ts @@ -0,0 +1,214 @@ +// ADR §B.2 steps 5–6: append accepted records into ≤ 1 MB storage rows (the +// arrival time on the row, never in the record — §8), then pack one gzipped +// NDJSON object per tenant, `` persisted in the same transaction that +// records the object's `key: = written` and deletes the packed rows. + +import { + buildResourceLogs, + encodeNdjson, + inboxKey, + inboxKeyStorageKey, + INBOX_ROW_MAX_BYTES, + pendingRowStorageKey, + SEQ_STORAGE_KEY, + type NormalisedRecord, + type OtlpResourceLogs, + type Tenant, +} from "@handsontable/demo-runtime/telemetry"; +import { deleteChunked, type StorageLike } from "./storage.js"; + +export interface PendingRow { + tenant: Tenant; + /** ADR §8: "the arrival time itself never becomes part of a stored + * record" — it lives here, on the row, set once when the row is written. */ + arrivalMs: number; + resourceLogs: OtlpResourceLogs[]; +} + +const ROW_PREFIX = "row:"; +/** Not a contract-named key — an `InboxWriter`-internal monotonic counter, + * separate from the pack `seq` (§8), so row numbering survives a restart + * the same way `seq` does. */ +const ROW_SEQ_STORAGE_KEY = "rowSeq"; + +export interface AppendResult { + writes: Record; + nextRowSeq: number; + bytesAdded: number; +} + +/** Chunks `records` into rows of at most {@link INBOX_ROW_MAX_BYTES} (a + * single oversized record is impossible here — `normalise/` already drops + * anything over `INBOX_RECORD_MAX_BYTES`, well under the row cap) and + * returns the `row:` entries to write, reading/advancing `rowSeq` from + * `storage` first. Pure: does not write — the caller commits these + * alongside the dedupe/fingerprint/heartbeat writes in one transaction. */ +export async function appendRows( + storage: StorageLike, + tenant: Tenant, + arrivalMs: number, + records: readonly NormalisedRecord[], +): Promise { + let rowSeq = (await storage.get(ROW_SEQ_STORAGE_KEY)) ?? 0; + const writes: Record = {}; + let bytesAdded = 0; + + let current: OtlpResourceLogs[] = []; + let currentBytes = 0; + const flush = () => { + if (current.length === 0) return; + writes[pendingRowStorageKey(rowSeq)] = { tenant, arrivalMs, resourceLogs: current }; + rowSeq += 1; + bytesAdded += currentBytes; + current = []; + currentBytes = 0; + }; + + for (const record of records) { + const resourceLogs = buildResourceLogs(record); + const size = new TextEncoder().encode(JSON.stringify(resourceLogs)).length; + if (currentBytes + size > INBOX_ROW_MAX_BYTES && current.length > 0) flush(); + current.push(resourceLogs); + currentBytes += size; + } + flush(); + + return { writes, nextRowSeq: rowSeq, bytesAdded }; +} + +export { ROW_SEQ_STORAGE_KEY }; + +async function gzip(text: string): Promise { + // Fully read the compressed stream before returning — a `put` against a + // still-draining `CompressionStream` output is a documented trap. + const stream = new Blob([text]).stream().pipeThrough(new CompressionStream("gzip")); + const buf = await new Response(stream).arrayBuffer(); + return new Uint8Array(buf); +} + +export interface PackedObject { + key: string; + bytes: number; + consumedRowKeys: string[]; +} + +/** ADR §B.2 step 6's "or at 4 MB stored" bounds the packed OBJECT itself: + * without it, a sustained ingest flood (~100 MB/min for one IP) could + * push the alarm's in-memory gzip past the DO's 128 MB memory before + * anything is packed, causing every later alarm to retry against an + * ever-larger pending set. The caller (`writer.ts#alarm()`) loops over + * leftover rows. */ +export const PACK_OBJECT_MAX_DECOMPRESSED_BYTES = 4 * 1024 * 1024; + +/** Packs pending rows for `tenant`, up to + * {@link PACK_OBJECT_MAX_DECOMPRESSED_BYTES} decompressed, into one + * gzipped NDJSON R2 object, keyed by the **first** packed row's arrival + * time — not `Date.now()`, so a retry after a crash lands on the + * identical key instead of producing a divergent object. `null` when + * nothing is pending. Writes to R2 directly (not transactional with DO + * storage); the caller commits `seq`/`key:`/row deletion right + * after, per ADR §B.2 step 6. */ +export async function packTenant( + storage: StorageLike, + bucket: R2Bucket, + tenant: Tenant, + rows: [string, PendingRow][], +): Promise { + const first = rows[0]; + if (!first) return null; + + // Take rows in order until the byte budget is spent; always takes at + // least one row, even if it alone is over budget — a stuck pending set + // is worse than one oversized object. + let budget = 0; + let cut = rows.length; + for (let i = 0; i < rows.length; i++) { + const row = rows[i]; + if (!row) break; + const rowBytes = new TextEncoder().encode(encodeNdjson(row[1].resourceLogs)).length; + if (i > 0 && budget + rowBytes > PACK_OBJECT_MAX_DECOMPRESSED_BYTES) { + cut = i; + break; + } + budget += rowBytes; + } + const taken = rows.slice(0, cut); + + const allLogs = taken.flatMap(([, row]) => row.resourceLogs); + const ndjson = encodeNdjson(allLogs); + const gz = await gzip(ndjson); + + const seq = (await storage.get(SEQ_STORAGE_KEY)) ?? 0; + // `taken[0]` is always `first` (the loop above always takes at least the + // row at index 0) — `?? first` only satisfies the type checker's + // (correct, in general) indexed-access uncertainty, not a real fallback. + const firstArrival = new Date((taken[0] ?? first)[1].arrivalMs); + const key = inboxKey(tenant, firstArrival, seq); + + await bucket.put(key, gz); + + return { key, bytes: gz.byteLength, consumedRowKeys: taken.map(([k]) => k) }; +} + +/** Commits a {@link PackedObject}: `seq += 1`, `key: = "written"`, the + * consumed rows deleted — one atomic write (ADR §B.2 step 6: "`` is … + * incremented in the same transaction that records the key"). */ +export async function commitPackedObject(storage: StorageLike, packed: PackedObject): Promise { + await storage.transaction(async (txn) => { + const seq = (await txn.get(SEQ_STORAGE_KEY)) ?? 0; + await txn.put({ + [SEQ_STORAGE_KEY]: seq + 1, + [inboxKeyStorageKey(packed.key)]: "written", + }); + // Many small rows can pack more than 128 into one object. + await deleteChunked(txn, packed.consumedRowKeys); + }); +} + +// ---- Bounded reads for the pack alarm -------------------------------------- +// `row:` is zero-padded, so native ascending `list()` order equals +// arrival order — a small, bounded page is enough to read in order. + +/** How many rows one `list()` page fetches while accumulating a batch — + * small so a single call's OWN memory footprint stays bounded (rows are + * ≤ `INBOX_ROW_MAX_BYTES`, so one page is ≤ ~8 MB). */ +export const ROW_LIST_PAGE_LIMIT = 8; +/** Target bytes per `collectRowBatch` call — matches + * {@link PACK_OBJECT_MAX_DECOMPRESSED_BYTES} so one batch is normally + * enough to fill one packed object, without page-boundary fragmentation. */ +const ROW_BATCH_TARGET_BYTES = PACK_OBJECT_MAX_DECOMPRESSED_BYTES; + +function rowByteSize(row: PendingRow): number { + return new TextEncoder().encode(encodeNdjson(row.resourceLogs)).length; +} + +/** Bounded, arrival-ordered batch of pending rows (any tenant mixed in — + * the caller groups by tenant): pages through `row:` in small chunks + * ({@link ROW_LIST_PAGE_LIMIT}) via an exclusive `start` cursor, + * accumulating until {@link ROW_BATCH_TARGET_BYTES} or nothing remains. + * Always takes at least one row, matching `packTenant`'s own rule. */ +export async function collectRowBatch(storage: StorageLike): Promise<[string, PendingRow][]> { + const rows: [string, PendingRow][] = []; + let bytes = 0; + let start: string | undefined; + for (;;) { + const page = await storage.list({ prefix: ROW_PREFIX, start, limit: ROW_LIST_PAGE_LIMIT }); + if (page.size === 0) break; + let lastKey: string | undefined; + let hitBudget = false; + for (const [key, row] of page) { + lastKey = key; + const size = rowByteSize(row); + if (rows.length > 0 && bytes + size > ROW_BATCH_TARGET_BYTES) { + hitBudget = true; + break; + } + rows.push([key, row]); + bytes += size; + } + if (hitBudget) break; + if (page.size < ROW_LIST_PAGE_LIMIT) break; // nothing pending beyond this page + start = `${lastKey}\0`; // exclusive: the next page starts strictly after lastKey + } + return rows; +} diff --git a/runner/workers/o11y/src/inbox/registry.ts b/runner/workers/o11y/src/inbox/registry.ts new file mode 100644 index 0000000000..2bf3afcdb3 --- /dev/null +++ b/runner/workers/o11y/src/inbox/registry.ts @@ -0,0 +1,105 @@ +// ADR §F.3 / contract §7, §8: the exact first-seen registry for error +// fingerprints — `fp:` is written once, on first sight, and +// never overwritten. `pruneFingerprintRegistry` is a bounded, +// cursor-paginated TTL sweep (never a full-prefix scan) called from +// `writer.ts#backlog()` alongside `ledger.ts#pruneLedger`. A TTL, not an +// LRU: the alert lookback only ever looks back to the last notified time. + +import { fingerprintStorageKey, fingerprintTimeIndexKey, FPTS_TIMESTAMP_DIGITS } from "@handsontable/demo-runtime/telemetry"; +import { deleteChunked, getManyChunked, type StorageLike } from "./storage.js"; + +const FP_PREFIX = "fp:"; +const FPTS_PREFIX = "fpts:"; +/** `fpts:` + the zero-padded ms + the separating `:` — everything after + * this offset in an `fpts:` key is the fingerprint verbatim (safe even + * when the fingerprint itself contains `:`). */ +const FPTS_FP_OFFSET = FPTS_PREFIX.length + FPTS_TIMESTAMP_DIGITS + 1; +/** Default TTL: long enough that a re-alert after pruning is rare and + * acceptable, while bounding registry growth to a multiple of daily + * fingerprint volume rather than all-time history. */ +export const FP_DEFAULT_TTL_MS = 90 * 24 * 60 * 60 * 1000; +/** Bounds one prune call's cost — see `dedupe.ts#HASH_PRUNE_BATCH_LIMIT` + * for the throughput arithmetic. */ +const FP_PRUNE_BATCH_LIMIT = 5000; +/** How many rows one prune call inspects (not necessarily deletes) — larger + * than the delete limit since most inspected rows, in steady state, are + * not stale. Bounds the call even when nothing is stale yet. */ +const FP_PRUNE_SCAN_LIMIT = 20000; + +/** Returns `fp:` → `nowMs` (plus its `fpts:` time-index twin) for + * every fingerprint not already present — commit these in the same + * transaction as the dedupe/row writes. */ +export async function newFingerprintWrites( + storage: StorageLike, + fingerprints: readonly string[], + nowMs: number, +): Promise> { + const unique = [...new Set(fingerprints)]; + if (unique.length === 0) return {}; + // A 200-item Faro batch (the ingest cap) can carry up to 200 unique + // fingerprints — over the real DO storage 128-key limit. + const existing = await getManyChunked(storage, unique.map(fingerprintStorageKey)); + const writes: Record = {}; + for (const fp of unique) { + const key = fingerprintStorageKey(fp); + if (!existing.has(key)) { + writes[key] = nowMs; + writes[fingerprintTimeIndexKey(nowMs, fp)] = nowMs; + } + } + return writes; +} + +/** `fpts:` prefix and per-key parsing, exported for `alerts/inbox-state.ts#newFingerprintsSince` + * (its own bounded, time-ordered read of this index — kept here so the + * `fpts:` key shape has exactly one owner). */ +export { FPTS_PREFIX }; +export function fpFromFptsKey(key: string): string { + return key.slice(FPTS_FP_OFFSET); +} + +export interface FingerprintPruneResult { + fpDeleted: number; + /** A stored cursor makes forward progress across the whole keyspace over + * successive calls (fingerprints don't embed a date, unlike `hash:`). + * `null` once a full lap completes with nothing left. */ + nextCursor: string | null; +} + +/** Bounded TTL sweep: reads at most `FP_PRUNE_SCAN_LIMIT` rows starting + * after `cursor` (wrapping to the beginning once the end of the keyspace is + * reached), deletes at most `FP_PRUNE_BATCH_LIMIT` of the ones older than + * `ttlMs`, and returns where the next call should resume. Never a + * `list({prefix: "fp:"})` with no bound. */ +export async function pruneFingerprintRegistry( + storage: StorageLike, + nowMs: number, + ttlMs: number = FP_DEFAULT_TTL_MS, + cursor: string | null = null, +): Promise { + const end = `${FP_PREFIX}￿`; // exclusive upper bound past every possible fp: key + const start = cursor ?? FP_PREFIX; + const page = await storage.list({ start, end, limit: FP_PRUNE_SCAN_LIMIT }); + + const toDelete: string[] = []; + let lastKey: string | undefined; + for (const [key, firstSeenMs] of page) { + lastKey = key; + if (toDelete.length < FP_PRUNE_BATCH_LIMIT && typeof firstSeenMs === "number" && nowMs - firstSeenMs > ttlMs) { + toDelete.push(key); + // Delete the `fpts:` time-index twin alongside its `fp:` entry — the + // exact firstSeenMs is right here as this row's own value, so the + // twin's key is reconstructible without a second read. + toDelete.push(fingerprintTimeIndexKey(firstSeenMs, key.slice(FP_PREFIX.length))); + } + } + if (toDelete.length > 0) await deleteChunked(storage, toDelete); + + // Reached the end of the keyspace (fewer rows than the scan limit came + // back) — resume from the beginning, so a quiet tail never starves the + // front. + const reachedEnd = page.size < FP_PRUNE_SCAN_LIMIT; + const nextCursor = reachedEnd ? null : lastKey ? `${lastKey}\0` : null; + + return { fpDeleted: toDelete.length, nextCursor }; +} diff --git a/runner/workers/o11y/src/inbox/storage.ts b/runner/workers/o11y/src/inbox/storage.ts new file mode 100644 index 0000000000..de301dc4ee --- /dev/null +++ b/runner/workers/o11y/src/inbox/storage.ts @@ -0,0 +1,147 @@ +// The minimal storage surface `InboxWriter`'s pure logic modules need — a +// structural subset of `DurableObjectStorage`/`DurableObjectTransaction`, +// so a plain `Map`-backed fake can stand in under `node --test`. `get`/ +// `getMany` are split rather than mirroring one overloaded method: +// `writer.ts` adapts `this.ctx.storage` to this shape with a few lines of +// glue. + +/** `start`/`end`/`limit`: a real `DurableObjectStorage.list` already + * accepts these, for bounded-per-call pruning sweeps. `memoryStorage()` + * honours them, including `prefix` combined with `start`. */ +export interface ListOptions { + prefix?: string; + /** Inclusive: only keys `>= start`. */ + start?: string; + /** Exclusive: only keys `< end`. */ + end?: string; + limit?: number; +} + +export interface StorageLike { + get(key: string): Promise; + getMany(keys: string[]): Promise>; + put(entries: Record): Promise; + delete(keys: string[]): Promise; + /** Always returns entries in ascending key order (matches + * `DurableObjectStorage.list`'s default, `reverse` never requested + * here). */ + list(options?: ListOptions): Promise>; + transaction(closure: (txn: StorageLike) => Promise): Promise; + getAlarm(): Promise; + setAlarm(scheduledTime: number): Promise; +} + +/** Cloudflare's documented SQLite-backed-DO storage-API limit: get/put/ + * delete each support up to 128 keys at a time (fetched 2026-09-24). + * Local `workerd` was observed accepting 500+ keys with no error, so + * every multi-key call in this codebase must chunk through + * {@link getManyChunked}/{@link putChunked}/{@link deleteChunked}; + * `memoryStorage()` throws above this limit so a missed call site fails + * a test instead of silently working locally and throwing only in + * production. */ +export const DO_STORAGE_MAX_KEYS_PER_CALL = 128; + +function chunks(items: readonly T[], size: number): T[][] { + const out: T[][] = []; + for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size)); + return out; +} + +/** `storage.getMany(keys)`, chunked to {@link DO_STORAGE_MAX_KEYS_PER_CALL} + * per call. Safe inside a `transaction()` closure's own `txn`. */ +export async function getManyChunked(storage: StorageLike, keys: readonly string[]): Promise> { + const out = new Map(); + for (const chunk of chunks(keys, DO_STORAGE_MAX_KEYS_PER_CALL)) { + if (chunk.length === 0) continue; + const part = await storage.getMany(chunk); + for (const [k, v] of part) out.set(k, v); + } + return out; +} + +/** `storage.put(entries)`, chunked to {@link DO_STORAGE_MAX_KEYS_PER_CALL} + * per call. Inside a `transaction()`, every chunk still commits as one + * atomic transaction — chunking only splits the underlying `put` calls. */ +export async function putChunked(storage: StorageLike, entries: Record): Promise { + const keys = Object.keys(entries); + for (const chunk of chunks(keys, DO_STORAGE_MAX_KEYS_PER_CALL)) { + if (chunk.length === 0) continue; + const part: Record = {}; + for (const k of chunk) part[k] = entries[k] as T; + await storage.put(part); + } +} + +/** `storage.delete(keys)`, chunked to {@link DO_STORAGE_MAX_KEYS_PER_CALL} + * keys per call — see {@link putChunked}'s transaction-atomicity note, + * which applies identically here. */ +export async function deleteChunked(storage: StorageLike, keys: readonly string[]): Promise { + let deleted = 0; + for (const chunk of chunks(keys, DO_STORAGE_MAX_KEYS_PER_CALL)) { + if (chunk.length === 0) continue; + deleted += await storage.delete(chunk); + } + return deleted; +} + +/** A `Map`-backed {@link StorageLike} for `node --test`. `transaction()` is + * a no-op wrapper — a fresh `memoryStorage()` given to a second + * `InboxWriter` simulates a restart between an append and the alarm. */ +export function memoryStorage(): StorageLike { + const data = new Map(); + let alarm: number | null = null; + + const self: StorageLike = { + async get(key: string): Promise { + return data.get(key) as T | undefined; + }, + async getMany(keys: string[]): Promise> { + if (keys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`memoryStorage().getMany: ${keys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL} — use getManyChunked()`); + } + const out = new Map(); + for (const k of keys) if (data.has(k)) out.set(k, data.get(k) as T); + return out; + }, + async put(entries) { + const keys = Object.keys(entries); + if (keys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`memoryStorage().put: ${keys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL} — use putChunked()`); + } + for (const [k, v] of Object.entries(entries)) data.set(k, v); + }, + async delete(keys) { + if (keys.length > DO_STORAGE_MAX_KEYS_PER_CALL) { + throw new Error(`memoryStorage().delete: ${keys.length} keys exceeds the DO storage limit of ${DO_STORAGE_MAX_KEYS_PER_CALL} — use deleteChunked()`); + } + let n = 0; + for (const k of keys) if (data.delete(k)) n++; + return n; + }, + async list(options?: ListOptions) { + const matches: [string, T][] = []; + for (const [k, v] of data) { + if (options?.prefix && !k.startsWith(options.prefix)) continue; + if (options?.start !== undefined && k < options.start) continue; + if (options?.end !== undefined && k >= options.end) continue; + matches.push([k, v as T]); + } + // A real `DurableObjectStorage.list` always returns ascending key + // order; sort explicitly so the fake matches real range-delete/ + // pagination behaviour. + matches.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)); + const limited = options?.limit !== undefined ? matches.slice(0, options.limit) : matches; + return new Map(limited); + }, + async transaction(closure) { + return closure(self); + }, + async getAlarm() { + return alarm; + }, + async setAlarm(scheduledTime) { + alarm = scheduledTime; + }, + }; + return self; +} diff --git a/runner/workers/o11y/src/inbox/writer.ts b/runner/workers/o11y/src/inbox/writer.ts new file mode 100644 index 0000000000..557a299cac --- /dev/null +++ b/runner/workers/o11y/src/inbox/writer.ts @@ -0,0 +1,402 @@ +// `InboxWriter` — the one owner of the inbox lifecycle (ADR §B.2). A thin +// RPC shell: every real decision lives in the pure, storage-agnostic +// modules next to this file (`dedupe.ts`, `registry.ts`, `pack.ts`) so +// they are unit-testable over `storage.ts#memoryStorage()` without a real +// DO. + +import { DurableObject } from "cloudflare:workers"; +import { + DRAINS_PAUSED_STORAGE_KEY, + HEARTBEAT_STORAGE_KEY, + PACK_ALARM_INTERVAL_MS, + PACK_AT_BYTES, + toAePoint, + wakeStorageKey, + type AlertState, + type Heartbeat, + type NormalisedRecord, + type Tenant, + type WakeState, +} from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem, InboxWriterApi, IngestResult } from "../env.js"; +import { checkDuplicates } from "./dedupe.js"; +import { newFingerprintWrites } from "./registry.js"; +import { o11ySelfIdentity } from "../normalise/respond.js"; +import { writePointFromDo } from "../normalise/points.js"; +// Aliased to `ledger*`: every one of these names also names a class method +// below with the identical signature; aliasing removes any doubt about +// which one a reader means, even though the bare import always wins in JS. +import { + commitKeys as ledgerCommitKeys, + computeBacklog as ledgerComputeBacklog, + currentWakeId as ledgerCurrentWakeId, + markKeysProvisional as ledgerMarkKeysProvisional, + nextWrittenKeys as ledgerNextWrittenKeys, + pruneLedger, + recentRejectionCount as ledgerRecentRejectionCount, + recordPartialReject as ledgerRecordPartialReject, + recordWakeReady as ledgerRecordWakeReady, + rejectKey as ledgerRejectKey, + reopenWindow as ledgerReopenWindow, + reopenWindowExceedsRetention, + resolveOverWakes, + takeReopenedFlag as ledgerTakeReopenedFlag, + type InboxObjectInfo, +} from "./ledger.js"; +import { appendRows, collectRowBatch, commitPackedObject, packTenant, ROW_SEQ_STORAGE_KEY } from "./pack.js"; +import { putChunked } from "./storage.js"; +import { pruneHashBuckets } from "./dedupe.js"; +import { pruneFingerprintRegistry } from "./registry.js"; +import type { StorageLike } from "./storage.js"; +import { + backlogOldestAgeMs, + newFingerprintsAfterKey, + readAlertMeta, + readAlertState, + readDrainsPaused, + readHeartbeat, + rejectedKeyCount, + writeAlertMeta, + writeAlertState, + writeCronHeartbeat, + writeDrainsPaused, +} from "../alerts/inbox-state.js"; +import { getGrafanaBoxStub } from "../box.js"; + +const CLEAN_MARKER_PREFIX = "state/wakes/"; +/** Where `pruneStorage` persists `pruneFingerprintRegistry`'s resume cursor + * between cron ticks — not a contract-named key (internal housekeeping + * state only, like `pack.ts`'s own `rowSeq`). */ +const FP_PRUNE_CURSOR_STORAGE_KEY = "fpPruneCursor"; + +/** Paginates `O11Y_INBOX.list()` under `inbox/` into the shape `ledger.ts` + * needs — cheaper than a `.head()` per key, which would cost one + * subrequest per backlog key on every cron tick. */ +async function listInboxObjects(bucket: R2Bucket): Promise { + const out: InboxObjectInfo[] = []; + let cursor: string | undefined; + for (;;) { + const page = await bucket.list({ prefix: "inbox/", cursor, limit: 1000 }); + for (const obj of page.objects) out.push({ key: obj.key, size: obj.size, uploaded: obj.uploaded }); + if (!page.truncated) break; + cursor = page.cursor; + } + return out; +} + +/** `this.ctx.storage` adapted to {@link StorageLike}. Also used, + * recursively, to adapt a real `transaction()` call's own `txn` — a + * `DurableObjectTransaction` has `get`/`getMany`/`put`/`delete`/`list` but + * not `transaction`/`getAlarm`/`setAlarm`; every call site in this file + * only uses the shared subset on a nested `StorageLike`. */ +function adaptStorage(storage: DurableObjectStorage): StorageLike { + return { + get: (key: string) => storage.get(key), + getMany: (keys: string[]) => storage.get(keys), + put: (entries) => storage.put(entries), + delete: (keys) => storage.delete(keys), + list: (options) => storage.list(options), + transaction: (closure) => storage.transaction((txn) => closure(adaptStorage(txn as unknown as DurableObjectStorage))), + getAlarm: () => storage.getAlarm(), + setAlarm: (t) => storage.setAlarm(t), + }; +} + +export class InboxWriter extends DurableObject implements InboxWriterApi { + async recordWake(wakeId: string, reason: "backlog" | "visit"): Promise { + const storage = adaptStorage(this.ctx.storage); + await storage.transaction(async (txn) => { + const existingWakes = await txn.list({ prefix: "wake:" }); + const writes: Record = {}; + for (const [key, wake] of existingWakes) { + if (!wake.over) writes[key] = { ...wake, over: true }; + } + writes[wakeStorageKey(wakeId)] = { startedAt: Date.now(), reason, over: false }; + await txn.put(writes); + }); + } + + async recordWakeReady(wakeId: string, readyMs: number): Promise { + await ledgerRecordWakeReady(adaptStorage(this.ctx.storage), wakeId, readyMs); + } + + async ingest(tenant: Tenant, arrivalMs: number, items: IngestItem[]): Promise { + const storage = adaptStorage(this.ctx.storage); + + const result = await storage.transaction(async (txn) => { + const hashes = items.map((i) => i.hash); + const dedupe = await checkDuplicates(txn, hashes, arrivalMs); + // Filter by OCCURRENCE (index), never by hash — the first copy of an + // in-batch repeat is the one stored, later copies are duplicates. See + // `DedupeResult.isDuplicate`. + const accepted = items.filter((_, idx) => !dedupe.isDuplicate[idx]); + + // A `record`-less item still goes through the dedupe/fingerprint + // bookkeeping above but must never produce a `row:` entry (§6: AE + // points only, never stored). + const append = await appendRows( + txn, + tenant, + arrivalMs, + accepted.map((i) => i.record).filter((r): r is NormalisedRecord => r !== undefined), + ); + + const fingerprints = accepted.map((i) => i.fingerprint).filter((fp): fp is string => Boolean(fp)); + const fpWrites = await newFingerprintWrites(txn, fingerprints, arrivalMs); + + const heartbeat = (await txn.get(HEARTBEAT_STORAGE_KEY)) ?? { lastCron: 0, lastIngest: 0 }; + + // A 200-item Faro batch (the per-request cap) can produce up to 200 + // dedupe.writes + 200*2 fpWrites (fp:/fpts: pairs) entries in one + // call — well over the real DO storage 128-key put() limit. + await putChunked(txn, { + ...dedupe.writes, + ...append.writes, + [ROW_SEQ_STORAGE_KEY]: append.nextRowSeq, + ...fpWrites, + [HEARTBEAT_STORAGE_KEY]: { ...heartbeat, lastIngest: arrivalMs } satisfies Heartbeat, + }); + + return { + results: items.map((i, idx) => ({ + hash: i.hash, + outcome: (dedupe.isDuplicate[idx] ? "duplicate" : "accepted") as "duplicate" | "accepted", + })), + bytesAdded: append.bytesAdded, + }; + }); + + // Alarm scheduling is outside the transaction — `getAlarm`/`setAlarm` + // are not part of `DurableObjectTransaction`'s surface. + await this.schedulePackAlarm(storage, result.bytesAdded); + + return { results: result.results }; + } + + // ---- Ledger, backlog, drain support ------------------------------------- + + async resolveWakes(): Promise { + const storage = adaptStorage(this.ctx.storage); + const { resolved } = await resolveOverWakes(storage, { + isBoxRunning: () => getGrafanaBoxStub(this.env).isAwake(), + markerExists: async (wakeId) => { + const head = await this.env.O11Y_LOKI_STATE.head(`${CLEAN_MARKER_PREFIX}${wakeId}/clean`); + return head !== null; + }, + }); + // Contract §5's `o11y.wake` point — written here since only the + // ledger learns whether a wake's stop was clean. `recordWakeReady` + // carries the wake-to-ready time here on both the clean and unclean + // path; a wake whose box never became ready writes 0 deliberately. + for (const w of resolved) { + writePointFromDo( + this.env, + this.ctx, + toAePoint( + "o11y.wake", + { count: 1, duration_ms: w.readyMs ?? 0 }, + { ...o11ySelfIdentity(this.env), reason: w.reason, outcome: w.clean ? "clean" : "unclean" }, + ), + ); + } + } + + async backlog(): Promise<{ oldestWrittenAgeMs: number; totalBytes: number; writtenCount: number; drainsPaused: boolean }> { + await this.resolveWakes(); + const storage = adaptStorage(this.ctx.storage); + // Housekeeping runs from the cron path, never the pack alarm (which + // only fires on ingest and would starve pruning during a quiet + // period) — see `pruneStorage`. + await this.pruneStorage(storage); + const drainsPaused = (await storage.get(DRAINS_PAUSED_STORAGE_KEY)) ?? false; + return ledgerComputeBacklog(storage, () => listInboxObjects(this.env.O11Y_INBOX), drainsPaused); + } + + /** Bounded housekeeping for the three storage prefixes that would + * otherwise never be deleted. Each sweep is independently try/caught. */ + private async pruneStorage(storage: StorageLike): Promise { + const nowMs = Date.now(); + try { + await pruneLedger(storage, nowMs); + } catch (err) { + console.error(JSON.stringify({ event: "o11y.prune.error", target: "ledger", message: String(err) })); + } + try { + await pruneHashBuckets(storage, nowMs); + } catch (err) { + console.error(JSON.stringify({ event: "o11y.prune.error", target: "hash", message: String(err) })); + } + try { + const cursor = (await storage.get(FP_PRUNE_CURSOR_STORAGE_KEY)) ?? null; + const result = await pruneFingerprintRegistry(storage, nowMs, undefined, cursor); + await storage.put({ [FP_PRUNE_CURSOR_STORAGE_KEY]: result.nextCursor }); + } catch (err) { + console.error(JSON.stringify({ event: "o11y.prune.error", target: "fingerprint", message: String(err) })); + } + } + + async nextWrittenKeys(limit: number, excludeTenants: Tenant[] = []): Promise { + return ledgerNextWrittenKeys(adaptStorage(this.ctx.storage), limit, excludeTenants); + } + + // Lets `box.ts#drainStepBody` know whether the batch it just pushed + // replayed any reopened keys, so its `o11y.drain` point can emit + // `reason: "reopen"`. + async takeReopenedFlag(inboxKeys: string[]): Promise { + return ledgerTakeReopenedFlag(adaptStorage(this.ctx.storage), inboxKeys); + } + + async markKeysProvisional(wakeId: string, keys: string[]): Promise { + await ledgerMarkKeysProvisional(adaptStorage(this.ctx.storage), wakeId, keys); + } + + /** See `ledger.ts#commitKeys` — a zero-bytes-pushed key commits directly, + * no wake/marker involved. */ + async commitKeys(keys: string[]): Promise { + await ledgerCommitKeys(adaptStorage(this.ctx.storage), keys); + } + + async rejectKey(key: string, reason: string): Promise { + await ledgerRejectKey(adaptStorage(this.ctx.storage), key, reason); + } + + /** Logs a rejection EVENT for a key whose ledger state stays + * `provisional`/`done:` — see `ledger.ts#recordPartialReject`. */ + async recordPartialReject(key: string, reason: string): Promise { + await ledgerRecordPartialReject(adaptStorage(this.ctx.storage), key, reason); + } + + /** Count of `rejectedEvent:` entries newer than `sinceMs` — the + * rejected-key alert resolves once rejections stop. */ + async recentRejectionCount(sinceMs: number): Promise { + return ledgerRecentRejectionCount(adaptStorage(this.ctx.storage), sinceMs); + } + + /** A window wider than {@link KEY_RETENTION_MS} is refused outright — + * nothing that old can still exist, so a wider request would otherwise + * scan for nothing while paying the full read cost. */ + async reopenWindow(fromMs: number, toMs: number): Promise<{ reopened: number }> { + if (reopenWindowExceedsRetention(fromMs, toMs)) { + throw new Error("reopenWindow: window exceeds the 7-day retention cap"); + } + const storage = adaptStorage(this.ctx.storage); + const active = await ledgerCurrentWakeId(storage); + return ledgerReopenWindow(storage, fromMs, toMs, active); + } + + async currentWakeId(): Promise { + return ledgerCurrentWakeId(adaptStorage(this.ctx.storage)); + } + + /** ADR §B.2 step 6: pack on a 60 s alarm, or immediately once a single + * ingest call crosses `PACK_AT_BYTES` — an approximation of "4 MB + * stored" as "4 MB in one request," not a running total; the 60 s alarm + * always catches the remainder. Never moves an alarm earlier than one + * already scheduled. */ + private async schedulePackAlarm(storage: StorageLike, bytesAdded: number): Promise { + const existing = await storage.getAlarm(); + if (bytesAdded >= PACK_AT_BYTES) { + await storage.setAlarm(Date.now()); + return; + } + if (existing === null) { + await storage.setAlarm(Date.now() + PACK_ALARM_INTERVAL_MS); + } + } + + /** ADR §B.2 step 6: gzipped NDJSON objects per tenant, committed + * atomically per object (`pack.ts#commitPackedObject`). Reads are + * bounded — `collectRowBatch` pages `row:` in small chunks up to one + * packed object's byte budget — and looped until nothing remains or + * `MAX_OBJECTS_PER_ALARM` objects have packed, rescheduling immediately + * when objects remain, so a sustained flood cannot grow past the DO's + * memory limit within one invocation. */ + async alarm(): Promise { + const MAX_OBJECTS_PER_ALARM = 25; + const storage = adaptStorage(this.ctx.storage); + + let packedCount = 0; + let more = false; + for (;;) { + if (packedCount >= MAX_OBJECTS_PER_ALARM) { + more = true; + break; + } + const batch = await collectRowBatch(storage); // bounded — see this method's own doc comment + if (batch.length === 0) break; + + const byTenant = new Map(); + for (const entry of batch) { + const list = byTenant.get(entry[1].tenant) ?? []; + list.push(entry); + byTenant.set(entry[1].tenant, list); + } + + let packedAny = false; + for (const [tenant, rows] of byTenant) { + if (packedCount >= MAX_OBJECTS_PER_ALARM) { + more = true; + break; + } + const packed = await packTenant(storage, this.env.O11Y_INBOX, tenant, rows); + if (!packed) continue; + await commitPackedObject(storage, packed); + packedCount++; + packedAny = true; + } + if (more) break; + if (!packedAny) break; // safety valve: nothing consumable in this batch + } + if (more) await storage.setAlarm(Date.now()); + } + + // ---- Alerts, watchdog, the o11y spend cap ------------------------------- + // Thin RPC shells only — every real rule lives in `alerts/inbox-state.ts`. + + async heartbeat(): Promise { + return readHeartbeat(adaptStorage(this.ctx.storage)); + } + + async stampCronHeartbeat(nowMs: number): Promise { + await writeCronHeartbeat(adaptStorage(this.ctx.storage), nowMs); + } + + async backlogOldestAgeMs(): Promise { + return backlogOldestAgeMs(adaptStorage(this.ctx.storage)); + } + + async rejectedKeyCount(): Promise { + return rejectedKeyCount(adaptStorage(this.ctx.storage)); + } + + async newFingerprintsAfterKey( + afterKey: string | null, + fallbackSinceMs: number, + ): Promise<{ entries: { key: string; name: string; firstSeenMs: number }[]; truncated: boolean }> { + return newFingerprintsAfterKey(adaptStorage(this.ctx.storage), afterKey, fallbackSinceMs); + } + + async alertState(rule: string): Promise { + return readAlertState(adaptStorage(this.ctx.storage), rule); + } + + async setAlertState(rule: string, state: AlertState): Promise { + await writeAlertState(adaptStorage(this.ctx.storage), rule, state); + } + + async getAlertMeta(key: string): Promise { + return readAlertMeta(adaptStorage(this.ctx.storage), key); + } + + async setAlertMeta(key: string, value: string): Promise { + await writeAlertMeta(adaptStorage(this.ctx.storage), key, value); + } + + async drainsPaused(): Promise { + return readDrainsPaused(adaptStorage(this.ctx.storage)); + } + + async setDrainsPaused(paused: boolean): Promise { + await writeDrainsPaused(adaptStorage(this.ctx.storage), paused); + } +} diff --git a/runner/workers/o11y/src/index.ts b/runner/workers/o11y/src/index.ts new file mode 100644 index 0000000000..a7e505452c --- /dev/null +++ b/runner/workers/o11y/src/index.ts @@ -0,0 +1,317 @@ +// The o11y worker's entry point (observability contract §1): real routing +// through `router.ts` for `POST /telemetry/collect|v1/logs|deploy|hooks/ +// sentry`, plus `/grafana/*` and `POST /grafana/_o11y/reopen` (below). +// `POST /telemetry/lite` registers itself via `./lite.ts`'s side-effect +// import. Durable Object classes are exported from here (Workers requires +// it) but defined in `box.ts`/`inbox/writer.ts`. + +import { toAePoint } from "@handsontable/demo-runtime/telemetry"; +import type { Env } from "./env.js"; +import "./lite.js"; +import { checkBrowserGates, checkPayloadEnvironment } from "./gates/browser.js"; +import { checkDeployGate } from "./gates/oidc.js"; +import { checkSentryHmac } from "./gates/sentry.js"; +import { checkExportSecret } from "./gates/secret.js"; +import { COLLECT_MAX_BYTES, OTLP_MAX_BYTES, SMALL_JSON_MAX_BYTES } from "./gates/limits.js"; +import { inboxWriter } from "./inbox/accessor.js"; +import { isDeployPayload, processDeployPayload } from "./normalise/deploy.js"; +import { countFaroItems, MAX_FARO_ITEMS_PER_BODY, processFaroBody } from "./normalise/faro.js"; +import { processOtlpBody } from "./normalise/otlp.js"; +import { processSentryPayload } from "./normalise/sentry.js"; +import { BodyTooLargeError, readCappedBytes, readCappedText } from "./normalise/read-body.js"; +import { recordInvalidItem, recordOversizeDrop, respondDrop, respondIngested, o11ySelfIdentity } from "./normalise/respond.js"; +import { writePoint } from "./normalise/points.js"; +import { findRoute, registerRoute } from "./router.js"; +import { runAlerts } from "./alerts/index.js"; +import { getGrafanaBoxStub } from "./box.js"; +import { handleGrafana } from "./grafana/proxy.js"; +import { handleReopen } from "./grafana/reopen.js"; +import { handleCallback, handleLogin, handleLogout, handleLogoutPage, handleSession } from "./grafana/login.js"; + +export { GrafanaBox } from "./box.js"; +export { InboxWriter } from "./inbox/writer.js"; +export { O11yHeartbeat } from "./heartbeat.js"; + +// ---- POST /telemetry/collect — Faro payloads from the authoring app ------------ + +async function handleCollect(req: Request, env: Env, ctx: ExecutionContext): Promise { + const gate = await checkBrowserGates(req, env, COLLECT_MAX_BYTES); + if (!gate.ok) return respondDrop(env, ctx, gate); + + let bytes: Uint8Array; + try { + bytes = await readCappedBytes(req, COLLECT_MAX_BYTES); + } catch (err) { + if (err instanceof BodyTooLargeError) return respondDrop(env, ctx, { ok: false, reason: "size", status: 413 }); + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + let body: unknown; + try { + body = JSON.parse(new TextDecoder().decode(bytes)); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + const declaredEnv = (body as { meta?: { app?: { environment?: string } } })?.meta?.app?.environment; + const envGate = checkPayloadEnvironment(declaredEnv, env); + if (!envGate.ok) return respondDrop(env, ctx, envGate); + + // Bound the whole batch before doing any real work — a real Faro batch + // never approaches this many items (SDK limit 50); unbounded, one 1 MB + // body inflated to ~16.7k stored records and AE points. + if (countFaroItems(body) > MAX_FARO_ITEMS_PER_BODY) { + return respondDrop(env, ctx, { ok: false, reason: "too_many_items", status: 400 }); + } + + const receivedAtMs = Date.now(); + const rawVersion = (body as { meta?: { app?: { version?: string } } })?.meta?.app?.version; + const service = { + name: "demos-authoring" as const, + // `meta.app.version` is client-supplied and unbounded — it becomes + // `service.version`, an AE blob, so it must be capped or it can push a + // point over Analytics Engine's per-point size limit. + version: typeof rawVersion === "string" && rawVersion.length > 0 ? rawVersion.slice(0, 64) : "unknown", + environment: env.O11Y_ENV, + }; + + let accepted = 0; + let duplicate = 0; + try { + const processed = await processFaroBody(body, env, service, receivedAtMs); + + // `withItem[i].ingestItem` is `ingestItems[i]`, and `IngestResult.results` + // is index-aligned with `ingestItems` (see below). + const withItem = processed.filter((p) => p.ingestItem); + const ingestItems = withItem.map((p) => p.ingestItem!); + + for (const p of processed) { + if (p.invalid) recordInvalidItem(env, ctx, p.invalid); + if (p.oversize) recordOversizeDrop(env, ctx, "Faro record exceeds 256 KB"); + // An item with no `ingestItem` writes its points unconditionally (it + // was already fully handled above). An `example.*` event carries a + // hash-only `ingestItem` so it goes through the dedupe transaction + // too, gated below on the actual outcome — no kind can double-count. + if (!p.ingestItem) { + for (const point of p.aePoints) writePoint(env, ctx, point); + } + } + + if (ingestItems.length > 0) { + const result = await inboxWriter(env).ingest("browser", receivedAtMs, ingestItems); + // Outcomes are matched to items BY INDEX, never by hash — two + // identical items can share a hash but get different outcomes. Every + // hash `InboxWriter.ingest` reports counts toward this route's + // `o11y.ingest` self-metric, whether or not it carries a stored + // `record`: an AE-only item is still a record the pipeline accepted. + withItem.forEach((p, idx) => { + const outcome = result.results[idx]?.outcome; + if (outcome === undefined) return; + outcome === "duplicate" ? duplicate++ : accepted++; + if (outcome === "accepted") { + for (const point of p.aePoints) writePoint(env, ctx, point); + } + }); + } + } catch (err) { + // `handleCollect`'s own backstop: every known throw site is fixed at + // its root, but a still-unknown shape here degrades to one accounted + // drop, never an unhandled exception. + console.warn("[o11y] handleCollect failed:", err instanceof Error ? err.message : String(err)); + recordInvalidItem(env, ctx, "handleCollect: unhandled batch failure"); + // Reaching this catch means nothing in this batch committed — + // `accepted`/`duplicate` stay zero. Answering 2xx here would claim a + // commit that never happened (ADR §B.2), and since Faro only retries + // on non-2xx, would silently drop the batch instead. A batch that + // legitimately commits nothing (all duplicates) never throws, so it + // still reaches the ordinary 204 below. + return new Response(JSON.stringify({ error: "unhandled_batch_failure" }), { + status: 500, + headers: { "content-type": "application/json" }, + }); + } + + return respondIngested(env, ctx, "collect", { accepted, duplicate }, bytes.byteLength); +} + +// ---- POST /telemetry/v1/logs — Cloudflare OTLP log export --------------------- + +async function handleOtlpLogs(req: Request, env: Env, ctx: ExecutionContext): Promise { + const secretGate = checkExportSecret(req, env); + if (!secretGate.ok) return respondDrop(env, ctx, secretGate); + + let bytes: Uint8Array; + try { + bytes = await readCappedBytes(req, OTLP_MAX_BYTES); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "size", status: 413 }); + } + + const contentType = req.headers.get("content-type") ?? "application/json"; + const receivedAtMs = Date.now(); + + let processed; + try { + processed = await processOtlpBody(bytes, contentType, env, receivedAtMs); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + let accepted = 0; + let duplicate = 0; + if (processed.droppedOversize > 0) { + for (let i = 0; i < processed.droppedOversize; i++) { + recordOversizeDrop(env, ctx, "OTLP record exceeds 256 KB"); + } + } + if (processed.items.length > 0) { + const result = await inboxWriter(env).ingest("worker", receivedAtMs, processed.items); + for (const r of result.results) r.outcome === "duplicate" ? duplicate++ : accepted++; + } + + return respondIngested(env, ctx, "v1/logs", { accepted, duplicate }, bytes.byteLength); +} + +// ---- POST /telemetry/deploy — CI deploy events --------------------------------- + +async function handleDeploy(req: Request, env: Env, ctx: ExecutionContext): Promise { + const gate = await checkDeployGate(req, env); + if (!gate.ok) return respondDrop(env, ctx, gate); + + let text: string; + try { + text = await readCappedText(req, SMALL_JSON_MAX_BYTES); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "size", status: 413 }); + } + + let body: unknown; + try { + body = JSON.parse(text); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + if (!isDeployPayload(body)) return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + + const receivedAtMs = Date.now(); + const item = await processDeployPayload(body, env, receivedAtMs); + const result = await inboxWriter(env).ingest("worker", receivedAtMs, [item]); + const accepted = result.results.filter((r) => r.outcome === "accepted").length; + const duplicate = result.results.filter((r) => r.outcome === "duplicate").length; + + return respondIngested(env, ctx, "deploy", { accepted, duplicate }, text.length); +} + +// ---- POST /telemetry/hooks/sentry — Sentry issue-alert webhook ---------------- + +async function handleSentryHook(req: Request, env: Env, ctx: ExecutionContext): Promise { + let text: string; + try { + text = await readCappedText(req, SMALL_JSON_MAX_BYTES); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "size", status: 413 }); + } + + const gate = await checkSentryHmac(req, env, text); + if (!gate.ok) return respondDrop(env, ctx, gate); + + let body: unknown; + try { + body = JSON.parse(text); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + const receivedAtMs = Date.now(); + const item = await processSentryPayload(body, env, receivedAtMs, req.headers.get("sentry-hook-timestamp")); + const result = await inboxWriter(env).ingest("worker", receivedAtMs, [item]); + const accepted = result.results.filter((r) => r.outcome === "accepted").length; + const duplicate = result.results.filter((r) => r.outcome === "duplicate").length; + + return respondIngested(env, ctx, "hooks/sentry", { accepted, duplicate }, text.length); +} + +registerRoute("POST", "/telemetry/collect", handleCollect); +registerRoute("POST", "/telemetry/v1/logs", handleOtlpLogs); +registerRoute("POST", "/telemetry/deploy", handleDeploy); +registerRoute("POST", "/telemetry/hooks/sentry", handleSentryHook); +// `"*"`, not `"GET"` — Grafana's frontend also queries via POST under +// `/grafana/*`. `POST /grafana/_o11y/reopen` is an exact route, which +// `router.ts`'s precedence (exact beats prefix) always wins over this +// catch-all. The five login/session/logout routes below are also exact +// routes for the same reason, and none of them ever wakes the box. +registerRoute("GET", "/grafana/_o11y/login", handleLogin); +registerRoute("GET", "/grafana/_o11y/callback", handleCallback); +registerRoute("POST", "/grafana/_o11y/session", handleSession); +registerRoute("GET", "/grafana/_o11y/logout", handleLogoutPage); +registerRoute("POST", "/grafana/_o11y/logout", handleLogout); +registerRoute("*", "/grafana/*", handleGrafana); +registerRoute("POST", "/grafana/_o11y/reopen", handleReopen); + +/** ADR §A/§B.1's ten-minute cron (`wrangler.jsonc`'s `triggers.crons`): + * reads the backlog (which resolves over-wakes as a side effect, ADR + * §B.3), writes the `o11y.backlog` self-metric, and wakes the box when the + * backlog is old or large enough — never while `drainsPaused` (the cost + * cap; this cron only reads the flag, never writes it). The alert + * evaluation cron shares the same `scheduled` handler rather than adding a + * second export (Workers allows only one). */ +async function handleScheduled(env: Env, ctx: ExecutionContext): Promise { + const writer = inboxWriter(env); + const backlog = await writer.backlog(); + + writePoint( + env, + ctx, + toAePoint( + "o11y.backlog", + { value: backlog.oldestWrittenAgeMs / 1000, bytes: backlog.totalBytes }, + o11ySelfIdentity(env), + ), + ); + + // ADR §A/§G: "never when drainsPaused" — reads the same flag + // `alerts/index.ts#canWakeForBacklog` exposes, inline rather than a + // second RPC round trip. A Grafana VISIT wake never reads this flag; the + // box still serves Grafana, and `box.ts#drainStep` refuses to drain. + if (backlog.drainsPaused) return; + + const oneHourMs = 60 * 60 * 1000; + const sixtyFourMb = 64 * 1024 * 1024; + if (backlog.oldestWrittenAgeMs <= oneHourMs && backlog.totalBytes <= sixtyFourMb) return; + + try { + await getGrafanaBoxStub(env).wake("backlog"); + } catch (err) { + // A wake failure (e.g. `recordWake` throwing) is retried by the very + // next tick — nothing here needs to escalate. + console.warn("[o11y] cron wake failed:", err instanceof Error ? err.message : String(err)); + } +} + +export default { + async fetch(request: Request, env: Env, ctx: ExecutionContext): Promise { + const url = new URL(request.url); + + const handler = findRoute(request.method, url.pathname); + if (handler) return handler(request, env, ctx); + + return new Response("Not Found", { status: 404 }); + }, + + // `handleScheduled` is the one real `scheduled` export (Workers allows + // only one). This tick does three independent things: stamps + // `heartbeat.lastCron`, runs the backlog scan/wake, and evaluates every + // alert. The backlog wake runs AFTER the alerts — `runAlerts` is what + // sets `drainsPaused` from this tick's spend-cap result, so running them + // side by side could let the crossing tick still wake and drain. + async scheduled(_controller: ScheduledController, env: Env, ctx: ExecutionContext): Promise { + ctx.waitUntil(inboxWriter(env).stampCronHeartbeat(Date.now())); + ctx.waitUntil( + runAlerts(env, ctx) + .catch((err) => { + console.error(JSON.stringify({ event: "o11y.cron.alerts_failed", message: String(err) })); + }) + .then(() => handleScheduled(env, ctx)), + ); + }, +} satisfies ExportedHandler; diff --git a/runner/workers/o11y/src/lite.ts b/runner/workers/o11y/src/lite.ts new file mode 100644 index 0000000000..b5a9236918 --- /dev/null +++ b/runner/workers/o11y/src/lite.ts @@ -0,0 +1,195 @@ +// `POST /telemetry/lite` — the §9 lite beacon from `/d` and `/embed`'s +// standalone reporter (ADR §C.5). Reuses the ingest machinery end-to-end: +// the browser gate (`checkBrowserGates`), the capped body reader, the +// shared converter (beacon order: `beaconToRecord` → `scrubTelemetry`, the +// *reverse* of the Faro path), `withResourceAttrDefaults`, `hashRecord`, +// `InboxWriter.ingest` on the `browser` tenant, and `respond.ts`'s single +// `o11y.ingest` point per outcome. +// +// One request is always exactly one beacon (§9's payload has no batch shape), +// unlike `/telemetry/collect`'s `TransportBody` array — so this route has no +// per-item loop and writes at most one `error.uncaught`/`web_vital` point. + +import { + beaconToRecord, + feedsNewFingerprintAlert, + fingerprint, + INBOX_RECORD_MAX_BYTES, + isValidLitePayload, + LITE_PAYLOAD_MAX_BYTES, + scrubTelemetry, + toAePoint, + type AePoint, + type LiteBeaconPayload, + type ServiceIdentity, + type Surface, +} from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem } from "./env.js"; +import { checkBrowserGates } from "./gates/browser.js"; +import { inboxWriter } from "./inbox/accessor.js"; +import { BodyTooLargeError, readCappedBytes } from "./normalise/read-body.js"; +import { scrubBodyText } from "./normalise/text-scrub.js"; +import { hashRecord } from "./normalise/hash.js"; +import { withResourceAttrDefaults, writePoint } from "./normalise/points.js"; +import { recordOversizeDrop, respondDrop, respondIngested } from "./normalise/respond.js"; +import { registerRoute } from "./router.js"; + +/** §3: the lite beacon has no natural build identity (a client with no + * bundle of its own cannot report its own `service.version`) — + * `"unknown"` is `withResourceAttrDefaults`'s own fallback for exactly + * this case, restated explicitly here since this is the only record + * shape built with this identity from the start. */ +function liteServiceIdentity(env: Env): ServiceIdentity { + return { name: "demos-embed", version: "unknown", environment: env.O11Y_ENV }; +} + +/** §7's fingerprint input for a lite `err` payload: the error's own name + * and (truncated, client-side) message — never the stack. + * `convert.ts#beaconBody`'s stored `body` includes the stack for a human + * reading Loki; folding it into the fingerprint too would mean a hashed + * chunk name or line number mints a "new" fingerprint on every rebuild + * that shifts one — and unlike `demo-runtime`, `d`/`embed` surfaces feed + * the new-fingerprint alert (`feedsNewFingerprintAlert`), so a false + * "new" here pages someone. */ +function liteErrorFingerprintMessage(payload: Extract): string { + return `${payload.n}: ${payload.m}`; +} + +async function handleLite(req: Request, env: Env, ctx: ExecutionContext): Promise { + // The contract's own payload cap doubles as this route's request-body cap + // (§9: "≤ 2 KB") — there is no larger "batch" shape to allow room for, the + // way `COLLECT_MAX_BYTES` allows for a whole Faro `TransportBody`. + const gate = await checkBrowserGates(req, env, LITE_PAYLOAD_MAX_BYTES); + if (!gate.ok) return respondDrop(env, ctx, gate); + + let bytes: Uint8Array; + try { + bytes = await readCappedBytes(req, LITE_PAYLOAD_MAX_BYTES); + } catch (err) { + if (err instanceof BodyTooLargeError) return respondDrop(env, ctx, { ok: false, reason: "size", status: 413 }); + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + let body: unknown; + try { + body = JSON.parse(new TextDecoder().decode(bytes)); + } catch { + return respondDrop(env, ctx, { ok: false, reason: "parse", status: 400 }); + } + + // The one place every field of an untrusted, client-crafted beacon is + // re-checked (`isValidLitePayload`): shape, per-field caps, and — + // decisively — the total serialized size again, so a payload that grew + // past 2 KB only once JSON-decoded is still caught. + if (!isValidLitePayload(body)) { + return respondDrop(env, ctx, { ok: false, reason: "invalid_item", status: 400 }); + } + + const receivedAtMs = Date.now(); + const service = liteServiceIdentity(env); + + // Beacon order (the reverse of Faro's): convert first, scrub the built + // record second — `beaconToRecord`'s own fields (`m`, `st`) never passed + // through `scrubTelemetry` before this point. + const record = beaconToRecord(body, { service, receivedAtMs }); + // Never `null` here: only a Faro item (`isFaroItem`) can make `scrubTelemetry` + // drop the record entirely (a console item, §3); the OTLP-record branch + // always returns its scrubbed clone. + const scrubbed = scrubTelemetry(record)!; + // `scrubTelemetry` only strips a query string from a *discrete* + // URL-shaped field, never one embedded inside free body text — and a + // beacon's `st` (a stack) routinely carries a bundler's cache-busting + // `?t=`/`?v=` on a chunk URL. + scrubbed.body = scrubBodyText(scrubbed.body); + withResourceAttrDefaults(scrubbed.resourceAttributes, env); + + // A lite web-vital beacon (`t !== "err"`) is AE-only, contract §6/§9's + // own ruling (measurements never need a stored Loki record; only an + // error report does). Errors are unaffected: `body.t === "err"` still + // stores its record, symbolication and all. + const storeRecord = body.t === "err"; + + if ( + storeRecord && + new TextEncoder().encode(JSON.stringify(scrubbed)).length > INBOX_RECORD_MAX_BYTES + ) { + // Unreachable in practice — the whole request body is already capped at + // `LITE_PAYLOAD_MAX_BYTES` (2 KB), far under `INBOX_RECORD_MAX_BYTES` + // (256 KB) — kept for the same defence-in-depth reason the Faro and + // OTLP paths both carry this check. Skipped entirely for a vital: there + // is nothing to store for it regardless of size. + recordOversizeDrop(env, ctx, "lite beacon record exceeds 256 KB"); + return respondIngested(env, ctx, "lite", { accepted: 0, duplicate: 0 }, bytes.byteLength); + } + + const aePoints: AePoint[] = []; + const common = { service_name: service.name, service_version: service.version, environment: service.environment }; + let itemFingerprint: string | undefined; + + if (body.t === "err") { + const fp = fingerprint(body.s, liteErrorFingerprintMessage(body)); + itemFingerprint = feedsNewFingerprintAlert(body.s as Surface) ? fp : undefined; + aePoints.push( + toAePoint("error.uncaught", { count: 1 }, { ...common, surface: body.s, fingerprint: fp, demo_id: body.demo }), + ); + } else { + aePoints.push( + toAePoint( + "web_vital", + { value: body.val }, + { + ...common, + surface: body.s, + framework: body.fw, + ht_major: body.ht, + reason: body.n, + device: body.dev, + demo_id: body.demo, + }, + ), + ); + } + + const hash = await hashRecord({ + body: scrubbed.body, + resourceAttributes: scrubbed.resourceAttributes, + attributes: scrubbed.attributes ?? {}, + // The beacon's own raw, un-clamped event time (§8: "the record's own raw + // source timestamp, exactly as received") — `ts` is epoch ms, restated as + // a string, the same way OTLP's raw `time_unix_nano` is. + rawEventTime: String(body.ts), + // The reporter's per-beacon id, hash-only (never in the stored record, + // attributes or a Loki label — no cardinality). A conditional spread, + // never `extra: body.id !== undefined ? {...} : undefined` — + // `stableStringify` walks `Object.entries`, so a present `extra` key + // holding `undefined` would still serialize and change the hash for + // every old-reporter beacon that has no `id` at all. With the spread, + // an id-less beacon hashes exactly as it did before this field existed. + ...(body.id !== undefined ? { extra: { beacon_id: body.id } } : {}), + }); + + // Same pattern as `normalise/faro.ts`: a hash-only item with no `record` + // still gets a real dedupe transaction (`InboxWriter.ingest`/`appendRows` + // skip a `record`-less item entirely), so a retried/redelivered vital + // beacon still cannot double-count its `web_vital` point. + const item: IngestItem = storeRecord + ? { hash, record: scrubbed, fingerprint: itemFingerprint } + : { hash, fingerprint: itemFingerprint }; + const result = await inboxWriter(env).ingest("browser", receivedAtMs, [item]); + let accepted = 0; + let duplicate = 0; + for (const r of result.results) r.outcome === "duplicate" ? duplicate++ : accepted++; + + // Only write this route's own metric point when the record was actually + // a NEW record — an unconditional write would mean a duplicated beacon + // (a `sendBeacon` retry, a redelivered request) writes a second + // `error.uncaught`/`web_vital` point even while the matching + // `o11y.ingest` point already says `duplicate`. + if (accepted > 0) { + for (const point of aePoints) writePoint(env, ctx, point); + } + + return respondIngested(env, ctx, "lite", { accepted, duplicate }, bytes.byteLength); +} + +registerRoute("POST", "/telemetry/lite", handleLite); diff --git a/runner/workers/o11y/src/normalise/browser-attrs.ts b/runner/workers/o11y/src/normalise/browser-attrs.ts new file mode 100644 index 0000000000..c52f4a394d --- /dev/null +++ b/runner/workers/o11y/src/normalise/browser-attrs.ts @@ -0,0 +1,62 @@ +// The contract's `HotAttrs` (`toAePoint`'s third argument) has fields — +// `reason`, `route_class`, `fingerprint`, `model`, `provider`, `device`, +// `bucket`, `ref`, `area` — that `attrs.ts#ALLOWED_ATTRIBUTE_KEYS` does not +// list, because that allowlist governs what a *stored, Loki-bound* record +// may carry (§3's small resource/structured-metadata set), not what +// Analytics Engine's §4 layout accepts. This module reads these under a +// `hot.` key in the item's `context`/`attributes`, mirroring the +// existing `hot.*` convention — with one exception: `hot.kind` is already +// reserved (§3: "the Faro item kind") and `convert.ts#faroItemToRecord` +// always overwrites it with `item.type`, so AE slot `blob17` (ADR-0042's +// "docs, starter, saved, import, payload" kind) is read from +// `hot.metric_kind` instead. +// +// Read from the item's **raw, pre-scrub** context/attributes — these values +// never reach storage (only `toAePoint`, never `buildResourceLogs`), so +// `scrubTelemetry`'s allowlist would otherwise strip every one of them +// before this module ever sees them. Each value still gets this module's +// own light sanitisation (`redactPreviewHosts` + `stripQueryAndFragment` + +// a length cap), since it is client-controlled and never passed through +// the authoritative scrubber. + +import { redactPreviewHosts } from "@handsontable/demo-runtime/monitor"; +import { type HotAttrs, stripQueryAndFragment } from "@handsontable/demo-runtime/telemetry"; + +const MAX_ATTR_LEN = 256; + +/** `HotAttrs` field name → the raw `hot.` context key this module reads + * it from. `hot.kind`, `hot.demo_id`, `hot.surface`, `hot.tier`, + * `hot.framework`, `hot.ht_major`, `hot.outcome` are deliberately absent — + * each of those is already part of the allowlisted resource/structured- + * metadata set and is read off the *converted* record instead (see + * `normalise/faro.ts`), not duplicated here. */ +const AE_ONLY_KEYS: ReadonlyArray<[keyof HotAttrs, string]> = [ + ["reason", "hot.reason"], + ["route_class", "hot.route_class"], + ["fingerprint", "hot.fingerprint"], + ["model", "hot.model"], + ["provider", "hot.provider"], + ["device", "hot.device"], + ["bucket", "hot.bucket"], + ["ref", "hot.ref"], + ["area", "hot.area"], + ["kind", "hot.metric_kind"], +]; + +function sanitize(value: string): string { + return redactPreviewHosts(stripQueryAndFragment(value)).slice(0, MAX_ATTR_LEN); +} + +/** Extracts the AE-only `HotAttrs` fields from a raw (pre-scrub) Faro item's + * merged `context`/`attributes` bag. Returns `{}` for `undefined`/empty + * input — every field stays optional, `toAePoint` only reads what a given + * metric's §5 row lists. */ +export function readAeOnlyAttrs(raw: Record | undefined): HotAttrs { + if (!raw) return {}; + const out: Record = {}; + for (const [field, key] of AE_ONLY_KEYS) { + const value = raw[key]; + if (typeof value === "string" && value.length > 0) out[field] = sanitize(value); + } + return out as HotAttrs; +} diff --git a/runner/workers/o11y/src/normalise/deploy.ts b/runner/workers/o11y/src/normalise/deploy.ts new file mode 100644 index 0000000000..be611fbfec --- /dev/null +++ b/runner/workers/o11y/src/normalise/deploy.ts @@ -0,0 +1,109 @@ +// ADR §C.2: "each deploy job posts `{service, sha, cf_version_id}` to +// `/telemetry/deploy`, which becomes a Loki line and a Grafana annotation." +// Worker tenant (§8: "worker" — every non-browser source). +// +// The record's **resource** attributes are the o11y worker's own +// self-identity (`service.name=demos-o11y`, `hot.surface=o11y`, via +// `withResourceAttrDefaults`'s `SURFACE_BY_SERVICE_NAME` mapping) — a +// deploy event is an annotation the o11y worker *reports about* a +// third-party deploy, not a record the deploying service itself emitted, +// the same way `o11y.ingest`'s own resource identity is always the o11y +// worker's (`normalise/respond.ts#o11ySelfIdentity`). The **body** is +// `JSON.stringify({event:"deploy", service, sha, cf_version_id})` — +// `event: "deploy"` is not named in ADR §C.2's `{service, sha, +// cf_version_id}`, added so a Grafana annotation query can filter on it +// without a body-text regex. + +import { msToUnixNano, scrubTelemetry, type NormalisedRecord } from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem } from "../env.js"; +import { hashRecord } from "./hash.js"; +import { o11ySelfIdentity } from "./respond.js"; +import { withResourceAttrDefaults } from "./points.js"; +import { scrubBodyText } from "./text-scrub.js"; + +export interface DeployPayload { + service: string; + sha: string; + cf_version_id: string; +} + +export function isDeployPayload(v: unknown): v is DeployPayload { + if (typeof v !== "object" || v === null) return false; + const d = v as Record; + return typeof d["service"] === "string" && typeof d["sha"] === "string" && typeof d["cf_version_id"] === "string"; +} + +// The payload itself carries no event time (§C.2 names only `service`, +// `sha`, `cf_version_id`), and there is no per-delivery header either — +// this is a plain authenticated POST from a CI step, not a webhook system +// with its own delivery id. A fixed `rawEventTime` would mean two genuinely +// different deploys that happen to redeploy the exact same `{service, sha, +// cf_version_id}` (e.g. a no-op redeploy, or a rollback) inside the same +// 24h dedupe window (`DEDUPE_WINDOW_MS`) hash identically and the second +// one silently vanishes. Bucketing the worker's own receive time to the +// minute gives each such event a distinct `rawEventTime` while still +// collapsing a genuine network-level retry of the same POST, which lands +// well inside the same one-minute bucket. +const RAW_EVENT_TIME_BUCKET_MS = 60_000; + +function bucketedReceivedAt(receivedAtMs: number): string { + return String(Math.floor(receivedAtMs / RAW_EVENT_TIME_BUCKET_MS)); +} + +export async function processDeployPayload( + payload: DeployPayload, + env: Env, + receivedAtMs: number, +): Promise { + const identity = o11ySelfIdentity(env); + const resourceAttributes = withResourceAttrDefaults( + { + "service.name": identity.service_name, + "service.version": identity.service_version, + }, + env, + ); + // `cf_version_id` can arrive as `""` — `isDeployPayload` only checks it + // is a string, and `master.yml`'s own `version_id=$(grep ...) || true` + // deliberately lets a wrangler wording change through as an empty string + // rather than failing the deploy job after the deploy already shipped. + // This ingest path must NEVER reject the event over it (the deploy still + // happened and is still worth recording), but an empty version id + // silently corrupts the ADR §C.2 deploy-correlation record — mark it + // both ways: a `console.warn` (reaches Workers Logs -> Loki) and a body + // field (NOT a §4/§5 `attributes` key — that bag is a fixed, + // contract-governed schema via `ALLOWED_ATTRIBUTE_KEYS`, which + // `scrubTelemetry` below strips; the body is free JSON, so a marker + // there survives scrubbing and is queryable via a body filter, e.g. + // `cf_version_id_missing":true`). + const versionIdMissing = payload.cf_version_id.length === 0; + if (versionIdMissing) { + console.warn(`[o11y] deploy event for "${payload.service}" (sha ${payload.sha}) arrived with an empty cf_version_id`); + } + let record: NormalisedRecord = { + body: JSON.stringify({ + event: "deploy", + service: payload.service, + sha: payload.sha, + cf_version_id: payload.cf_version_id, + ...(versionIdMissing ? { cf_version_id_missing: true } : {}), + }), + timeUnixNano: msToUnixNano(receivedAtMs), + resourceAttributes, + attributes: {}, + }; + // CI-controlled input is lower risk than Sentry's (`sentry.ts`'s own doc + // comment), but the OIDC gate authenticates the *deployer*, not the + // content of `service`/`sha`; running the same authoritative scrub costs + // nothing here and keeps every worker-tenant record on the same + // guarantee. + record = scrubTelemetry(record)!; + record.body = scrubBodyText(record.body); + const hash = await hashRecord({ + body: record.body, + resourceAttributes: record.resourceAttributes, + attributes: {}, + rawEventTime: bucketedReceivedAt(receivedAtMs), + }); + return { hash, record }; +} diff --git a/runner/workers/o11y/src/normalise/faro.ts b/runner/workers/o11y/src/normalise/faro.ts new file mode 100644 index 0000000000..cb42693d9b --- /dev/null +++ b/runner/workers/o11y/src/normalise/faro.ts @@ -0,0 +1,488 @@ +// ADR §B.2 step 1 (Faro half) + §6's item → Analytics Engine / inbox table. +// Order: `scrubTelemetry(item)` → `faroItemToRecord(scrubbedItem, …)`. +// Faro's transport posts a `TransportBody` (one shared `meta` plus separate +// `exceptions`/`logs`/`measurements`/`events`/`traces` arrays), not an array +// of items; this module reconstructs `{type, payload, meta}` items from it. +// `traces` is never unpacked (ADR §C.4): any item there is invalid. + +import { + ATTR_HOT_DEMO_ID, + ATTR_HOT_FRAMEWORK, + ATTR_HOT_HT_MAJOR, + ATTR_HOT_OUTCOME, + ATTR_HOT_SURFACE, + ATTR_HOT_TIER, + fingerprint as computeFingerprint, + faroItemToRecord, + feedsNewFingerprintAlert, + INBOX_RECORD_MAX_BYTES, + isValidFingerprint, + METRICS, + scrubTelemetry, + toAePoint, + type AePoint, + type HtMajor, + type MetricName, + type MetricValues, + type ScrubbableFaroItem, + type ServiceIdentity, + type Surface, + type Tier, +} from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem } from "../env.js"; +import { readAeOnlyAttrs } from "./browser-attrs.js"; +import { hashRecord } from "./hash.js"; +import { withResourceAttrDefaults } from "./points.js"; +import { scrubAttributeValues, scrubBodyText } from "./text-scrub.js"; + +/** Bounds a Faro `TransportBody` batch, generous headroom over the SDK's + * own limit (50 items) — an unbounded batch let a client inflate AE + * points/dedupe load (measured ~16.7k items in one 1 MB body). + * `handleCollect` rejects the whole request above this, like an oversized + * body. */ +export const MAX_FARO_ITEMS_PER_BODY = 200; + +/** Total item count across every unpacked kind (`traces` included, since a + * trace item still counts toward the cap even though it is always + * rejected as invalid) — used by `index.ts#handleCollect` before this + * module does any real work on the batch. */ +export function countFaroItems(body: unknown): number { + if (typeof body !== "object" || body === null) return 0; + const wire = body as Record; + let count = 0; + for (const key of ["exceptions", "logs", "measurements", "events", "traces"]) { + const list = wire[key]; + if (Array.isArray(list)) count += list.length; + } + return count; +} + +const ITEM_KIND_BY_BODY_KEY: Readonly> = { + exceptions: "exception", + logs: "log", + measurements: "measurement", + events: "event", +}; + +const LITE_VITAL_KEYS: Readonly> = { + lcp: "LCP", + inp: "INP", + cls: "CLS", + ttfb: "TTFB", +}; + +/** + * Server-side backstop for a client that skips the shared noise gates in + * Faro's `beforeSend` (`eventGate.ts#isUnhandledNoise`/ + * `isOfficeScannerRejection`). Duplicated, not imported: keep in sync by + * hand. `isForeignUnhandled`/`isEdgelessForeignSessionStart` stay + * browser-gate-only — both need browser-only context this path never gets. + */ +const SERVER_SIDE_UNHANDLED_NOISE: readonly RegExp[] = [ + /^ResizeObserver loop/i, + /^AbortError/i, + /Failed to fetch/i, + /Load failed/i, +]; +const SERVER_SIDE_INJECTED_SCANNER_MESSAGES: readonly RegExp[] = [ + /Object Not Found Matching Id/i, // Microsoft Outlook/Office safelink scanner +]; + +/** True for an unhandled exception item whose message/type matches one of + * the shared noise gates — mirrors `eventGate.ts`'s own + * `mechanism.handled === false` discriminator (an explicit, handled report + * that merely quotes this text must never be silently dropped). */ +function isServerSideNoiseException(value: string | undefined, type: string | undefined, handled: boolean): boolean { + if (handled) return false; + const patterns = [...SERVER_SIDE_UNHANDLED_NOISE, ...SERVER_SIDE_INJECTED_SCANNER_MESSAGES]; + return patterns.some((re) => re.test(value ?? "") || re.test(type ?? "")); +} + +export interface ProcessedFaroItem { + /** Absent for a console-dropped, unrecoverable or oversize item. An + * `example.*` event also carries a hash-only `ingestItem` (`record` + * absent) so `InboxWriter.ingest`'s dedupe transaction covers it too, + * without storing anything (§6). */ + ingestItem?: IngestItem; + /** When {@link ingestItem} is set, write these only for a hash + * `InboxWriter.ingest` reports `"accepted"`, never `"duplicate"` — a + * retried batch must not double-count a point. When absent, these have + * no hash to gate on and are written unconditionally. */ + aePoints: AePoint[]; + /** Set when this item could not be converted/validated at all — the caller + * writes one `invalid_item` `o11y.ingest` point and moves on (never a + * 500). Unset for a console-drop, an oversize record or an `example.*` + * event: those are intentional, not a failure. */ + invalid?: string; + /** Set when the built record alone (well-formed, otherwise storable) + * exceeds `INBOX_RECORD_MAX_BYTES` (ADR §B.2 step 1, "drop records over + * 256 KB"). Distinct from `invalid` so the caller writes a + * `reason: "size"` point, not `reason: "invalid_item"`. */ + oversize?: boolean; +} + +function metricValuesFromPayload(values: Record | undefined): MetricValues { + if (!values) return {}; + const out: MetricValues = {}; + for (const key of ["count", "duration_ms", "value", "usd", "tokens_in", "tokens_out", "bytes", "cap"] as const) { + if (typeof values[key] === "number") out[key] = values[key]; + } + return out; +} + +function browserHotAttrs(resourceAttributes: Record) { + return { + surface: resourceAttributes[ATTR_HOT_SURFACE] as Surface | undefined, + tier: resourceAttributes[ATTR_HOT_TIER] as Tier | undefined, + framework: resourceAttributes[ATTR_HOT_FRAMEWORK], + ht_major: resourceAttributes[ATTR_HOT_HT_MAJOR] as HtMajor | undefined, + outcome: resourceAttributes[ATTR_HOT_OUTCOME], + }; +} + +function processMeasurement( + raw: Record, + resourceAttributes: Record, + demoId: string | undefined, + aeOnly: ReturnType, + service: ServiceIdentity, +): AePoint[] { + const type = typeof raw["type"] === "string" ? raw["type"] : ""; + const values = raw["values"] as Record | undefined; + const common = { service_name: service.name, service_version: service.version, environment: service.environment }; + + if (type === "web-vitals") { + const points: AePoint[] = []; + for (const [key, value] of Object.entries(values ?? {})) { + const reason = LITE_VITAL_KEYS[key.toLowerCase()]; + if (!reason || typeof value !== "number") continue; + points.push( + toAePoint( + "web_vital", + { value }, + { + ...common, + surface: browserHotAttrs(resourceAttributes).surface, + framework: browserHotAttrs(resourceAttributes).framework, + ht_major: browserHotAttrs(resourceAttributes).ht_major, + reason, + device: aeOnly.device, + demo_id: demoId, + }, + ), + ); + } + return points; + } + + if ((METRICS as Record)[type] && METRICS[type as MetricName].emittedBy.includes("browser")) { + return [ + toAePoint(type as MetricName, metricValuesFromPayload(values), { + ...common, + ...browserHotAttrs(resourceAttributes), + ...aeOnly, + }), + ]; + } + return []; +} + +function processExampleEvent( + name: string, + resourceAttributes: Record, + aeOnly: ReturnType, + service: ServiceIdentity, +): AePoint[] { + if (!(name in METRICS)) return []; + const common = { service_name: service.name, service_version: service.version, environment: service.environment }; + return [ + toAePoint(name as MetricName, { count: 1 }, { ...common, ...browserHotAttrs(resourceAttributes), ...aeOnly }), + ]; +} + +/** Picks the fingerprint a client offered, validated, or falls back to + * computing it server-side. `wireFingerprint` (Faro's own field) is + * preferred; `aeOnlyFingerprint` is the next fallback; neither is trusted + * verbatim (must match §7's `:<16 hex>` shape). `fallbackMessage` + * is the last resort — the `type: value` head only, never `record.body`'s + * stack lines, which shift on every deploy and would turn a recurring + * defect into a fresh `fp:` entry each release. */ +function resolveFingerprint( + wireFingerprint: string | undefined, + aeOnlyFingerprint: string | undefined, + surface: string, + fallbackMessage: string, +): string { + if (wireFingerprint !== undefined && isValidFingerprint(wireFingerprint)) return wireFingerprint; + if (aeOnlyFingerprint !== undefined && isValidFingerprint(aeOnlyFingerprint)) return aeOnlyFingerprint; + return computeFingerprint(surface, fallbackMessage); +} + +/** The `type: value` head of a Faro exception payload, with NO stack — + * deliberately mirrors `convert.ts#faroBody`'s own exception-head + * construction (duplicated here rather than imported, same tradeoff as + * `SERVER_SIDE_UNHANDLED_NOISE` above). `o11y-normalise.test.mjs` pins this + * shape directly, so a future drift between the two shows up as a failing + * test, not a silent mismatch. */ +function exceptionFingerprintMessage(payload: { type?: string; value?: string }): string { + const value = payload.value ?? ""; + return payload.type ? `${payload.type}: ${value}` : value; +} + +/** Inputs that can change this item's output (an AE point, an alert) but + * never reach `hashRecord`'s `body`/`attributes` (`hash.ts#PreHashRecord.extra`). + * - `type`: `faroBody()`'s measurement case stringifies only `values`, so + * two different measurements with the same `values` would hash the same. + * - `aeOnly`: `hot.*` AE-only attributes never reach a stored record's + * `attributes` (`browser-attrs.ts`), so they'd otherwise be invisible here. + * - `sessionId`: a Faro META field, never copied into `context`/`attributes`, + * so `faroItemToRecord` never sees it; read from the raw `meta` param. */ +function hashExtra( + payload: Record, + aeOnly: ReturnType, + meta: unknown, +): Record { + const extra: Record = {}; + const type = payload["type"]; + if (typeof type === "string" && type.length > 0) extra["type"] = type; + for (const [key, value] of Object.entries(aeOnly)) { + if (typeof value === "string" && value.length > 0) extra[`ae.${key}`] = value; + } + const sessionId = (meta as { session?: { id?: unknown } } | null | undefined)?.session?.id; + if (typeof sessionId === "string" && sessionId.length > 0) extra["session.id"] = sessionId; + return extra; +} + +function processException( + fallbackMessage: string, + resourceAttributes: Record, + demoId: string | undefined, + handled: boolean, + aeOnly: ReturnType, + service: ServiceIdentity, + wireFingerprint: string | undefined, +): { points: AePoint[]; fingerprint?: string } { + const surface = (resourceAttributes[ATTR_HOT_SURFACE] as Surface | undefined) ?? "authoring"; + const fp = resolveFingerprint(wireFingerprint, aeOnly.fingerprint, surface, fallbackMessage); + const common = { service_name: service.name, service_version: service.version, environment: service.environment }; + const point = handled + ? toAePoint("error.handled", { count: 1 }, { ...common, surface, route_class: aeOnly.route_class, fingerprint: fp }) + : toAePoint("error.uncaught", { count: 1 }, { ...common, surface, fingerprint: fp, demo_id: demoId }); + return { points: [point], fingerprint: feedsNewFingerprintAlert(surface) ? fp : undefined }; +} + +/** Unpacks a Faro `TransportBody`, scrubs/converts/validates every item, + * and returns one {@link ProcessedFaroItem} per item, each with its + * `IngestItem.hash` already computed (ADR §B.2 step 2) — the route + * handler hashes nothing itself. */ +export async function processFaroBody( + body: unknown, + env: Env, + service: ServiceIdentity, + receivedAtMs: number, +): Promise { + const results: ProcessedFaroItem[] = []; + if (typeof body !== "object" || body === null) return [{ aePoints: [], invalid: "not an object" }]; + const wire = body as Record; + const meta = wire["meta"]; + if (typeof meta !== "object" || meta === null) return [{ aePoints: [], invalid: "missing meta" }]; + + if (Array.isArray(wire["traces"]) || (wire["traces"] && typeof wire["traces"] === "object")) { + results.push({ aePoints: [], invalid: "trace item (not exported)" }); + } + + const jobs: Promise[] = []; + for (const [bodyKey, kind] of Object.entries(ITEM_KIND_BY_BODY_KEY)) { + const list = wire[bodyKey]; + if (!Array.isArray(list)) continue; + for (const rawPayload of list) { + jobs.push(processOneItem(kind, rawPayload as Record, meta, env, service, receivedAtMs)); + } + } + results.push(...(await Promise.all(jobs))); + return results; +} + +async function processOneItem( + type: ScrubbableFaroItem["type"], + payload: Record, + meta: unknown, + env: Env, + service: ServiceIdentity, + receivedAtMs: number, +): Promise { + // An untrusted client can put a `null`/non-object entry inside a Faro + // batch array (`{"logs":[null]}` is valid JSON); guard against that + // shape here rather than letting the destructure below throw. + if (typeof payload !== "object" || payload === null) { + return { aePoints: [], invalid: "item is not an object" }; + } + + const rawContext = { + ...((payload["context"] as Record | undefined) ?? {}), + ...((payload["attributes"] as Record | undefined) ?? {}), + }; + const aeOnly = readAeOnlyAttrs(rawContext); + const handled = rawContext["handled"] === "true"; + // Faro's own `pushError({ fingerprint })` option lands in + // `payload.fingerprint`, a sibling of `context`/`attributes`, not inside + // either — `readAeOnlyAttrs` (which only reads `context`) never sees it. + // Length-capped defensively before the validation regex in + // `resolveFingerprint` runs against untrusted input. + const rawWireFingerprint = payload["fingerprint"]; + const wireFingerprint = + typeof rawWireFingerprint === "string" && rawWireFingerprint.length <= 128 ? rawWireFingerprint : undefined; + + const item: ScrubbableFaroItem = { type, payload: payload as never, meta: meta as never }; + let scrubbed: ScrubbableFaroItem | null; + try { + scrubbed = scrubTelemetry(item); + } catch (err) { + // `scrubTelemetry` itself can throw on a malformed nested shape; this + // per-item boundary is the "never a 500" backstop for that. + return { aePoints: [], invalid: err instanceof Error ? err.message : String(err) }; + } + if (scrubbed === null) return { aePoints: [] }; // console item, intentionally dropped (§3) + + // Dropped exactly like a console item (see SERVER_SIDE_UNHANDLED_NOISE + // above): no stored record, no AE point, no fingerprint — the same thing + // Sentry/Faro's own `beforeSend` returning `null` does browser-side. + if ( + type === "exception" && + isServerSideNoiseException(scrubbed.payload.value, scrubbed.payload.type, handled) + ) { + return { aePoints: [] }; + } + + let record; + try { + record = faroItemToRecord(scrubbed, { service, receivedAtMs }); + } catch (err) { + return { aePoints: [], invalid: err instanceof Error ? err.message : String(err) }; + } + // Snapshot before `withResourceAttrDefaults` fills `hot.outcome = "none"` + // for the stored record (ADR-0041 §L.15, "every label populated"): + // `toAePoint` throws on `outcome`/`reason` for a metric whose §5 row has + // no such slot, so reusing the defaulted bag for AE point attrs would + // turn a metric like `example.open` into a spurious throw. + // `clientResourceAttributes` is what `browserHotAttrs()` reads instead. + const clientResourceAttributes = { ...record.resourceAttributes }; + withResourceAttrDefaults(record.resourceAttributes, env); + // `stripCodeFrame` (via `scrubText`) and the attribute allowlist need a + // second pass over fields `faroItemToRecord` itself builds (exception + // `type`, event `name`, stack-frame `function` text folded into `body`), + // which the first `scrubTelemetry` call on the raw item never saw. Run it + // again here on the OTLP-record branch, exactly like `lite.ts`/`otlp.ts` + // do for their own converted records — it never returns `null` for that + // branch (only a Faro item can be dropped as a console item). + record = scrubTelemetry(record)!; + // `scrubTelemetry` strips query strings only from discrete URL fields, + // never from an embedded URL inside the built body text; this closes + // that gap. + record.body = scrubBodyText(record.body); + // An allowlisted attribute/resource-attribute value only gets + // `redactPreviewHosts` inside `scrubTelemetry` — a query string or an + // embedded user-agent in `context`/a diagnostic tag value would survive + // otherwise. Same extra pass `body` gets, applied to every attribute + // value. + record.attributes = scrubAttributeValues(record.attributes); + record.resourceAttributes = scrubAttributeValues(record.resourceAttributes) ?? record.resourceAttributes; + const demoId = record.attributes?.[ATTR_HOT_DEMO_ID]; + + // Metric extraction (a crafted `outcome`/`reason`/attribute value throws + // inside `toAePoint`) is isolated in its own try/catch, deliberately + // separate from record storage below it — a malformed metric attribute + // must cost only its own point, never the underlying log record, which is + // already valid at the OTLP level regardless of what the metric extraction + // made of it. + let aePoints: AePoint[] = []; + let storeRecord = true; + let itemFingerprint: string | undefined; + try { + if (type === "event") { + const name = typeof scrubbed.payload.name === "string" ? scrubbed.payload.name : ""; + if (name.startsWith("example.")) { + // Set BEFORE calling the extractor: the catch block below reads + // `storeRecord` to decide whether a throwing extractor should + // surface as `invalid` (nothing to salvage) or fall through to + // still storing the record. Setting the flag AFTER a call that can + // itself throw would leave it at its default `true` on that path — + // a crafted `example.*` attribute that made + // `processExampleEvent`/`toAePoint` throw would then store a record + // anyway, contradicting §6 ("AE points only, never stored"). + storeRecord = false; // §6: example.* events are AE points only, never stored + aePoints = processExampleEvent(name, clientResourceAttributes, aeOnly, service); + } + } else if (type === "measurement") { + // A Faro measurement (incl. `web-vitals`, `processMeasurement`'s + // other branch below) is ~99% of the browser Loki tenant's lines and + // drained bytes, and no dashboard reads a stored measurement record — + // the AE point above is the only consumer (ADR §F.1). Reuses the + // hash-only `ingestItem` path `example.*` events already take above. + // + // Set BEFORE calling `processMeasurement`: a crafted + // `hot.outcome`/`hot.reason` that makes `toAePoint` throw must + // surface as `invalid`, not fall through and store a record with + // zero AE points (§6 "none" ruling). + storeRecord = false; + aePoints = processMeasurement(payload, clientResourceAttributes, demoId, aeOnly, service); + } else if (type === "exception") { + const ex = processException( + exceptionFingerprintMessage(scrubbed.payload), + clientResourceAttributes, + demoId, + handled, + aeOnly, + service, + wireFingerprint, + ); + aePoints = ex.points; + itemFingerprint = ex.fingerprint; + } + // "log" and a non-"example." event: inbox record only, no AE point (§6). + } catch (err) { + // The point extraction failed; if this item was never going to be + // stored anyway (an "example.*" event with a bad attribute), there is + // nothing left to salvage — surface it as invalid. Otherwise fall + // through and still store the record; the caller sees zero points for + // this item and can tell from `invalid` that the metric was dropped. + if (!storeRecord) return { aePoints: [], invalid: err instanceof Error ? err.message : String(err) }; + aePoints = []; + } + + if (!storeRecord) { + // An `example.*` event skips row storage (§6: AE points only) but must + // still go through `InboxWriter.ingest`'s hash/dedupe transaction, so a + // retried/redelivered batch does not double-count its AE point. Reuses + // the exact hash shape `hashRecord` computes for a stored record below + // (same fields, including the raw client `timestamp` — not the clamped + // one — so two genuine clicks a browser reports with distinct + // timestamps never collapse into one) purely for dedupe: no `record` is + // attached, so `InboxWriter.ingest`/`appendRows` skip a `record`-less + // item entirely — nothing is ever stored for it. + const hash = await hashRecord({ + body: record.body, + resourceAttributes: record.resourceAttributes, + attributes: record.attributes ?? {}, + rawEventTime: scrubbed.payload.timestamp ?? "", + extra: hashExtra(payload, aeOnly, meta), + }); + return { aePoints, ingestItem: { hash } }; + } + + // `pack.ts`'s row-chunking (and the contract's own "records over 256 KB + // are dropped" rule, §8) assumes normalise enforces this on every ingest + // path, so the Faro path checks it here the same way the OTLP path does. + if (new TextEncoder().encode(JSON.stringify(record)).length > INBOX_RECORD_MAX_BYTES) { + return { aePoints, oversize: true }; + } + + const hash = await hashRecord({ + body: record.body, + resourceAttributes: record.resourceAttributes, + attributes: record.attributes ?? {}, + rawEventTime: scrubbed.payload.timestamp ?? "", + extra: hashExtra(payload, aeOnly, meta), + }); + return { aePoints, ingestItem: { hash, record, fingerprint: itemFingerprint } }; +} diff --git a/runner/workers/o11y/src/normalise/hash.ts b/runner/workers/o11y/src/normalise/hash.ts new file mode 100644 index 0000000000..d056a2bfa8 --- /dev/null +++ b/runner/workers/o11y/src/normalise/hash.ts @@ -0,0 +1,52 @@ +// ADR §B.2 step 2: "Hash each record (SHA-256) at this point, before any +// arrival-time value exists, so a redelivered body hashes identically." +// +// A `NormalisedRecord` (`@handsontable/demo-runtime/telemetry`) already +// carries its *final*, clamped `timeUnixNano` — hashing that would break +// ADR-0041 §L.4 for the zero-timestamp case, since `InboxWriter.arrivalMs` +// differs "seconds apart" between two deliveries and the clamp falls back +// to it. What must be identical across a redelivery is the **content**: +// `body`, `resourceAttributes`, `attributes`, and the record's own *raw*, +// un-clamped source timestamp (Faro's `payload.timestamp` string, or +// OTLP's raw `time_unix_nano`/`observed_time_unix_nano` before the +// received_at fallback) — genuinely part of the original body, unlike the +// arrival time, so it stays in the hash to keep two real events with +// identical text but different real timestamps distinct. + +import { sha256Hex } from "../gates/util.js"; + +export interface PreHashRecord { + body: string; + resourceAttributes: Record; + attributes?: Record; + /** The record's own raw source timestamp, exactly as received (before any + * clamp/fallback) — `""` when the source carried none at all, which is + * itself a stable, redelivery-identical value. */ + rawEventTime: string; + /** Additional content that distinguishes two otherwise-identical records + * but is not itself part of `body`/`attributes` — `normalise/faro.ts` + * uses this for the raw Faro `payload.type`, the AE-only `hot.*` + * attributes, and the Faro session id (none of which reach a stored + * record's `attributes`). Left `undefined` (not an empty object) by + * every other caller of `hashRecord`, so their hash output stays + * stable for an identical redelivery. */ + extra?: Record; +} + +/** Deterministic JSON: object keys sorted recursively, so two structurally + * equal objects with differently-ordered keys hash identically (the two + * attribute bags are built from `Object.entries`/`for..of` over payloads + * whose own key order is not guaranteed to match across a redelivery). */ +function stableStringify(value: unknown): string { + if (Array.isArray(value)) return `[${value.map(stableStringify).join(",")}]`; + if (value !== null && typeof value === "object") { + const entries = Object.entries(value as Record).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)); + return `{${entries.map(([k, v]) => `${JSON.stringify(k)}:${stableStringify(v)}`).join(",")}}`; + } + return JSON.stringify(value); +} + +/** SHA-256 hex over the stable-stringified {@link PreHashRecord}. */ +export async function hashRecord(record: PreHashRecord): Promise { + return sha256Hex(stableStringify(record)); +} diff --git a/runner/workers/o11y/src/normalise/otlp-protobuf.ts b/runner/workers/o11y/src/normalise/otlp-protobuf.ts new file mode 100644 index 0000000000..df62d96706 --- /dev/null +++ b/runner/workers/o11y/src/normalise/otlp-protobuf.ts @@ -0,0 +1,171 @@ +// Hand-rolled OTLP `ExportLogsServiceRequest` protobuf decoder, wire-primitive +// only (`@bufbuild/protobuf/wire`'s `BinaryReader`, no generated/reflective +// message types: `protobufjs`'s reflective decode path uses `new +// Function(...)`, which workerd disallows). Decodes into the same flat +// shape `otlp-json.ts` produces from the JSON variant, so `normalise/otlp.ts` +// has one shared "what to do with a decoded ResourceLogs" step regardless +// of which wire format arrived. +// +// Proto shapes decoded (github.com/open-telemetry/opentelemetry-proto, +// `opentelemetry/proto/{logs,common,resource}/v1`), only the fields this +// contract reads: +// +// ExportLogsServiceRequest { repeated ResourceLogs resource_logs = 1; } +// ResourceLogs { Resource resource = 1; repeated ScopeLogs scope_logs = 2; } +// Resource { repeated KeyValue attributes = 1; } +// ScopeLogs { repeated LogRecord log_records = 2; } +// LogRecord { fixed64 time_unix_nano = 1; fixed64 observed_time_unix_nano = 11; +// string severity_text = 3; AnyValue body = 5; +// repeated KeyValue attributes = 6; } +// KeyValue { string key = 1; AnyValue value = 2; } +// AnyValue { string string_value = 1; bool bool_value = 2; int64 int_value = 3; +// double double_value = 4; bytes bytes_value = 7; } (oneof) +// +// Only scalar `AnyValue` kinds are decoded to a string: +// `array_value`/`kvlist_value` (nested `AnyValue` collections) are rendered +// as `"[unsupported: array]"`/`"[unsupported: kvlist]"` rather than +// recursively decoded — no captured export observed a nested value under +// `hot.*`/`service.*`/`deployment.*` (the only attributes this contract +// keeps), and every allowlisted key is a plain string by contract. + +import { BinaryReader } from "@bufbuild/protobuf/wire"; + +export interface DecodedLogRecord { + timeUnixNano?: string; + observedTimeUnixNano?: string; + severityText?: string; + body?: string; + attributes: Record; +} + +export interface DecodedResourceLogs { + resourceAttributes: Record; + logRecords: DecodedLogRecord[]; +} + +function anyValueToString(bytes: Uint8Array): string { + const r = new BinaryReader(bytes); + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + switch (fieldNo) { + case 1: // string_value + return r.string(); + case 2: // bool_value + return String(r.bool()); + case 3: // int_value + return String(r.int64()); + case 4: // double_value + return String(r.double()); + case 7: // bytes_value + return btoa(String.fromCharCode(...r.bytes())); + case 5: + r.skip(wireType); + return "[unsupported: array]"; + case 6: + r.skip(wireType); + return "[unsupported: kvlist]"; + default: + r.skip(wireType); + } + } + return ""; +} + +function readKeyValue(bytes: Uint8Array): [string, string] { + const r = new BinaryReader(bytes); + let key = ""; + let value = ""; + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + if (fieldNo === 1) key = r.string(); + else if (fieldNo === 2) value = anyValueToString(r.bytes()); + else r.skip(wireType); + } + return [key, value]; +} + +function readAttributes(r: BinaryReader): Record { + const out: Record = {}; + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + if (fieldNo === 1) { + const [k, v] = readKeyValue(r.bytes()); + out[k] = v; + } else { + r.skip(wireType); + } + } + return out; +} + +function readLogRecord(bytes: Uint8Array): DecodedLogRecord { + const r = new BinaryReader(bytes); + const rec: DecodedLogRecord = { attributes: {} }; + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + switch (fieldNo) { + case 1: + rec.timeUnixNano = String(r.fixed64()); + break; + case 11: + rec.observedTimeUnixNano = String(r.fixed64()); + break; + case 3: + rec.severityText = r.string(); + break; + case 5: + rec.body = anyValueToString(r.bytes()); + break; + case 6: { + const [k, v] = readKeyValue(r.bytes()); + rec.attributes[k] = v; + break; + } + default: + r.skip(wireType); + } + } + return rec; +} + +function readScopeLogs(bytes: Uint8Array, out: DecodedLogRecord[]): void { + const r = new BinaryReader(bytes); + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + if (fieldNo === 2) out.push(readLogRecord(r.bytes())); + else r.skip(wireType); + } +} + +function readResourceLogs(bytes: Uint8Array): DecodedResourceLogs { + const r = new BinaryReader(bytes); + const out: DecodedResourceLogs = { resourceAttributes: {}, logRecords: [] }; + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + if (fieldNo === 1) { + // Resource { repeated KeyValue attributes = 1; } + const resourceBytes = r.bytes(); + const rr = new BinaryReader(resourceBytes); + out.resourceAttributes = readAttributes(rr); + } else if (fieldNo === 2) { + readScopeLogs(r.bytes(), out.logRecords); + } else { + r.skip(wireType); + } + } + return out; +} + +/** Decodes an `ExportLogsServiceRequest` protobuf body into a flat list of + * {@link DecodedResourceLogs}. Throws on a truncated/malformed body — the + * caller (`normalise/otlp.ts`) turns that into a `400`, never a `500`. */ +export function decodeOtlpProtobuf(bytes: Uint8Array): DecodedResourceLogs[] { + const r = new BinaryReader(bytes); + const out: DecodedResourceLogs[] = []; + while (r.pos < r.len) { + const [fieldNo, wireType] = r.tag(); + if (fieldNo === 1) out.push(readResourceLogs(r.bytes())); + else r.skip(wireType); + } + return out; +} diff --git a/runner/workers/o11y/src/normalise/otlp.ts b/runner/workers/o11y/src/normalise/otlp.ts new file mode 100644 index 0000000000..95fb55459b --- /dev/null +++ b/runner/workers/o11y/src/normalise/otlp.ts @@ -0,0 +1,301 @@ +// ADR §B.2 step 1 (Cloudflare export half): decode Cloudflare's export +// (protobuf or JSON); keep only allowlisted attributes; hoist `hot.*` and +// `service.*` to resource attributes; run the server-side scrubber (§E.4). +// Plus §C.2's OTLP timestamp rule: ADR §C.2 only clamps **browser and +// beacon** item timestamps, never OTLP — an OTLP record keeps its real +// `time_unix_nano`, falling back to `observed_time_unix_nano`, then +// `received_at` (a replayed sandbox-probe fixture days later must still +// pass exit criterion 3, ADR-0041 §L.3). Every timestamp stays the +// decimal-nanosecond string OTLP itself uses — never round-tripped through +// a `number` of milliseconds, which would lose precision past 2^53 +// nanoseconds (about 104 days). + +import { + ATTR_SERVICE_NAME, + hoistAttributes, + INBOX_RECORD_MAX_BYTES, + isValidFingerprint, + msToUnixNano, + RESOURCE_ATTRS, + sanitizeResourceAttributes, + scrubTelemetry, + SERVICE_NAMES, + type NormalisedRecord, + type ScrubbableOtlpRecord, +} from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem } from "../env.js"; +import { hashRecord } from "./hash.js"; +import { decodeOtlpProtobuf, type DecodedLogRecord, type DecodedResourceLogs } from "./otlp-protobuf.js"; +import { withResourceAttrDefaults } from "./points.js"; +import { scrubBodyText } from "./text-scrub.js"; + +// ---- OTLP JSON decode ----------------------------------------------------- + +interface OtlpAnyValueJson { + stringValue?: string; + intValue?: string | number; + doubleValue?: number; + boolValue?: boolean; + bytesValue?: string; +} +interface OtlpKeyValueJson { + key: string; + value?: OtlpAnyValueJson; +} +function anyValueJsonToString(v: OtlpAnyValueJson | undefined): string { + if (!v) return ""; + if (v.stringValue !== undefined) return v.stringValue; + if (v.boolValue !== undefined) return String(v.boolValue); + if (v.intValue !== undefined) return String(v.intValue); + if (v.doubleValue !== undefined) return String(v.doubleValue); + if (v.bytesValue !== undefined) return v.bytesValue; + return ""; +} +function attrsJsonToRecord(attrs: OtlpKeyValueJson[] | undefined): Record { + const out: Record = {}; + for (const kv of attrs ?? []) out[kv.key] = anyValueJsonToString(kv.value); + return out; +} + +/** Decodes the OTLP JSON `ExportLogsServiceRequest` shape into the same flat + * {@link DecodedResourceLogs} list the protobuf decoder produces. Throws on + * a body that is not that shape at all (not JSON, or missing + * `resourceLogs`) — the caller turns that into a `400`. */ +export function decodeOtlpJson(text: string): DecodedResourceLogs[] { + const parsed = JSON.parse(text) as { resourceLogs?: unknown }; + if (!Array.isArray(parsed.resourceLogs)) throw new Error("otlp json: missing resourceLogs"); + return parsed.resourceLogs.map((rl) => { + const r = rl as { + resource?: { attributes?: OtlpKeyValueJson[] }; + scopeLogs?: Array<{ logRecords?: unknown[] }>; + }; + const resourceAttributes = attrsJsonToRecord(r.resource?.attributes); + const logRecords: DecodedLogRecord[] = []; + for (const scope of r.scopeLogs ?? []) { + for (const lr of scope.logRecords ?? []) { + const rec = lr as { + timeUnixNano?: string | number; + observedTimeUnixNano?: string | number; + severityText?: string; + body?: OtlpAnyValueJson; + attributes?: OtlpKeyValueJson[]; + }; + logRecords.push({ + timeUnixNano: rec.timeUnixNano !== undefined ? String(rec.timeUnixNano) : undefined, + observedTimeUnixNano: rec.observedTimeUnixNano !== undefined ? String(rec.observedTimeUnixNano) : undefined, + severityText: rec.severityText, + body: anyValueJsonToString(rec.body), + attributes: attrsJsonToRecord(rec.attributes), + }); + } + } + return { resourceAttributes, logRecords }; + }); +} + +// ---- Shared decode → NormalisedRecord pipeline ----------------------------- + +/** `"0"`, `""` and `undefined` are all "no real value" for the OTLP + * timestamp fallback chain (§C.2) — Cloudflare's export, like any OTLP + * exporter, may send an explicit `"0"` rather than omitting the field. */ +function isRealTimestamp(v: string | undefined): v is string { + return v !== undefined && v !== "" && v !== "0"; +} + +/** Real Cloudflare invocation-log exports carry the ray id under + * `cloudflare.ray_id`, not the contract's `cf.ray` — without this remap, + * `hoistAttributes` (which only recognises the contract's own key names) + * silently drops it, even though `cf.ray` is named as structured metadata + * every record should carry (§3). Applied before `hoistAttributes`, so it + * works whether the source key arrived as a resource or record attribute. */ +const CLOUDFLARE_KEY_REMAP: Readonly> = { + "cloudflare.ray_id": "cf.ray", +}; + +function remapCloudflareKeys(attrs: Record): Record { + const out: Record = {}; + for (const [key, value] of Object.entries(attrs)) { + out[CLOUDFLARE_KEY_REMAP[key] ?? key] = value; + } + return out; +} + +/** Cloudflare's real OTLP export stamps the resource `service.name` with + * the deployed Worker's own SCRIPT name (`handsontable-demos-api`), never + * the contract's short name (§3's closed `SERVICE_NAMES` set). Left + * unmapped, `apiFingerprintFeed` below (which checks + * `finalResourceAttrs["service.name"] === "demos-api"` exactly) could + * never match a real record. Strips the shared `handsontable-` prefix + * before `hoistAttributes`, strictly after the `bodyJsonAttrs` spread, so + * a body-JSON key can never win this remap either. */ +const CLOUDFLARE_SCRIPT_NAME_PREFIX = "handsontable-"; +const CONTRACT_SERVICE_NAMES: ReadonlySet = new Set(SERVICE_NAMES); + +function remapCloudflareServiceName(attrs: Record): Record { + const raw = attrs[ATTR_SERVICE_NAME]; + if (typeof raw !== "string" || !raw.startsWith(CLOUDFLARE_SCRIPT_NAME_PREFIX)) return attrs; + const stripped = raw.slice(CLOUDFLARE_SCRIPT_NAME_PREFIX.length); + if (!CONTRACT_SERVICE_NAMES.has(stripped)) return attrs; + return { ...attrs, [ATTR_SERVICE_NAME]: stripped }; +} + +/** A Worker's own structured `console.log(JSON.stringify({...}))` line + * (`lines.ts`'s shape) arrives through Cloudflare's OTLP export as opaque + * BODY TEXT — without this, those fields never reach Loki as structured + * metadata (ADR §E.4). Parsed here and merged into the same attribute bag + * a true OTLP attribute lands in. A body-JSON key must never SPOOF a real + * resource attribute: every `RESOURCE_ATTRS` key is stripped from this + * function's output, and the call site (`toIngestItem`) additionally + * gives the body-JSON bag the LOWEST merge priority. */ +const RESOURCE_ATTR_KEY_SET = new Set(RESOURCE_ATTRS.map((a) => a.key)); + +/** ADR §A: "Tier-2 container stdout lands in the API worker's logs" — the + * same export this function parses. `lines.ts` stamps every one of its + * own lines with a closed-set `"log.kind"` sentinel; a body missing that + * exact marker stays opaque body text (contract §3's "authored code … + * console output" rule). + * + * Known gap (ADR §M): the sentinel is body text, not cryptographically + * bound to `lines.ts` — authored stdout that prints the same shape is + * indistinguishable here if it ever reaches this Worker's own + * `console.log` (unconfirmed on a real account). Accepted: a forged line + * can only mint an `fp:` entry and a notify-only Slack line — never + * Sentry, PII or code execution. */ +const TRUSTED_BODY_JSON_LOG_KINDS: ReadonlySet = new Set(["api.request", "error"]); + +function tryParseJsonBodyAttrs(body: string): Record { + if (!body || body.trimStart()[0] !== "{") return {}; + let parsed: unknown; + try { + parsed = JSON.parse(body); + } catch { + return {}; + } + if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) return {}; + const obj = parsed as Record; + if (!TRUSTED_BODY_JSON_LOG_KINDS.has(String(obj["log.kind"]))) return {}; + const out: Record = {}; + for (const [key, value] of Object.entries(obj)) { + if (RESOURCE_ATTR_KEY_SET.has(key)) continue; // never let body content spoof a resource attribute + if (value === null || value === undefined) continue; + if (typeof value === "string") out[key] = value; + else if (typeof value === "number" || typeof value === "boolean") out[key] = String(value); + // Nested objects/arrays inside the JSON body (none in lines.ts's own + // shape today) are skipped — attributes are flat strings only, same + // rule a real OTLP attribute already follows. + } + return out; +} + +/** + * Feeds the API worker's own `error.handled`/diagnostic reports into + * `InboxWriter`'s first-seen registry, so a server-side failure class + * notifies too once `SENTRY_SCOPE` flips to `uncaught`. + * + * All four conditions must hold: the REAL resource `service.name` is + * `demos-api` (never a body-JSON key); `log.kind` is `"error"`; the value + * matches contract §7's shape via the same validator the browser path + * uses; and the record is not Tier-2 container stdout, by construction — + * not fully guaranteed (see the known-gap doc above). Deliberately NOT + * `hot.surface !== "demo-runtime"`: that defaults to `"none"` for a + * worker-tenant record, which would admit anything under the same test. + */ +const API_FINGERPRINT_LOG_KIND = "error"; + +function apiFingerprintFeed( + bodyJsonAttrs: Record, + finalResourceAttrs: Record, +): string | undefined { + const candidate = bodyJsonAttrs["hot.fingerprint"]; + if (finalResourceAttrs[ATTR_SERVICE_NAME] !== "demos-api") return undefined; + if (bodyJsonAttrs["log.kind"] !== API_FINGERPRINT_LOG_KIND) return undefined; + if (typeof candidate !== "string" || !isValidFingerprint(candidate)) return undefined; + return candidate; +} + +export interface OtlpProcessResult { + items: IngestItem[]; + /** Records decoded but dropped (over the 256 KB cap) — accounted as + * `dropped`/`reason=size` by the caller, not `invalid_item`: these are + * well-formed, just too large. */ + droppedOversize: number; +} + +async function toIngestItem( + resourceLogs: DecodedResourceLogs, + record: DecodedLogRecord, + env: Env, + receivedAtMs: number, +): Promise { + // `bodyJsonAttrs` merges with the LOWEST priority of the three — a real + // resource or OTLP record attribute must always win over anything + // inferred from body text (RESOURCE_ATTR_KEY_SET above is the second, + // independent layer of that same guarantee). + const bodyJsonAttrs = tryParseJsonBodyAttrs(record.body ?? ""); + const merged = remapCloudflareServiceName( + remapCloudflareKeys({ ...bodyJsonAttrs, ...resourceLogs.resourceAttributes, ...record.attributes }), + ); + const hoisted = hoistAttributes(merged); + const resourceAttributes = sanitizeResourceAttributes(hoisted.resourceAttributes); + const attributes = hoisted.attributes; + + const scrubbable: ScrubbableOtlpRecord = { body: record.body, attributes, resourceAttributes }; + const scrubbed = scrubTelemetry(scrubbable) as ScrubbableOtlpRecord; + + const finalResourceAttrs = withResourceAttrDefaults(scrubbed.resourceAttributes ?? {}, env); + + const rawEventTime = isRealTimestamp(record.timeUnixNano) + ? record.timeUnixNano + : isRealTimestamp(record.observedTimeUnixNano) + ? record.observedTimeUnixNano + : undefined; + const timeUnixNano = rawEventTime ?? msToUnixNano(receivedAtMs); + + const normalised: NormalisedRecord = { + body: scrubBodyText(scrubbed.body ?? ""), + timeUnixNano, + resourceAttributes: finalResourceAttrs, + attributes: scrubbed.attributes, + severityText: record.severityText, + }; + + if (new TextEncoder().encode(JSON.stringify(normalised)).length > INBOX_RECORD_MAX_BYTES) return "oversize"; + + const hash = await hashRecord({ + body: normalised.body, + resourceAttributes: normalised.resourceAttributes, + attributes: normalised.attributes ?? {}, + rawEventTime: rawEventTime ?? "", + }); + const fingerprint = apiFingerprintFeed(bodyJsonAttrs, finalResourceAttrs); + return { hash, record: normalised, fingerprint }; +} + +/** Decodes and processes an already-size-capped OTLP export body (JSON or + * protobuf, by `contentType`) into ready-to-store {@link IngestItem}s. Never + * clamps a timestamp (see the file header); always fills the §3 resource + * attribute defaults (`withResourceAttrDefaults`). Throws only on a body + * that cannot be decoded at all (malformed JSON/protobuf) — the caller + * turns that into a `400`, never a `500`. */ +export async function processOtlpBody( + bytes: Uint8Array, + contentType: string, + env: Env, + receivedAtMs: number, +): Promise { + const isProtobuf = contentType.toLowerCase().includes("protobuf"); + const decoded = isProtobuf + ? decodeOtlpProtobuf(bytes) + : decodeOtlpJson(new TextDecoder().decode(bytes)); + + const items: IngestItem[] = []; + let droppedOversize = 0; + for (const resourceLogs of decoded) { + for (const record of resourceLogs.logRecords) { + const result = await toIngestItem(resourceLogs, record, env, receivedAtMs); + if (result === "oversize") droppedOversize++; + else items.push(result); + } + } + return { items, droppedOversize }; +} diff --git a/runner/workers/o11y/src/normalise/points.ts b/runner/workers/o11y/src/normalise/points.ts new file mode 100644 index 0000000000..ea788a93ff --- /dev/null +++ b/runner/workers/o11y/src/normalise/points.ts @@ -0,0 +1,113 @@ +// One place to (a) pick the Analytics Engine sink (real binding in +// production, the local ClickHouse shim in `O11Y_ENV === "local"`, per +// contract §10 — "No local emulation" for the real binding) and (b) fill in +// the eight §3 resource attributes every stored record must carry +// (ADR-0041 §L.15 needs every key present; several sources have no natural +// value for `hot.tier`/`hot.framework`/`hot.ht_major`/`hot.outcome`). + +import { + ATTR_DEPLOYMENT_ENVIRONMENT_NAME, + ATTR_HOT_FRAMEWORK, + ATTR_HOT_HT_MAJOR, + ATTR_HOT_OUTCOME, + ATTR_HOT_SURFACE, + ATTR_HOT_TIER, + ATTR_SERVICE_NAME, + ATTR_SERVICE_VERSION, + bindingSink, + clickhouseSink, + type AeSink, + type AePoint, +} from "@handsontable/demo-runtime/telemetry"; +import type { Env } from "../env.js"; + +/** `service.name` → `hot.surface` default for records with no other + * signal. `demos-api` is the fallback (o11y worker exports none of its + * own, ADR §B.6). */ +const SURFACE_BY_SERVICE_NAME: Readonly> = { + "demos-authoring": "authoring", + "demos-api": "api", + "demos-o11y": "o11y", + "demos-embed": "embed", +}; + +/** + * Fills every §3 resource-attribute key `mutable` lacks, in place. + * `"none"` for `hot.tier`/`hot.framework`/`hot.ht_major`/`hot.outcome`. + * `deployment.environment.name` falls back to `env.O11Y_ENV`; `hot.surface` + * to the table above; `service.version` to `"unknown"` (a real Cloudflare + * export carries `service.name` but never `service.version`). + */ +export function withResourceAttrDefaults( + mutable: Record, + env: Env, +): Record { + mutable[ATTR_SERVICE_VERSION] ??= "unknown"; + mutable[ATTR_DEPLOYMENT_ENVIRONMENT_NAME] ??= env.O11Y_ENV; + mutable[ATTR_HOT_SURFACE] ??= SURFACE_BY_SERVICE_NAME[mutable[ATTR_SERVICE_NAME] ?? ""] ?? "api"; + mutable[ATTR_HOT_TIER] ??= "none"; + mutable[ATTR_HOT_FRAMEWORK] ??= "none"; + mutable[ATTR_HOT_HT_MAJOR] ??= "none"; + mutable[ATTR_HOT_OUTCOME] ??= "none"; + return mutable; +} + +// Keyed by `env` object identity (`WeakMap`), not `O11Y_ENV`'s string +// value: caching by value would make every test in one process share a +// fake sink across every `env` fixture with the same `O11Y_ENV`. +const sinkByEnv = new WeakMap(); + +/** Production: `bindingSink(env.RUNNER_EVENTS)`. Local: `clickhouseSink` + * against `env.RUNNER_EVENTS_CLICKHOUSE_URL`, falling back to + * `http://localhost:8123` — must agree with + * `alerts/ae-query.ts#runAnalyticsEngineSqlApi`'s own default, or a + * non-default local ClickHouse silently gets zero points while alert + * queries read an empty table. Cached per `env` (cheap; stateless). */ +export function aeSink(env: Env): AeSink { + const cached = sinkByEnv.get(env); + if (cached) return cached; + const sink = + env.O11Y_ENV === "local" + ? clickhouseSink(env.RUNNER_EVENTS_CLICKHOUSE_URL ?? "http://localhost:8123", { + user: "default", + password: env.AE_SQL_TOKEN, + }) + : bindingSink(env.RUNNER_EVENTS); + sinkByEnv.set(env, sink); + return sink; +} + +/** `aeSink(env).writeDataPoint(point)` can throw SYNCHRONOUSLY from the + * real binding (an over-limit point) before any `.then`/`.catch` can run, + * so it's wrapped in a `try` alongside the async-rejection `.catch` below + * — both must land in the same "never throws into the caller" contract + * this function's doc comment promises. */ +function writeDataPointSafely(sink: AeSink, point: AePoint): Promise { + try { + return Promise.resolve(sink.writeDataPoint(point)).catch((err: unknown) => { + console.warn("[o11y] writeDataPoint failed:", err instanceof Error ? err.message : String(err)); + }); + } catch (err) { + console.warn("[o11y] writeDataPoint threw synchronously:", err instanceof Error ? err.message : String(err)); + return Promise.resolve(); + } +} + +/** Writes one point, never throwing into the caller: a local ClickHouse + * outage must not fail ingest (§B.1 "ingest never waits for the box"). + * `ctx.waitUntil` keeps the write off the critical path; `.catch` stops + * an unhandled-rejection log on every local run. */ +export function writePoint(env: Env, ctx: ExecutionContext, point: AePoint): void { + ctx.waitUntil(writeDataPointSafely(aeSink(env), point)); +} + +/** The same fire-and-forget write, from inside a Durable Object rather + * than a route handler — a DO method has no `ExecutionContext`, but + * `DurableObjectState` has its own `waitUntil`. Kept as a second, + * narrower-typed function rather than widening {@link writePoint}'s + * `ctx`: `ExecutionContext` also declares + * `passThroughOnException`/`tracing`/`abort`, which `DurableObjectState` + * lacks. */ +export function writePointFromDo(env: Env, ctx: DurableObjectState, point: AePoint): void { + ctx.waitUntil(writeDataPointSafely(aeSink(env), point)); +} diff --git a/runner/workers/o11y/src/normalise/read-body.ts b/runner/workers/o11y/src/normalise/read-body.ts new file mode 100644 index 0000000000..f7baccd01e --- /dev/null +++ b/runner/workers/o11y/src/normalise/read-body.ts @@ -0,0 +1,56 @@ +// The actual enforcement behind `gates/limits.ts`'s size cap (ADR §B.5): +// `Content-Length` is only ever a hint (`limits.ts#contentLengthExceeds`) — +// this reads the body incrementally and refuses once the cap is crossed, +// which also matters for a `Content-Encoding: gzip` request, since Cloudflare +// Workers do not auto-decompress an *incoming* request body (unlike a +// `fetch()` *response*): the wire bytes could be small while the decompressed +// bytes blow the cap, exactly the gzip-bomb shape a hint on the compressed +// size cannot catch. + +export class BodyTooLargeError extends Error { + constructor(maxBytes: number) { + super(`body exceeds ${maxBytes} bytes`); + this.name = "BodyTooLargeError"; + } +} + +/** Reads `req`'s body, transparently gunzipping when `Content-Encoding` + * names `gzip`, refusing (throwing {@link BodyTooLargeError}) once the + * decompressed byte count exceeds `maxBytes`. Reads only as much of the + * stream as it takes to detect the overflow — it does not first buffer the + * whole thing and check after. */ +export async function readCappedBytes(req: Request, maxBytes: number): Promise { + if (!req.body) return new Uint8Array(0); + + const encoding = req.headers.get("content-encoding"); + const stream = encoding?.toLowerCase().includes("gzip") + ? req.body.pipeThrough(new DecompressionStream("gzip")) + : req.body; + + const reader = stream.getReader(); + const chunks: Uint8Array[] = []; + let total = 0; + try { + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + total += value.byteLength; + if (total > maxBytes) throw new BodyTooLargeError(maxBytes); + chunks.push(value); + } + } finally { + reader.releaseLock(); + } + + const out = new Uint8Array(total); + let offset = 0; + for (const chunk of chunks) { + out.set(chunk, offset); + offset += chunk.byteLength; + } + return out; +} + +export async function readCappedText(req: Request, maxBytes: number): Promise { + return new TextDecoder().decode(await readCappedBytes(req, maxBytes)); +} diff --git a/runner/workers/o11y/src/normalise/respond.ts b/runner/workers/o11y/src/normalise/respond.ts new file mode 100644 index 0000000000..dee5865133 --- /dev/null +++ b/runner/workers/o11y/src/normalise/respond.ts @@ -0,0 +1,120 @@ +// The one place every route handler goes to answer a request and, in the +// same call, write the `o11y.ingest` Analytics Engine point ADR §B.5 +// requires ("every drop writes an `o11y.ingest` point with its reason") +// and §B.2 requires for an accepted/duplicate batch. Centralising this is +// also what makes ADR-0041 §L.4 provable: a batch with N accepted and M +// duplicate records writes exactly one `accepted` point (count=N) and one +// `duplicate` point (count=M) — never one point per record, which would +// make "delivered twice produces one copy" indistinguishable from "N +// separate deliveries" in a query. + +import { toAePoint, type CommonResourceAttrs } from "@handsontable/demo-runtime/telemetry"; +import type { Env } from "../env.js"; +import type { GateDrop } from "../gates/types.js"; +import { writePoint } from "./points.js"; + +/** `o11y.ingest`'s own emitter identity (§5: "Emitted by: o11y worker") — + * not derived from the request, always this Worker's own `service.*`. + * Falls back to `"dev"` under `wrangler dev`, where no deploy script sets + * `SERVICE_VERSION`. */ +export function o11ySelfIdentity(env: Env): CommonResourceAttrs { + return { + service_name: "demos-o11y", + service_version: env.SERVICE_VERSION ?? "dev", + environment: env.O11Y_ENV, + }; +} + +const JSON_HEADERS = { "content-type": "application/json" } as const; + +/** Answers a gate rejection and records it. `bytes` is the request's + * (compressed, on-wire) size when known — `0` is fine for a gate that never + * read the body. */ +export function respondDrop(env: Env, ctx: ExecutionContext, drop: GateDrop, bytes = 0): Response { + writePoint( + env, + ctx, + toAePoint( + "o11y.ingest", + { count: 1, bytes }, + { ...o11ySelfIdentity(env), reason: drop.reason, outcome: "dropped" }, + ), + ); + const headers: Record = { ...JSON_HEADERS }; + // A 429 without `Retry-After` leaves Faro guessing its back-off. + if (drop.retryAfterSeconds !== undefined) headers["retry-after"] = String(drop.retryAfterSeconds); + return new Response(JSON.stringify({ error: drop.reason }), { status: drop.status, headers }); +} + +/** One dropped record inside an otherwise-accepted batch (a + * client-controlled `outcome`/`reason`/`item.type` that fails `toAePoint`'s + * or `faroItemToRecord`'s runtime validation) — the batch itself still + * answers `2xx` for its other records; this only accounts the one item. */ +export function recordInvalidItem(env: Env, ctx: ExecutionContext, detail: string): void { + writePoint( + env, + ctx, + toAePoint( + "o11y.ingest", + { count: 1, bytes: 0 }, + { ...o11ySelfIdentity(env), reason: "invalid_item", outcome: "dropped" }, + ), + ); + console.warn("[o11y] dropped invalid item:", detail); +} + +/** One well-formed record dropped only for being over `INBOX_RECORD_MAX_BYTES` + * (256 KB, ADR §B.2 step 1) inside an otherwise-accepted batch — distinct + * from {@link recordInvalidItem} so an operator querying `o11y.ingest` by + * `reason="size"` to watch for oversized payloads actually sees something. + * Used by both the OTLP and Faro ingest paths. */ +export function recordOversizeDrop(env: Env, ctx: ExecutionContext, detail: string): void { + writePoint( + env, + ctx, + toAePoint( + "o11y.ingest", + { count: 1, bytes: 0 }, + { ...o11ySelfIdentity(env), reason: "size", outcome: "dropped" }, + ), + ); + console.warn("[o11y] dropped oversize record:", detail); +} + +/** Answers an accepted request after the `InboxWriter` commit, writing one + * `accepted` point (if any records were newly stored) and one `duplicate` + * point (if any were deduped) — `reason` is the route's short name + * (`"collect"`, `"lite"`, `"v1/logs"`, `"deploy"`, `"hooks/sentry"`), so + * `o11y.ingest`'s per-source volume is queryable the same way a dropped + * request's gate name is. */ +export function respondIngested( + env: Env, + ctx: ExecutionContext, + route: string, + counts: { accepted: number; duplicate: number }, + bytes: number, +): Response { + if (counts.accepted > 0) { + writePoint( + env, + ctx, + toAePoint( + "o11y.ingest", + { count: counts.accepted, bytes }, + { ...o11ySelfIdentity(env), reason: route, outcome: "accepted" }, + ), + ); + } + if (counts.duplicate > 0) { + writePoint( + env, + ctx, + toAePoint( + "o11y.ingest", + { count: counts.duplicate, bytes: 0 }, + { ...o11ySelfIdentity(env), reason: route, outcome: "duplicate" }, + ), + ); + } + return new Response(null, { status: 204 }); +} diff --git a/runner/workers/o11y/src/normalise/sentry.ts b/runner/workers/o11y/src/normalise/sentry.ts new file mode 100644 index 0000000000..3ed26946ac --- /dev/null +++ b/runner/workers/o11y/src/normalise/sentry.ts @@ -0,0 +1,78 @@ +// ADR §E.2: "The issue-alert webhook (new issue, regression, resolved) +// becomes a Loki line with issue id, title, release and link." Worker +// tenant. Sentry's internal-integration issue-alert payload +// (https://docs.sentry.io/product/integrations/integration-platform/webhooks/#issue-alerts): +// `{action, data: {issue: {id, shortId, title, level, permalink, ...}}}` +// roughly — every field read here is optional and falls back to `"unknown"`, +// since the exact shape is not pinned by this contract and the test fixture +// webhook replay is hand-built, not captured from a real Sentry account. + +import { msToUnixNano, scrubTelemetry, type NormalisedRecord } from "@handsontable/demo-runtime/telemetry"; +import type { Env, IngestItem } from "../env.js"; +import { hashRecord } from "./hash.js"; +import { withResourceAttrDefaults } from "./points.js"; +import { scrubBodyText } from "./text-scrub.js"; + +function str(v: unknown, fallback = "unknown"): string { + return typeof v === "string" && v.length > 0 ? v : fallback; +} + +/** No signature-level shape validation here (the HMAC gate already + * authenticated the sender) — reads defensively, never throws on a missing + * field, so a Sentry payload shape drift becomes a less-informative log + * line, never a `500`. */ +export async function processSentryPayload( + payload: unknown, + env: Env, + receivedAtMs: number, + // A fixed `rawEventTime` would mean an issue that flips regression -> + // resolved -> regression inside one 24h dedupe window (`DEDUPE_WINDOW_MS`) + // produces two records with an IDENTICAL derived body (same `action`, + // same `title`/`issueId`/`release`) — the second, genuinely new + // regression silently dedupes away. Sentry sends a real per-delivery + // `Sentry-Hook-Timestamp` header + // (https://docs.sentry.io/product/integrations/integration-platform/webhooks/#headers) + // — the caller (`index.ts`) passes it through here; falls back to the + // worker's own bucketed receive time (same scheme as `deploy.ts`) when + // the header is absent, so a hand-built payload still gets some + // per-event distinction. + rawEventTime: string | null, +): Promise { + const p = (typeof payload === "object" && payload !== null ? payload : {}) as Record; + const data = (typeof p["data"] === "object" && p["data"] !== null ? p["data"] : {}) as Record; + const issue = (typeof data["issue"] === "object" && data["issue"] !== null ? data["issue"] : {}) as Record< + string, + unknown + >; + + const action = str(p["action"]); + const issueId = str(issue["id"] ?? issue["shortId"]); + const title = str(issue["title"]); + const release = str((issue["lastRelease"] as Record | undefined)?.["version"] ?? issue["release"]); + const link = str(issue["permalink"] ?? issue["url"], ""); + + const resourceAttributes = withResourceAttrDefaults( + { "service.name": "demos-api", "service.version": release === "unknown" ? "unknown" : release }, + env, + ); + let record: NormalisedRecord = { + body: `sentry ${action}: ${title} [${issueId}] release=${release}${link ? ` ${link}` : ""}`, + timeUnixNano: msToUnixNano(receivedAtMs), + resourceAttributes, + attributes: {}, + }; + // Sentry's own issue `title`/`permalink` routinely embed a preview host + // (a session credential), a query string, an email or a user-agent, none + // of which contract §3 allows. Run the same authoritative pass every + // other ingest path runs (`lite.ts`'s own order: convert, then + // `scrubTelemetry`, then this worker's own extra text pass). + record = scrubTelemetry(record)!; + record.body = scrubBodyText(record.body); + const hash = await hashRecord({ + body: record.body, + resourceAttributes: record.resourceAttributes, + attributes: {}, + rawEventTime: rawEventTime && rawEventTime.length > 0 ? rawEventTime : String(Math.floor(receivedAtMs / 60_000)), + }); + return { hash, record }; +} diff --git a/runner/workers/o11y/src/normalise/text-scrub.ts b/runner/workers/o11y/src/normalise/text-scrub.ts new file mode 100644 index 0000000000..8603d7069c --- /dev/null +++ b/runner/workers/o11y/src/normalise/text-scrub.ts @@ -0,0 +1,63 @@ +// `scrubTelemetry`'s `stripQueryAndFragment` only runs on discrete +// URL-shaped fields, never on a message/body string that merely *contains* +// a URL — a real Cloudflare export line can embed one anyway (the Sandbox +// SDK's stale-preview-URL warning, ADR §D). This extra pass runs on every +// OTLP record's `body` (and, defensively, Faro's converted `body`) after +// `scrubTelemetry`, never instead of it. + +import { + redactIpInText, + stripQueryAndFragment, + stripUrlQueriesInText, + truncateForScrub, +} from "@handsontable/demo-runtime/telemetry"; + +// Defined once in the runtime package (ADR §E.4); re-exported since +// `pipeline/o11y-redos.test.mjs`/`o11y-normalise.test.mjs` import it here. +export { redactIpInText }; + +/** Blanks a `Mozilla/ ()` UA prefix embedded in free + * text (`scrub.ts#reduceBrowserMeta` only reduces the structured + * `meta.browser.userAgent` field). Every quantifier is bounded against + * ReDoS (measured 40k chars ~1.2s unbounded); `o11y-redos.test.mjs` pins + * the timing. */ +const USER_AGENT_PATTERN = /Mozilla\/[\d.]{1,32}\s{0,16}\([^)]{0,512}\)[^\s,;]{0,256}/gi; + +export function redactUserAgentInText(text: string): string { + return text.replace(USER_AGENT_PATTERN, ""); +} + +/** Blanks a standard email shape embedded in free text (contract §3 bans + * it). Bounded per RFC 5321 §4.5.3.1 (local ≤64, domain ≤253 octets) + * against ReDoS (measured 80k chars ~4.3s unbounded); `o11y-redos.test.mjs` + * pins the timing. */ +const EMAIL_PATTERN = /[a-z0-9._%+-]{1,64}@[a-z0-9.-]{1,253}\.[a-z]{2,63}/gi; + +export function redactEmailInText(text: string): string { + return text.replace(EMAIL_PATTERN, ""); +} + +/** The extra pass run on every stored record's free body text, beyond what + * `scrubTelemetry` guarantees. `truncateForScrub` bounds `text` to + * `SCRUB_TEXT_MAX_CHARS` (= `INBOX_RECORD_MAX_BYTES`, §8) before any pass + * runs. */ +export function scrubBodyText(text: string): string { + return redactIpInText(redactEmailInText(redactUserAgentInText(stripUrlQueriesInText(truncateForScrub(text))))); +} + +/** `scrubTelemetry`'s OTLP-record branch scrubs preview hosts from every + * attribute value but never strips a query string or UA the way + * `scrubBodyText` does for `body`. Uses `stripQueryAndFragment`, not + * `stripUrlQueriesInText`: an attribute value is the WHOLE field, not + * text that merely *embeds* a URL, so it gets `scrub.ts`'s + * `meta.page.url` treatment. */ +export function scrubAttributeValues( + attrs: Record | undefined, +): Record | undefined { + if (!attrs) return attrs; + const out: Record = {}; + for (const [key, value] of Object.entries(attrs)) { + out[key] = redactIpInText(redactEmailInText(redactUserAgentInText(stripQueryAndFragment(truncateForScrub(value))))); + } + return out; +} diff --git a/runner/workers/o11y/src/router.ts b/runner/workers/o11y/src/router.ts new file mode 100644 index 0000000000..edd034dd4d --- /dev/null +++ b/runner/workers/o11y/src/router.ts @@ -0,0 +1,59 @@ +// COMMON.md pinned interface 2. `registerRoute(method, path, handler)` — +// callers plug their own routes in through this, never by editing +// `index.ts`'s dispatch logic directly. + +import type { Env } from "./env.js"; + +export type RouteMethod = "GET" | "POST" | "*"; +export type RouteHandler = (req: Request, env: Env, ctx: ExecutionContext) => Promise; + +interface Route { + method: RouteMethod; + path: string; + isPrefix: boolean; + handler: RouteHandler; +} + +const routes: Route[] = []; + +/** `path` is exact, or a prefix when it ends in `/*`. Throws on a duplicate + * `method`+`path` registration — a silent second registration shadowing + * the first would be a much harder bug to find than a boot-time throw. */ +export function registerRoute(method: RouteMethod, path: string, handler: RouteHandler): void { + const isPrefix = path.endsWith("/*"); + if (routes.some((r) => r.method === method && r.path === path)) { + throw new Error(`registerRoute: duplicate registration for ${method} ${path}`); + } + routes.push({ method, path, isPrefix, handler }); +} + +function matches(route: Route, method: string, pathname: string): boolean { + if (route.method !== "*" && route.method !== method) return false; + if (route.isPrefix) return pathname.startsWith(route.path.slice(0, -1)); + return pathname === route.path; +} + +/** + * Finds the best match for `method`/`pathname`: an exact-path route beats + * every prefix route, and among prefix routes the longest `path` wins (so + * `/grafana/_o11y/reopen` beats `/grafana/*` regardless of registration + * order). + */ +export function findRoute(method: string, pathname: string): RouteHandler | null { + let best: Route | null = null; + for (const route of routes) { + if (!matches(route, method, pathname)) continue; + if (best === null) { + best = route; + continue; + } + const bestIsExact = !best.isPrefix; + const routeIsExact = !route.isPrefix; + if (routeIsExact && !bestIsExact) { + best = route; // exact beats prefix + } else if (routeIsExact === bestIsExact && route.path.length > best.path.length) { + best = route; // longer prefix (or, among exacts, cannot tie — unique per method+path) + } + } + return best?.handler ?? null; +} diff --git a/runner/workers/o11y/tsconfig.json b/runner/workers/o11y/tsconfig.json new file mode 100644 index 0000000000..c25c7ff496 --- /dev/null +++ b/runner/workers/o11y/tsconfig.json @@ -0,0 +1,9 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "noEmit": true, + "types": ["@cloudflare/workers-types"], + "lib": ["ES2022"] + }, + "include": ["src"] +} diff --git a/runner/workers/o11y/wrangler.jsonc b/runner/workers/o11y/wrangler.jsonc new file mode 100644 index 0000000000..1d468f8cb2 --- /dev/null +++ b/runner/workers/o11y/wrangler.jsonc @@ -0,0 +1,110 @@ +{ + // The observability Worker (ADR-0041 rev. 3) — o11y worker + InboxWriter DO + + // GrafanaBox DO/Container. Deployed to the main Handsontable Cloudflare + // account (NOT the sandbox), same account as workers/api. + "$schema": "node_modules/wrangler/config-schema.json", + "name": "handsontable-demos-o11y", + "account_id": "15111272c53ed0aaf84a908f0c9c7f8b", + "main": "src/index.ts", + "compatibility_date": "2026-09-01", + + // No workers.dev URL, no preview URLs (contract §1) — every request reaches + // this Worker through the `--routes` flags in the `deploy` script + // (package.json), never a Cloudflare-issued subdomain. + "workers_dev": false, + "preview_urls": false, + + // ADR §B.6: the observer does not observe itself — no exported Workers + // Logs, no traces, invocation logs off. Self-metrics go to Analytics + // Engine like every other metric, never through its own ingest routes. + "observability": { + "enabled": true, + "logs": { "enabled": true, "invocation_logs": false, "persist": true, "destinations": [] }, + "traces": { "enabled": false } + }, + + // ADR §B.3: drain alarm loop budget, at most 300000 per ADR. A single + // `drainStep` pushes up to `DRAIN_BATCH_SIZE` (10) objects through + // gzip/hash/symbolication; 30s measured too tight on the sandbox probe. + "limits": { "cpu_ms": 120000 }, + + // ADR §A: the backlog cron; the alert evaluation cron extends the same + // `scheduled()` handler rather than adding a second trigger. + "triggers": { "crons": ["*/10 * * * *"] }, + + // Public, non-secret config (contract §2). + "vars": { + "O11Y_ENV": "production", + // The Handsontable login broker's base URL (ADR-0007), same value as + // workers/api/wrangler.jsonc's own var. Public, not a secret — every + // consumer of this broker already commits it the same way. + "LOGIN_BROKER_URL": "https://mcp-auth-proxy-j0tb.onrender.com", + "GITHUB_OIDC_REPOSITORY": "handsontable/examples", + // The expected `workflow_ref` OIDC claim for the deploy job — must be + // kept in sync with the real deploy workflow's file path and the ref + // it runs from. + "GITHUB_OIDC_WORKFLOW_REF": "handsontable/examples/.github/workflows/master.yml@refs/heads/master", + // Duplicates the `account_id` above because a Worker has no runtime way + // to read its own account id, and GrafanaBox needs it to build the + // Loki bucket's R2 S3 endpoint and the Analytics Engine SQL API URL. + "CLOUDFLARE_ACCOUNT_ID": "15111272c53ed0aaf84a908f0c9c7f8b" + // SERVICE_VERSION is intentionally absent: the real deploy script sets + // it via `--var SERVICE_VERSION:$GITHUB_SHA`, not a committed value — + // every reader falls back to `"dev"` when unset. + }, + + // R2 — EU jurisdiction on every bucket (contract §2). Bucket names must exist + // once R2 is enabled for this account (`wrangler r2 bucket create --jurisdiction eu `), + // the same one-time step `AGENTS.md` notes for the API worker's bucket. + "r2_buckets": [ + { "binding": "O11Y_INBOX", "bucket_name": "handsontable-demos-o11y-inbox", "jurisdiction": "eu" }, + { "binding": "O11Y_LOKI_STATE", "bucket_name": "handsontable-demos-o11y-loki", "jurisdiction": "eu" }, + { "binding": "O11Y_MAPS", "bucket_name": "handsontable-demos-o11y-maps", "jurisdiction": "eu" } + ], + + // Analytics Engine — `runner_events` (contract §4), shared with the API + // worker's own `RUNNER_EVENTS` binding. + "analytics_engine_datasets": [{ "binding": "RUNNER_EVENTS", "dataset": "runner_events" }], + + // o11y usage metering/spend, later `AdminReads` (ADR-0043). Bound to the + // named `O11yUsage` `WorkerEntrypoint`, not the default export — a named + // RPC entrypoint has no HTTP route, unreachable even though the API + // worker's own deploy routes are the wildcard `*.demos.handsontable.com/*`. + "services": [{ "binding": "API", "service": "handsontable-demos-api", "entrypoint": "O11yUsage" }], + + // `INBOX_WRITER` (dedupe, fingerprint registry, pack/commit) and + // `GRAFANA_BOX`. EU jurisdiction for both is a *runtime* call + // (`env.INBOX_WRITER.jurisdiction("eu")`), not a wrangler.jsonc field. + "durable_objects": { + "bindings": [ + { "class_name": "InboxWriter", "name": "INBOX_WRITER" }, + { "class_name": "GrafanaBox", "name": "GRAFANA_BOX" } + ] + }, + "migrations": [{ "new_sqlite_classes": ["InboxWriter", "GrafanaBox"], "tag": "v1" }], + + // `image` is a path to the Dockerfile (not its directory). EU jurisdiction + // is `constraints.jurisdiction`, not a top-level `jurisdiction` key + // (ADR-0041 §A/§H). `standard-1` = 1/2 vCPU, 4 GiB memory, 8 GB disk, + // matching `containers/o11y/compose.yml`'s local limits. `max_instances: 1`: + // exactly one box at a time (ADR §A "one sleeping box"). `scheduling_policy` + // deliberately left unset, matching Cloudflare's own Containers example. + "containers": [ + { + "class_name": "GrafanaBox", + "image": "../../containers/o11y/Dockerfile", + "instance_type": "standard-1", + "max_instances": 1, + "constraints": { "jurisdiction": "eu" } + } + ], + + // ADR §B.5's rate-limiting binding, gating `collect`/`lite`. + // `ratelimits[].namespace_id` is a free-form string the deploy author + // picks, not a dashboard-provisioned id. `1001` is scoped to this Worker + // only. `100` req/`60`s per `cf-connecting-ip`: an authoring tab flushes + // every 5 s, an embed view sends about 0.4 (runbook "Ingest rate limit"). + "ratelimits": [ + { "name": "RATE_LIMITER", "namespace_id": "1001", "simple": { "limit": 100, "period": 60 } } + ] +}