From f8d48925862d329b95480f56fa6bfcadf874e06a Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 16:22:34 +0800
Subject: [PATCH 01/22] docs: plan browser-run checker and interactive Test
Co-Authored-By: Claude Opus 5.5
---
.../plans/2026-10-07-browser-test-judge.md | 216 ++++++++++++++++++
...0-07-browser-test-judge-programs-design.md | 172 ++++++++++++++
2 files changed, 388 insertions(+)
create mode 100644 docs/superpowers/plans/2026-10-07-browser-test-judge.md
create mode 100644 docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
diff --git a/docs/superpowers/plans/2026-10-07-browser-test-judge.md b/docs/superpowers/plans/2026-10-07-browser-test-judge.md
new file mode 100644
index 000000000..7fd9707df
--- /dev/null
+++ b/docs/superpowers/plans/2026-10-07-browser-test-judge.md
@@ -0,0 +1,216 @@
+# Browser checker and interactive Test: implementation plan
+
+> **For Claude:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development to implement this plan task by task.
+
+**Goal:** Checker and interactive problems get Test verdicts computed entirely in the browser. The browser compiles and runs the problem's own checker or interactor. The server keeps only a source read endpoint. The server-side half of #641 and the `hidden` workspace visibility go away, with no feature flag, no fallback and no dead code left behind.
+
+**Spec:** `docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md`.
+
+**Owner decisions (2026-10-07):**
+- No authoring check and no notice beyond a single "students can read this program" line; authors test their own problems.
+- Interactive ships together with checker. No "not available yet" state.
+- Judge programs compile in the student's browser. No test worker.
+- No `TEST_JUDGE_ENABLED` and no fallback.
+- Endpoint authorization is problem view access.
+- Mobile follows the existing toolchain preload.
+- The "Judged on the server" badge and the busy banner go.
+
+**Upstream gate:** merge only after a `@wasm-oj` release with browser `interact` passing `startupEntropyBytes` (new forge PR), #93 (Python interactors) and #95 (interactive metering). Until then, develop against locally built forge packages. Never patch forge inside NOJV.
+
+**Branch:** `feat/browser-test-judge`, cut from origin/main.
+
+---
+
+## Task 0: Forge PR (parallel, outside NOJV)
+
+- Branch from forge `main` in the session scratchpad's forge clone.
+- In `src/runtime/runner.worker.ts` `interactiveCoreProgram`, pass `startupEntropyBytes: prepared.startupEntropyBytes`.
+- Add a browser test that runs `interact` end to end. A C interactor is enough. Use the repo's existing browser test harness, e.g. the strict-CSP suite or the Vitest browser setup.
+- Open the PR from TakalaWang/forge; CODEOWNERS requests JacobLinCool.
+- Comment on #93 and #95 that NOJV now needs them for browser interactive Test.
+
+## Task 1: Remove the server side
+
+Reapply commit `6bdacae9` from `feat/test-execution-only` with `git cherry-pick -n`, then review every hunk. It removes:
+- the test worker, its workflows, queue partitions, chart and image layers, plus `@wasm-oj/server` and its patch;
+- the Test API, the Redis lock, the limiter and the application test-judge module;
+- `TEST_JUDGE_ENABLED`;
+- the core request, record and cache-key code, and `build-artifact-wire`;
+- the storage and Redis keys;
+- the worker env vars;
+- the edit page's build status and `checkSamples` action;
+- their tests.
+
+Fix up after the cherry-pick:
+- **Restore in core** what the browser needs:
+ - `judge/test-judge-verdict.ts` (`checkerCaseVerdict`, `interactiveCaseVerdict`, `truncateUtf8`);
+ - `judge/wasm-oj-verdict.ts`;
+ - `judge/python-judge-wrappers.ts`;
+ - `judgeProgramCompileInput` and `JudgeProgramSource`, without `WASM_OJ_SERVER_*` or the cache key;
+ - `interactiveContestantSupported`.
+
+ Restore their tests too: `test-judge-verdict`, `wasm-oj-verdict`, `judge-program-sources`, and the compile-input part of `test-judge-program`.
+- **Capability:** keep a simplified `staticTestCapability({ isSpecialEnv })`, or inline it if it is only `special_env`. Make sure nothing still takes `testJudgeEnabled` or `judgeLanguage`.
+- **WEB-05 coverage:** move the contest participation and contest-window cases from the deleted `test-judge-domain.test.ts` (lines 527–715 on main) onto `listCodeDrafts`.
+- **`assertProblemContextAllowed`:** keep it unexported unless something outside `code-draft.ts` uses it.
+- **Lockfile:** `pnpm install --frozen-lockfile` must pass with a minimal lock diff.
+- **Renovate:** restore what `.github/renovate.json` said about `@wasm-oj/*` before #641, and drop the rust-image rule.
+- **Messages:** delete the keys only the removed UI used: `admin_checkSamples*`, `admin_checkingSamples`, `admin_judgeProgram*`, `admin_testJudgeDisabled`, `admin_testPythonInteractorUnsupported`, `editor_testJudgeBusy`, `editor_testTooLarge`, `editor_judgingOnServer`, `editor_judgedOnServer`. Grep each key before deleting it.
+
+## Task 2: Judge-program source endpoint
+
+- **Application:** add `getJudgeProgramSource(userId, problemId, context)` next to the problem queries, not in a new test-judge module.
+ - Authorization is the same view access the problem page uses for that context.
+ - The source is read through the verified script pointer.
+ - It returns `{ role, language, source, sha256 }`, or 404 for standard and `special_env` problems.
+- **Web:** add `apps/web/src/routes/api/problems/[id]/judge-program/+server.ts`, using `apiHandler` with the standard limiter. Parse the context the way `/api/drafts` does.
+- Register the route in `tests/unit/security/exam-confinement-api-allowlist.test.ts` as `exam-scoped`.
+- **Tests:** application cases for practice, assignment membership, a running exam, an ended exam or contest where the problem is still viewable, a forbidden context, 404s, and a corrupt pointer. Add an HTTP integration test for the route.
+
+## Task 3: Browser judge-program preparation
+
+- **New `apps/web/src/lib/services/judge-program.ts`:**
+ - `prepareJudgeProgram(problemId, context)` fetches the source and builds `judgeProgramCompileInput(source, WASM_OJ_LIBCXX_PCH_HEADER)` with the shared browser engine;
+ - the result is memoized per `problemId` + `sha256` for the page session;
+ - it exposes progress and returns `{ ok: true, artifact } | { ok: false, diagnostics }`.
+ - It preloads the judge program's toolchain(s) alongside the student's: clang for C++, the Python runtime for Python.
+ - On each editor mount it refetches the source and rebuilds only if `sha256` changed.
+- **`Editor.svelte` and `EditorActionBar.svelte`:**
+ - start preparation in the existing toolchain-preload effect;
+ - the Test button shows a preparing label (reuse the toolchain progress style) and waits;
+ - a build failure disables Test with "This problem's checker/interactor failed to build", and the panel shows the diagnostics;
+ - a fetch failure disables Test with "Couldn't load this problem's checker/interactor" after the existing retry pattern.
+- **Tests:**
+ - unit tests for the service with a fake engine and fetch: Python, C++, a failing build, memoization and a `sha256` change;
+ - component tests for the button states.
+
+## Task 4: Browser checker Test
+
+- **`browser-local-run.ts`:** add `runBrowserChecker(artifact, { input, answer, output, timeLimitMs })`:
+ - args `/judge/input /judge/answer /judge/feedback`;
+ - files `/judge/input`, `/judge/answer` and `/judge/feedback/.keep`;
+ - stdin = the student's output;
+ - output path `/judge/feedback/teammessage.txt`;
+ - `validatorTimeoutMs`, 512 MiB;
+ - result mapped through `checkerCaseVerdict`.
+- **`use-editor-run.svelte.ts`:**
+ - Delete `requestTestJudge` use, the server error codes, `TEST_DISABLING_CODES`, `UNJUDGED_SAMPLE_CODES`, `serialiseBuildArtifact` and `MAX_CASE_STDOUT_BYTES` truncation.
+ - Checker path: compile, run the samples, then run the checker on every sample that exited normally. Map run cases to samples by input, as today. Custom cases stay execution-only.
+- **`submission-service.ts`:** delete `requestTestJudge` and `testJudgeErrorCode`.
+- **`EditorBottomPanel.svelte`:** remove the server badge and the `serverNotice` banner. Gate the SE explanation on a browser "judged" flag. Keep the `teammessage` and Executed blocks.
+- **`apps/web/src/lib/types/index.ts`:** drop `serverJudged` and `serverNotice`. The transcript type becomes local if the core schema is gone.
+- **Tests:**
+ - `editor-client-test`: replace the server-request cases with browser AC, WA and SE, plus custom execution-only cases;
+ - `editor-output-comparison`: drop the server badge and notice cases;
+ - add a unit test for `runBrowserChecker` wiring.
+
+## Task 5: Browser interactive Test
+
+- **`browser-local-run.ts`:** add `runBrowserInteraction(contestant, interactor, { interactorInput, limits })` with `engine.interact`. Use the contestant limits and the official interactor limits, and map through `interactiveCaseVerdict`. A wall-limit stop is TLE. Cap the transcript (move `TEST_JUDGE_TRANSCRIPT_BYTES` into web if core no longer needs it).
+- **`use-editor-run.svelte.ts`:** the interactive path compiles the contestant, then runs every selected sample (`interactorInput`) and every custom case through the interactor.
+- **Custom cases on interactive problems:** the case editor's input field is the interactor input. Use the existing interactor-input label and the `customCasesAllowed` switch. Reword `editor_interactiveTestNote` so it no longer says custom cases are impossible.
+- JS/TS contestants stay disabled with the existing reason (`interactiveContestantSupported`).
+- **Tests:**
+ - unit tests for the interaction wiring with a fake engine;
+ - component tests for the transcript and for interactive custom cases;
+ - the real browser check in Task 8.
+
+## Task 6: Judge-tab note
+
+Add one line in the checker and interactor sections of `JudgeTab.svelte`, editors only: "Students can read this program when they press Test." Reword `admin_checkerHelpBody` and `admin_interactorHelpBody` so they no longer imply the program only ever runs in an isolated container. Add a small component test (none exists for `JudgeTab`).
+
+## Task 7: Remove `hidden` workspace visibility
+
+- **Migration:** follow `20260708000000_drop_userstatus_disabled`:
+ 1. `UPDATE … SET visibility='readonly' WHERE visibility='hidden'`;
+ 2. create the new enum;
+ 3. `ALTER COLUMN … USING`;
+ 4. rename;
+ 5. drop the old enum.
+
+ Add the `-- expand-contract-ok:` line for `scripts/check-migrations.mjs`.
+- **Schema and docs:** update `problem.prisma` and regenerate `DATABASE.generated.md`.
+- **Core:** `workspaceFileVisibilitySchema` becomes `["editable", "readonly"]`. The judge-snapshot parser maps a legacy `hidden` to `readonly`, and drops its duplicate pointer schema if one still exists.
+- **Application:** drop the content blanking in `details.ts` and the type in `workspace.ts`.
+- **UI:**
+ - `StudentProblemView.svelte`: lock prefix, third badge option and hidden placeholder;
+ - `editor-bindings.ts`: the filter becomes language-only; rename `publicFiles`;
+ - `ReferenceSolutionSection.svelte`: the filter;
+ - `WorkspaceFileEditor.svelte`: the option and type;
+ - `WorkspaceFileList.svelte`: the fallback icon;
+ - `WorkspaceFilesSection.svelte`: the warning;
+ - `types/index.ts`.
+- **Messages:** delete `admin_fileHidden`, `admin_workspaceHiddenTestNote`, `workspace_fileHidden*` and `workspace_visibilityHidden`. Reword `admin_workspaceFilesHintMultiFile` and `problemDetail_multiFileHelp`.
+- **Tests:**
+ - fixtures in `merge-sandbox-sources`, `db-read-model` (the blanking test goes), `problem-page-data`, `problem-queries`, `editor-workspace-parity`, `workspace-section`, `editor-client-test` and `tests/e2e/course-problem-library.test.ts`;
+ - a migration test;
+ - a legacy snapshot test.
+
+## Task 8: Docs, decisions, verification
+
+**Decisions:**
+- JDG-15 is rewritten per the spec.
+- JDG-26, SEC-15 and OPS-21 are withdrawn. Keep each heading and index line; `27fd8226` has the format.
+- JDG-05 is scoped to official judging.
+- Edit JDG-12, PRB-03, PRB-22, WEB-05 and OPS-18.
+- PRB-01 and PRB-09 drop `hidden`.
+- Every ID token in the README must still match a heading (`tests/unit/docs/doc-links.test.ts`).
+
+**Living docs:**
+- `ARCHITECTURE.md`, `JUDGE_PIPELINE.md` (Test sections and the visibility table), `FRONTEND.md`, `REDIS.md`, `DATABASE.md`;
+- `DEPLOYMENT.md`, `RELIABILITY.md`, `SECURITY.md`, `THREAT_MODEL.md`;
+- `QUALITY_SCORE.md`: drop the server-Test items, and add the Wasm official-judging fast path as an item to evaluate;
+- `PRODUCT_SENSE.md`, `docs/features/problem-test.md` (rewrite), `docs/features/contests.md`;
+- runbooks `judge-queue.md` and `getting-started.md`;
+- the READMEs of the worker, chart, storage and temporal, and `.env.example`.
+
+Delete this plan and the spec.
+
+**Verification:**
+1. `pnpm ci:verify` (long; run it in the background with a long timeout), `pnpm lint:helm` and `pnpm install --frozen-lockfile`.
+2. Run the touched integration tests against an isolated Postgres and Redis.
+3. **Browser check** on a seeded local stack: the worktree dev server on port 5174 with an isolated database, and forge built locally until the release, then the release. Take screenshots.
+ - `any-two-sum` and `shortest-route-plan`: AC and WA, with `teammessage`.
+ - A C++ checker fixture, and a broken one: Test disabled, diagnostics shown.
+ - `guess-the-number`, `multi-interactive-bisect`, `noisy-oracle-hunt` and `interactive-peak` with C++ and Python contestants: AC, WA, an infinite loop that TLEs at the instruction budget, and a custom interactor input.
+ - A JS contestant on an interactive problem: disabled.
+ - A multi-file problem: no `hidden` option.
+ - A standard problem: unchanged.
+ - A student on an ended exam's problem that is still viewable: checker Test works.
+4. A worker image build without the WASM-OJ layers.
+5. Run `git grep` over the dead-code checklist below; expect zero hits.
+
+## Cleanup of earlier attempts
+
+- **Code:** the only test worker that ever landed is #641's `worker-test` Deployment. Task 1 removes it together with its `K8S_RUNTIME_CLASS_NAME` env, service account and RBAC, PDB entry, values and image layers. The gVisor executor attempt of 2026-10-07 was stopped before any commit, and its branch is deleted.
+- **Production:** nothing to undo. On 2026-10-07 no `worker-test`, executor, Service or NetworkPolicy existed in production (read-only `kubectl` check), because #641 was never released.
+- **Local:** after this PR merges, delete the merged or abandoned worktrees and branches:
+ - `feat/checker-interactive-test` (#641);
+ - `fix/source-map-js-audit` (#642);
+ - `feat/test-execution-only` (stopped; its server-removal commit is reapplied in Task 1);
+ - `docs/checker-interactive-test-spec`, after switching the main checkout back to `main`.
+
+## After merge
+
+1. Bump to the forge release, then release NOJV.
+2. The seed rows need the manual admin edits #641 listed:
+ - checker re-uploads for `any-two-sum`, `course-order` and `shortest-route-plan`;
+ - interaction notes, interactor inputs and transcripts for the four interactive problems.
+
+ Restate them in the PR body.
+3. No backfill is needed: nothing is precompiled.
+
+---
+
+## Dead-code checklist (`git grep` must find nothing outside history)
+
+| Area | Must be gone |
+| --- | --- |
+| Worker | `WORKER_MODE.*test`, `nojv-worker-test`, `TEST_JUDGE_SLOTS`, `WASM_OJ_RUNTIME_DIR`, `WASM_OJ_TOOLCHAIN_DIR`, `WASM_OJ_CACHE_DIR`, `@wasm-oj/server`, `@wasm-oj__server` |
+| Temporal | `test-judge` queue, `testJudgeWorkflow`, `testJudgeProgramBuildWorkflow`, `runTestJudgeWorkflow`, `dispatchTestJudgeProgramBuild`, `TEST_JUDGE_TASK_QUEUE` |
+| Application | `runTestJudge`, `buildTestJudgeProgram`, `withUserTestJudgeLock`, `checkSamplesWithChecker`, `getJudgeProgramStatus`, `isTestJudgeEnabled`, `TEST_JUDGE_ENABLED` |
+| Storage, Redis, limiter | `test-judge-requests/`, `test-judge-programs/`, `testJudgeRequestKey`, `testJudgeProgramKey`, `testJudgeInFlight`, `rl:test-judge`, `testJudgeApiHandler` |
+| Core | `testJudgeRequestSchema`, `testJudgeResponseSchema`, `testJudgeStoredRequestSchema`, `storedJudgeProgramSchema`, `testJudgeProgramCacheKey`, `WASM_OJ_SERVER_IDENTITY`, `serialiseBuildArtifact`, `boundedTestJudgeOutput` |
+| Web | `requestTestJudge`, `serverJudged`, `serverNotice`, `JudgeProgramTestStatus`, `checkSamples` |
+| Message keys | `admin_checkSamples*`, `admin_judgeProgram*`, `admin_testJudgeDisabled`, `editor_testJudgeBusy`, `editor_testTooLarge`, `editor_judgingOnServer`, `editor_judgedOnServer`, `admin_fileHidden`, `admin_workspaceHiddenTestNote`, `workspace_fileHidden*`, `workspace_visibilityHidden` |
+| Infra | `worker-test`, `test-executor`, `TEST_JUDGE_EXECUTOR`, `worker-test.deployment.yaml`, `infra/docker/wasm-oj-toolchains`, the WASM-OJ stages in `worker.Dockerfile`, `"hidden"` as a workspace visibility |
diff --git a/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md b/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
new file mode 100644
index 000000000..5da6071c7
--- /dev/null
+++ b/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
@@ -0,0 +1,172 @@
+# Browser Test runs checkers and interactors
+
+**Status:** Direction decided by the owner on 2026-10-07, after the final review round that day · **Touches:** JDG-03, JDG-05, JDG-12, JDG-15, JDG-26, OPS-18, OPS-21, PRB-01, PRB-03, PRB-09, PRB-22, SEC-12, SEC-15, WEB-05
+
+## Problem
+
+#641 (merged 2026-10-06, not released) added Test for checker and interactive problems by running the problem's checker or interactor on a server test worker. That worker runs TA-authored judge programs, plus the student's Wasm on interactive problems. The only thing between that code and the container's object-storage keys (every problem's hidden testcases) and Redis URL is the Wasm runtime.
+
+The owner's direction:
+- Test is the student's own run, so all of it happens in the student's browser: the judge program is compiled and run there too.
+- Judge programs become readable by students. Keeping them robust is the authors' responsibility, and authors test their own problems.
+- For Test, the server keeps nothing but a read endpoint for the judge program's source.
+- The `hidden` workspace visibility goes away; it never gave confidentiality.
+- Checker and interactive problems ship together, with no feature flag and no fallback.
+
+## Goals
+
+- Test on checker and interactive problems shows verdicts (AC/WA/TLE/RE/SE), the checker's `teammessage` and the interaction transcript, all computed in the browser.
+- The server executes and compiles nothing for Test. The test worker, its queue, build cache, Test API and the worker image's WASM-OJ layers are removed.
+- The judge program is prepared in the background when the editor opens, so pressing Test does not wait on it.
+- Hidden testcase data never leaves the server. Test uses only sample data, which the statement already makes public.
+- Multi-file problems keep only `editable` and `readonly` files.
+
+## Non-goals
+
+- Official judging is unchanged: Submit, verdicts and the sandbox, and official judging still compiles the judge program from source in its per-stage gVisor Pod.
+- No authoring-time check or build status for judge programs; authors press Test on their own problem.
+- `special_env` (Advanced) still has no Test.
+- No per-problem toggle for what students can read.
+- No multi-file precompile. The student's editable files change on every Test, and forge's in-browser object cache already reuses unchanged files.
+- A Wasm fast path for official judging is a separate future item (Quality Ledger).
+
+## What becomes public, and the accepted risk
+
+| Data | Test sends it to the browser |
+| --- | --- |
+| Checker / interactor source | yes |
+| Sample input, output, interactor input | yes (already in the statement) |
+| `readonly` workspace files | yes (already today) |
+| Hidden testcases (input or answer) | **never** (SEC-12) |
+
+The accepted risk: bugs in a judge program, such as a missing validity check or an off-by-one query limit, become easier to find. Those bugs exist whether or not the source is public, and authors own them. The judge tab states once that students can read the program.
+
+The docs record safe authoring practice:
+- A checker only verifies, and reads the optimum from `judge_answer` (the testcase's expected-output field), as the seed checkers do.
+- An interactor reads its secret from `judge_input`.
+
+## Design
+
+### Judge program delivery
+
+`GET /api/problems/[id]/judge-program?context=…` returns `{ role, language, source, sha256 }` for checker and interactive problems, and 404 otherwise.
+
+- The source is read through the existing verified script pointer.
+- **Authorization:** anyone who may view the problem in that context. Checker Test therefore keeps working after an exam or contest ends, like standard Test.
+- The route is classified `exam-scoped` in the exam-confinement allowlist and uses the standard API rate limiter.
+
+### Preparing the judge program in the browser
+
+When the editor opens on a checker or interactive problem, it:
+1. fetches the source;
+2. builds it with `@wasm-oj/browser`, using core's `judgeProgramCompileInput` (the DOMjudge Python wrapper, or the C++ `bits/stdc++.h` shim with the PCH-only-when-included rule):
+ - **Python** is packaged into a runtime bundle at once, with no compile;
+ - **C++** compiles in the background;
+3. preloads whatever toolchains the judge program needs, next to the student's own: clang for a C++ judge program, the Python runtime for a Python one.
+
+The built program stays in memory for the page session, keyed by `sha256`. It is rebuilt when the editor reopens with different source.
+
+Test waits for it, as it already waits for the student's toolchain, and the button shows that it is preparing. If the judge program fails to build, Test is disabled with a reason and the panel shows the compiler diagnostics. The source is public, so its diagnostics are too.
+
+### Checker problems
+
+1. Compile the student's program and run each selected sample with `stdin = sample.input`, as today.
+2. For each sample whose run exited normally, run the checker with:
+ - args `/judge/input /judge/answer /judge/feedback`;
+ - files: input = `sample.input`, answer = `sample.output`;
+ - stdin = the student's stdout;
+ - output path `/judge/feedback/teammessage.txt`;
+ - official judging's `validatorTimeoutMs` and 512 MiB.
+3. Map the result with `checkerCaseVerdict`: exit 42 is AC, 43 is WA, anything else is SE. Show `teammessage`.
+4. Custom cases stay execution-only, because they have no `judge_answer` in the author's format.
+
+### Interactive problems
+
+- `engine.interact(contestant, interactor, …)` from `@wasm-oj/browser` runs both programs. The interactor gets `sample.interactorInput` as `/judge/input`.
+- Map the result with `interactiveCaseVerdict`, and show the transcript, contestant stderr and time. A contestant stopped by its wall limit is TLE.
+- Custom cases take an interactor input, which the student types as the secret, and the real interactor judges them.
+- JS/TS contestants stay unavailable until upstream streams QuickJS stdin.
+
+### Upstream dependency
+
+This PR merges only after a `@wasm-oj` release that contains:
+
+| Change | Status | Why |
+| --- | --- | --- |
+| Browser `interact` passes `startupEntropyBytes` | new PR | `runner.worker.ts` `interactiveCoreProgram` omits the field, but runtime-core requires it (`startup_entropy_bytes: u64`, no serde default), so every browser `interact` fails while decoding its request |
+| Python (runtime-bundle) interactors | wasm-oj/forge#93 | all four interactive problems use Python interactors |
+| In-module interactive metering | wasm-oj/forge#95 | a CPU-bound contestant must stop at its instruction budget instead of keeping the student's tab busy until the wall limit |
+
+Until that release, NOJV develops against locally built packages. It then pins `@wasm-oj/browser` and the toolchains to the release. Forge must not be patched inside NOJV (JDG-15).
+
+### Removing `hidden` workspace visibility
+
+- On 2026-10-07 production had 9 `editable` files, 9 `readonly` files and **0 `hidden`** files (read-only query).
+- A migration sets any `hidden` rows to `readonly`, then recreates the `WorkspaceFileVisibility` enum without `hidden`.
+- Stored judge snapshots map a legacy `hidden` to `readonly`, so old submissions still rejudge.
+- The UI, application, seeds and tests drop it.
+- PRB-01 records the rejection: `hidden` gave no confidentiality, nobody used it, and it misled authors.
+
+### Removed from #641
+
+- The whole server side:
+ - the `WORKER_MODE=test` worker and its Deployment;
+ - the `test-judge` queue and its partitions;
+ - the Test and build workflows, and the build-after-save dispatch;
+ - the `test-judge-programs/` cache and the Test API;
+ - the Redis lock and the `rl:test-judge` limiter;
+ - `TEST_JUDGE_ENABLED`, `WASM_OJ_*` and `TEST_JUDGE_SLOTS`;
+ - `@wasm-oj/server` and its EPIPE patch;
+ - the worker image's WASM-OJ layers, `infra/docker/wasm-oj-toolchains/` and the related Renovate pins.
+- In core: the request, response and record schemas, the cache key and server identity, and `build-artifact-wire`.
+- On the edit page: the judge-program build status and the "check samples with the checker" action.
+
+### Kept from #641
+
+- `interactorInput` and `interactionFormat`.
+- The seed fixes.
+- "Executed" for custom cases without expected output.
+- The contest code-draft authorization fix (WEB-05).
+- The PCH-only-with-`` rule.
+- In core: the verdict helpers, Python wrappers, C++ shim and `judgeProgramCompileInput`, now used by the browser.
+
+## Decision log changes
+
+- **JDG-15:** rewritten as "Test runs entirely in the browser, judge programs included".
+ - Rejected:
+ - running judge programs on the server (#641, withdrawn before release);
+ - compiling them on the server (that keeps a server WASM-OJ runtime and worker for a preview feature);
+ - execution-only Test;
+ - exposure toggles.
+ - Rules:
+ - what is public, and that authors own their judge programs;
+ - the server neither compiles nor executes anything for Test;
+ - Test never receives non-sample testcase data.
+- **JDG-26, SEC-15 and OPS-21:** withdrawn, each with its reason and a pointer to JDG-15. IDs are never reused.
+- **JDG-05:** the "validators never seen" rule is scoped to official judging.
+- **JDG-12, PRB-03, PRB-22 and WEB-05:** the server-Test wording goes.
+- **PRB-01 and PRB-09:** `hidden` goes; PRB-01 gets a Rejected line.
+- **OPS-18:** the wording #641 added goes.
+
+## Testing
+
+- **Unit:**
+ - the browser checker and interactor runs with a fake engine (args, files, verdict mapping);
+ - judge-program preparation (Python packaging, C++ compile, failure);
+ - Test capability per problem type;
+ - endpoint authorization (exam, contest window, assignment membership, practice);
+ - the visibility migration;
+ - legacy-`hidden` snapshot parsing.
+- **Component:**
+ - Test button states (preparing, build failed, JS/TS on interactive problems);
+ - the checker result panel with `teammessage`;
+ - the transcript panel;
+ - interactive custom cases;
+ - the judge-tab note.
+- **Browser check on seeds:**
+ - `any-two-sum` and `shortest-route-plan`: AC and WA;
+ - a C++ checker fixture, and a broken one that disables Test;
+ - `guess-the-number` and the other three interactive problems: AC, WA, and TLE from an infinite loop;
+ - a multi-file problem showing no `hidden` option;
+ - standard problems unchanged.
+- `pnpm ci:verify`, `pnpm lint:helm`, and a worker image build without the WASM-OJ layers.
From 0a684105f27f369213506ad649fbc475b883a07f Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 16:38:18 +0800
Subject: [PATCH 02/22] refactor: remove server-side Test judging
Test will run checkers and interactors in the student's browser, so the
server half of #641 goes: the test-judge API route, limiter, Redis lock
and storage keys, the application test-judge domain and its build
dispatch after a judge-config save, the test-judge queue, workflows and
WORKER_MODE=test, the WASM-OJ runtime layers and toolchains in the
worker image, the worker-test chart Deployment, web's TEST_JUDGE_ENABLED,
and @wasm-oj/server with its patch. Core drops the request, response and
record schemas, the judge-program cache key and server identity, and the
artifact wire format. The edit page no longer shows a judge-program build
status or offers "check samples with the checker". Renovate updates
@wasm-oj/* and the rust image again.
Core keeps what the browser will use: the checker and interactive verdict
mapping, truncateUtf8, the WASM-OJ termination verdict, the DOMjudge
Python wrappers, judgeProgramCompileInput and
interactiveContestantSupported. The Test capability is no longer a
server field; the editor disables Test on special_env problems itself.
Until browser judging lands, checker and interactive Test report that
Test isn't available for the problem. The contest participation and
window checks on code drafts keep their coverage on listCodeDrafts.
Co-Authored-By: Claude Opus 5.5
---
.env.example | 10 +-
.github/renovate.json | 12 -
apps/web/messages/en.json | 16 -
apps/web/messages/zh-TW.json | 16 -
.../features/problem/editors/Editor.svelte | 8 +-
.../problem/editors/use-editor-run.svelte.ts | 236 +----
.../features/problem/tabs/JudgeTab.svelte | 22 +-
.../tabs/judge/JudgeProgramTestStatus.svelte | 169 ----
apps/web/src/lib/server/env.ts | 2 -
apps/web/src/lib/server/shared/api-handler.ts | 5 -
.../web/src/lib/server/shared/rate-limiter.ts | 1 -
.../src/lib/services/submission-service.ts | 49 --
apps/web/src/lib/types/index.ts | 5 +-
.../problems/[problemId]/edit/+page.server.ts | 30 +-
.../problems/[problemId]/edit/+page.svelte | 1 -
.../api/problems/[id]/test-judge/+server.ts | 46 -
apps/worker/README.md | 33 +-
apps/worker/package.json | 2 -
.../src/activities/test-judge-bundle.ts | 1 -
apps/worker/src/activities/test-judge.ts | 353 --------
apps/worker/src/env.ts | 6 +-
apps/worker/src/otel.ts | 4 +-
apps/worker/src/test-judge/judge-program.ts | 143 ---
apps/worker/src/test-judge/runtime.ts | 180 ----
apps/worker/src/worker-app.ts | 54 --
apps/worker/src/workflows/activity-options.ts | 1 -
apps/worker/src/workflows/index.ts | 1 -
apps/worker/src/workflows/test-judge.ts | 43 -
infra/charts/nojv/README.md | 6 +-
infra/charts/nojv/templates/pdb.yaml | 3 -
.../charts/nojv/templates/web.deployment.yaml | 2 -
infra/charts/nojv/templates/worker-rbac.yaml | 11 -
.../templates/worker-test.deployment.yaml | 164 ----
infra/charts/nojv/values-single-machine.yaml | 2 -
infra/charts/nojv/values.yaml | 12 -
infra/docker/temporal-dynamic-config.yaml | 4 -
.../wasm-oj-toolchains/package-lock.json | 47 -
infra/docker/wasm-oj-toolchains/package.json | 8 -
infra/docker/worker.Dockerfile | 31 -
infra/flux/temporal-values.yaml | 4 -
infra/gcp/gke/temporal/helm-values.ha.yaml | 4 -
packages/application/src/code-draft.ts | 2 +-
packages/application/src/index.ts | 1 -
packages/application/src/problem/details.ts | 16 +-
.../application/src/problem/judge-config.ts | 7 -
.../src/problem/mutations/judge-config.ts | 39 +-
.../application/src/shared/orchestration.ts | 8 -
.../src/shared/test-judge-enabled.ts | 3 -
packages/application/src/test-judge/index.ts | 344 --------
packages/core/src/index.ts | 3 -
.../core/src/judge/build-artifact-wire.ts | 68 --
packages/core/src/judge/test-capability.ts | 29 +-
packages/core/src/judge/test-judge-program.ts | 20 -
.../core/src/judge/test-judge-response.ts | 81 --
packages/core/src/judge/test-judge-verdict.ts | 3 +-
packages/core/src/schemas/test-judge.ts | 167 ----
packages/core/src/workflow-types.ts | 21 -
packages/redis/src/keys.ts | 2 -
packages/storage/README.md | 4 +-
packages/storage/src/index.ts | 3 -
packages/storage/src/keys.ts | 10 -
packages/temporal/README.md | 4 +-
packages/temporal/src/dispatch.ts | 64 +-
.../temporal/src/orchestration-adapter.ts | 4 -
packages/temporal/src/task-queues.ts | 1 -
patches/@wasm-oj__server@0.2.3.patch | 31 -
pnpm-lock.yaml | 18 -
pnpm-workspace.yaml | 2 -
scripts/classify-image-impact.mjs | 1 -
.../web/editor-output-comparison.test.ts | 4 +-
tests/component/web/editor-shortcuts.test.ts | 1 -
.../web/editor-test-button-state.test.ts | 50 +-
.../web/editor-toolchain-preload.test.ts | 1 -
.../fixtures/judge-program-status-host.svelte | 19 -
.../web/judge-program-test-status.test.ts | 267 ------
.../application/code-drafts.test.ts | 173 ++++
.../application/submission-sweep.test.ts | 2 -
.../application/test-judge-author.test.ts | 273 ------
.../application/test-judge-domain.test.ts | 818 ------------------
.../db/lifecycle-cancellation-outbox.test.ts | 2 -
tests/integration/http/test-judge.test.ts | 274 ------
.../judge/test-judge-runtime.test.ts | 399 ---------
.../temporal/test-judge-priority.test.ts | 81 --
tests/setup/integration-setup.ts | 5 +-
tests/tsconfig.json | 6 -
.../application/assignment-mutations.test.ts | 2 -
.../application/contest-time-window.test.ts | 2 -
.../course-assignment-lifecycle.test.ts | 2 -
.../application/exam-create-lifecycle.test.ts | 2 -
.../application/exam-publish-delete.test.ts | 2 -
.../application/lifecycle-reconciler.test.ts | 2 -
.../application/registry-gc-trigger.test.ts | 2 -
tests/unit/core/test-capability.test.ts | 86 +-
tests/unit/core/test-judge-program.test.ts | 120 ---
tests/unit/core/test-judge-response.test.ts | 76 --
tests/unit/core/test-judge-schema.test.ts | 198 -----
tests/unit/core/test-judge-verdict.test.ts | 2 -
tests/unit/infra/env-manifest-parity.test.ts | 51 --
tests/unit/infra/image-impact.test.ts | 1 -
.../infra/temporal-queue-partitions.test.ts | 8 +-
tests/unit/infra/wasm-oj-pins.test.ts | 73 --
tests/unit/infra/workload-disruption.test.ts | 30 +-
tests/unit/storage/keys.test.ts | 12 -
.../unit/temporal/test-judge-dispatch.test.ts | 148 ----
tests/unit/web/editor-client-test.test.ts | 338 +-------
tests/unit/web/problem-edit-actions.test.ts | 111 +--
tests/unit/web/rate-limit-wrappers.test.ts | 31 +-
.../activity-bundle-registration.test.ts | 10 +-
tests/unit/worker/env.test.ts | 53 --
tests/unit/worker/mailer-startup.test.ts | 11 +-
tests/unit/worker/test-judge-activity.test.ts | 684 ---------------
.../worker/test-judge-engine-pool.test.ts | 203 -----
tests/unit/worker/test-judge-program.test.ts | 379 --------
tests/unit/worker/worker-app.test.ts | 92 +-
.../unit/worker/workflow-registration.test.ts | 4 +-
115 files changed, 308 insertions(+), 7505 deletions(-)
delete mode 100644 apps/web/src/lib/components/features/problem/tabs/judge/JudgeProgramTestStatus.svelte
delete mode 100644 apps/web/src/routes/api/problems/[id]/test-judge/+server.ts
delete mode 100644 apps/worker/src/activities/test-judge-bundle.ts
delete mode 100644 apps/worker/src/activities/test-judge.ts
delete mode 100644 apps/worker/src/test-judge/judge-program.ts
delete mode 100644 apps/worker/src/test-judge/runtime.ts
delete mode 100644 apps/worker/src/workflows/test-judge.ts
delete mode 100644 infra/charts/nojv/templates/worker-test.deployment.yaml
delete mode 100644 infra/docker/wasm-oj-toolchains/package-lock.json
delete mode 100644 infra/docker/wasm-oj-toolchains/package.json
delete mode 100644 packages/application/src/shared/test-judge-enabled.ts
delete mode 100644 packages/application/src/test-judge/index.ts
delete mode 100644 packages/core/src/judge/build-artifact-wire.ts
delete mode 100644 packages/core/src/judge/test-judge-response.ts
delete mode 100644 packages/core/src/schemas/test-judge.ts
delete mode 100644 patches/@wasm-oj__server@0.2.3.patch
delete mode 100644 tests/component/web/fixtures/judge-program-status-host.svelte
delete mode 100644 tests/component/web/judge-program-test-status.test.ts
delete mode 100644 tests/integration/application/test-judge-author.test.ts
delete mode 100644 tests/integration/application/test-judge-domain.test.ts
delete mode 100644 tests/integration/http/test-judge.test.ts
delete mode 100644 tests/integration/judge/test-judge-runtime.test.ts
delete mode 100644 tests/integration/temporal/test-judge-priority.test.ts
delete mode 100644 tests/unit/core/test-judge-response.test.ts
delete mode 100644 tests/unit/core/test-judge-schema.test.ts
delete mode 100644 tests/unit/infra/wasm-oj-pins.test.ts
delete mode 100644 tests/unit/temporal/test-judge-dispatch.test.ts
delete mode 100644 tests/unit/worker/env.test.ts
delete mode 100644 tests/unit/worker/test-judge-activity.test.ts
delete mode 100644 tests/unit/worker/test-judge-engine-pool.test.ts
delete mode 100644 tests/unit/worker/test-judge-program.test.ts
diff --git a/.env.example b/.env.example
index ac2cc1a40..7aae600f8 100644
--- a/.env.example
+++ b/.env.example
@@ -73,16 +73,8 @@ SANDBOX_PIDS_LIMIT=64
# Worker
PORT=8080
WORKER_CONCURRENCY=4
-# WORKER_MODE: "all" (default), "judge" (sandbox only), "platform" (lifecycle/plagiarism only),
-# "test" (checker and interactive Test on the test-judge queue; needs both WASM-OJ dirs below)
+# WORKER_MODE: "all" (default), "judge" (sandbox only), "platform" (lifecycle/plagiarism only)
WORKER_MODE=all
-# Checker and interactive Test through the WASM-OJ test judge. "all" serves the
-# test-judge queue only when both WASM-OJ dirs are set (docs/runbooks/getting-started.md).
-TEST_JUDGE_ENABLED=false
-# WASM_OJ_RUNTIME_DIR=/home/you/src/wasm-oj-forge/crates/runtime-core/target/release
-# WASM_OJ_TOOLCHAIN_DIR=/home/you/.cache/nojv-wasm-oj-toolchains
-# WASM_OJ_CACHE_DIR=/tmp/wasm-oj
-# TEST_JUDGE_SLOTS=2
# Minutes a submission may stay pending/running before the platform sweeper marks
# it system_error and terminates its judge. Range 10–1440; defaults to 10.
SUBMISSION_PENDING_TIMEOUT_MINUTES=10
diff --git a/.github/renovate.json b/.github/renovate.json
index aba640832..16381cd87 100644
--- a/.github/renovate.json
+++ b/.github/renovate.json
@@ -95,12 +95,6 @@
"matchDepTypes": ["packageManager", "engines"],
"enabled": false
},
- {
- "description": "WASM-OJ packages are pinned to the forge release in the worker image and checked by the pin test; they move by hand with a forge upgrade",
- "matchManagers": ["npm"],
- "matchPackageNames": ["/^@wasm-oj\\//"],
- "enabled": false
- },
{
"matchManagers": ["github-actions"],
"schedule": ["* 9-11 1-7 * 1"],
@@ -122,12 +116,6 @@
"matchPackageNames": ["node"],
"enabled": false
},
- {
- "description": "The rust image only builds the pinned WASM-OJ runtime; each bump replaces an 82 MB worker layer, so it moves by hand with a forge upgrade",
- "matchDatasources": ["docker"],
- "matchPackageNames": ["rust"],
- "enabled": false
- },
{
"description": "alpine/k8s tracks the cluster's kubectl minor",
"matchDatasources": ["docker"],
diff --git a/apps/web/messages/en.json b/apps/web/messages/en.json
index a0b52a9e4..282e9f7e8 100644
--- a/apps/web/messages/en.json
+++ b/apps/web/messages/en.json
@@ -272,11 +272,6 @@
"admin_cancel": "Cancel",
"admin_checkerHelpBody": "Your validator follows the DOMjudge output-validator interface and runs in its own isolated container — the student's program never sees the answer. It is invoked as `validator ` with the student output on stdin. It renders an accept/reject verdict only (no partial scoring). Python: `judge_input`, `judge_answer`, `team_output` are pre-bound as strings, and helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the right feedback files and exit. C++: read `argv[1]` / `argv[2]` and stdin yourself, write `teammessage.txt` into the `argv[3]` feedback dir, and exit 42 (accept) or 43 (wrong) — no judge header required. Feedback is shown to the student. Validator timeout is 30 seconds; a crash or hang becomes a system error and does not count against the student.",
"admin_checkerHelpTitle": "How to write a checker",
- "admin_checkSamples": "Check samples with the checker",
- "admin_checkSamplesRejected": "The checker must accept each sample's output, because students' Test uses it as the answer. Fix the checker or the sample.",
- "admin_checkSamplesSaveFirst": "Save your changes first to check samples against them.",
- "admin_checkSamplesUnavailable": "The test judge is unavailable right now. Try again later.",
- "admin_checkingSamples": "Checking samples…",
"admin_compareCaseSensitive": "Case-sensitive comparison",
"admin_compareFloatTolerance": "Float tolerance (ε)",
"admin_compareFloatToleranceHint": "Leave empty for exact matching. When set, numeric tokens match within absolute OR relative error ε. For anything more, write a Checker.",
@@ -322,11 +317,6 @@
"admin_interactorLanguage": "Interactor language",
"admin_judgeChecker": "Checker script",
"admin_judgeInteractive": "Interactive",
- "admin_judgeProgramDiagnostics": "Build output",
- "admin_judgeProgramTestFailed": "This judge program can't run in Test. Submissions are still judged normally.",
- "admin_judgeProgramTestPending": "Preparing this judge program for Test…",
- "admin_judgeProgramTestReady": "Test can run this judge program.",
- "admin_judgeProgramTestUnavailable": "Couldn't check this judge program right now.",
"admin_judgeStandard": "Standard (stdin/stdout diff)",
"admin_judgeType": "Judge type",
"admin_judgeTypeHint": "How each testcase is evaluated.",
@@ -403,8 +393,6 @@
"admin_tabBasicInfo": "Basic Info",
"admin_tabReports": "Reports",
"admin_tabJudge": "Judge Settings",
- "admin_testJudgeDisabled": "Test judging isn't enabled on this server.",
- "admin_testPythonInteractorUnsupported": "Test doesn't support Python interactors yet.",
"admin_tabLocked": "Please complete and save basic info first",
"admin_tabLockedFields": "Fill in and save these in Basic Info first: {fields}",
"admin_tabOverview": "Overview",
@@ -1150,12 +1138,8 @@
"editor_clientTestLanguage": "The browser runtime for this language is unavailable. Test runs entirely on your device.",
"editor_testInteractiveLanguage": "Test can't run interactive problems in JavaScript or TypeScript yet. Switch to another language, or use Submit to have your code judged.",
"editor_testNoInteractiveSamples": "This problem has no sample with an interactor input, so there is nothing to test. Use Submit to have your code judged.",
- "editor_testJudgeBusy": "Test is busy right now. Try again in a moment.",
- "editor_testJudgeProgramBuildFailed": "This problem's checker or interactor can't run in Test. Submit still judges your code normally.",
"editor_testUnavailableForProblem": "Test isn't available for this problem right now. Submit still judges your code normally.",
"editor_testUnsupportedProblemType": "This problem type doesn't support Test. Use Submit to have your code judged.",
- "editor_testTooLarge": "This Test request is too large to send to the server. Reduce your program's size or output, or use Submit to have your code judged.",
- "editor_judgingOnServer": "Judging on the server…",
"editor_judgeSystemError": "The judge failed on this case. This is a problem on the judge's side, not with your program.",
"editor_interactiveTestNote": "Test runs your program against this problem's interactor on each sample's interactor input; the interactor judges your replies live. You can't add your own cases to an interactive problem.",
"editor_checkerCasesNote": "This problem's checker judges the samples only. Cases you add or change just run without a verdict, so check their output yourself.",
diff --git a/apps/web/messages/zh-TW.json b/apps/web/messages/zh-TW.json
index 8b17b5745..a8247b922 100644
--- a/apps/web/messages/zh-TW.json
+++ b/apps/web/messages/zh-TW.json
@@ -272,11 +272,6 @@
"admin_cancel": "取消",
"admin_checkerHelpBody": "你的 validator 採用 DOMjudge 輸出驗證器介面,在獨立隔離容器中執行——學生程式永遠看不到答案。呼叫方式為 `validator `,學生輸出由 stdin 傳入。只輸出接受/錯誤判決(不做部分計分)。Python:`judge_input`、`judge_answer`、`team_output` 已預先注入為字串,輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入對應回饋檔並結束。C++:自行讀取 `argv[1]` / `argv[2]` 與 stdin,把 `teammessage.txt` 寫進 `argv[3]` 的 feedback 目錄,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。回饋訊息會顯示給學生。Validator 超時上限 30 秒;掛掉或逾時記為系統錯誤,不影響學生判決。",
"admin_checkerHelpTitle": "如何撰寫 Checker",
- "admin_checkSamples": "用 checker 檢查範例",
- "admin_checkSamplesRejected": "checker 必須接受每個範例的輸出,因為學生的測試會拿它當作答案。請修正 checker 或範例。",
- "admin_checkSamplesSaveFirst": "請先儲存變更,才能用新的設定檢查範例。",
- "admin_checkSamplesUnavailable": "測試評測目前無法使用,請稍後再試。",
- "admin_checkingSamples": "正在檢查範例…",
"admin_compareCaseSensitive": "區分大小寫比對",
"admin_compareFloatTolerance": "浮點容差(ε)",
"admin_compareFloatToleranceHint": "留空為精確比對。設定後,數值 token 在絕對或相對誤差 ε 內即視為相符。需要更複雜的判定請改寫 Checker。",
@@ -322,11 +317,6 @@
"admin_interactorLanguage": "Interactor 語言",
"admin_judgeChecker": "Checker 腳本",
"admin_judgeInteractive": "互動題",
- "admin_judgeProgramDiagnostics": "編譯輸出",
- "admin_judgeProgramTestFailed": "這個評測程式無法在測試中執行,提交仍會照常評測。",
- "admin_judgeProgramTestPending": "正在準備讓測試使用這個評測程式…",
- "admin_judgeProgramTestReady": "測試可以使用這個評測程式。",
- "admin_judgeProgramTestUnavailable": "目前無法檢查這個評測程式。",
"admin_judgeStandard": "標準(stdin/stdout 比對)",
"admin_judgeType": "評測類型",
"admin_judgeTypeHint": "每筆測資的評測方式。",
@@ -403,8 +393,6 @@
"admin_tabBasicInfo": "題目資訊",
"admin_tabReports": "內容檢舉",
"admin_tabJudge": "判題設定",
- "admin_testJudgeDisabled": "此伺服器未啟用測試評測。",
- "admin_testPythonInteractorUnsupported": "測試尚不支援 Python interactor。",
"admin_tabLocked": "請先完成並儲存基本資訊",
"admin_tabLockedFields": "請先在「基本資訊」填寫並儲存:{fields}",
"admin_tabOverview": "總覽",
@@ -1150,12 +1138,8 @@
"editor_clientTestLanguage": "目前無法使用此語言的瀏覽器執行環境。Test 全程在你的裝置執行。",
"editor_testInteractiveLanguage": "互動題的測試目前還不支援 JavaScript 與 TypeScript。請改用其他語言,或直接提交評測。",
"editor_testNoInteractiveSamples": "此題沒有附 Interactor 輸入的範例,無法測試。請直接提交評測。",
- "editor_testJudgeBusy": "測試目前忙碌中,請稍後再試。",
- "editor_testJudgeProgramBuildFailed": "此題的 checker 或 interactor 無法在測試中執行,提交仍會照常評測。",
"editor_testUnavailableForProblem": "此題目前無法使用測試,提交仍會照常評測。",
"editor_testUnsupportedProblemType": "此題型不支援測試,請直接提交評測。",
- "editor_testTooLarge": "這次測試的資料太大,無法送到伺服器。請縮小程式或輸出,或直接提交評測。",
- "editor_judgingOnServer": "伺服器評測中…",
"editor_judgeSystemError": "評測程式在這組測資出錯,這是評測端的問題,不是你的程式造成的。",
"editor_interactiveTestNote": "測試會用每組範例的 Interactor 輸入,讓你的程式和本題的 interactor 對話,由 interactor 即時判斷你的回應。互動題無法自行新增測資。",
"editor_checkerCasesNote": "本題的 checker 只評測範例測資。你新增或修改的測資只會執行、不會判定對錯,請自行檢查輸出。",
diff --git a/apps/web/src/lib/components/features/problem/editors/Editor.svelte b/apps/web/src/lib/components/features/problem/editors/Editor.svelte
index efc3b5991..082ad4e2e 100644
--- a/apps/web/src/lib/components/features/problem/editors/Editor.svelte
+++ b/apps/web/src/lib/components/features/problem/editors/Editor.svelte
@@ -223,17 +223,13 @@
$effect(() => () => runController.markDestroyed());
let testDisabledReason = $derived.by(() => {
- const capability = problem.testCapability;
- if (!capability.available) {
- return capability.reason === "special_env"
- ? m.editor_testUnsupportedProblemType()
- : m.editor_testUnavailableForProblem();
- }
+ if (isSpecialEnv) return m.editor_testUnsupportedProblemType();
if (problem.judgeType === "interactive") {
if (!interactiveContestantSupported(language)) return m.editor_testInteractiveLanguage();
if (!problem.samples.some((sample) => sample.interactorInput?.trim()))
return m.editor_testNoInteractiveSamples();
}
+ if (problem.judgeType !== "standard") return m.editor_testUnavailableForProblem();
return runController.testDisabledReason;
});
diff --git a/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts b/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
index eb4c43b32..dee9431c4 100644
--- a/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
+++ b/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
@@ -1,36 +1,23 @@
-import {
- MAX_CASE_STDOUT_BYTES,
- interactiveContestantSupported,
- serialiseBuildArtifact,
- type JudgeConfig,
- type JudgeType,
- type Language,
- type SubmissionContext,
- type SubmissionResult,
- type SubmissionRunCase,
- type TestJudgeCaseResult,
+import type {
+ JudgeConfig,
+ JudgeType,
+ Language,
+ SubmissionContext,
+ SubmissionResult,
+ SubmissionRunCase,
} from "@nojv/core";
import { m } from "$lib/paraglide/messages.js";
import {
- requestTestJudge,
submissionRequestValidationError,
SubmissionRequestError,
- type SubmissionRequest,
} from "$lib/services/submission-service";
import { submitProblem } from "$lib/services/problem-submission";
import { toasts } from "$lib/stores/toast";
import {
- browserCaseResult,
- browserLocalErrorResult,
- browserLocalSubmissionResult,
browserToolchainPercent,
- compileBrowserLocally,
preloadBrowserToolchain,
- runBrowserCases,
runBrowserLocally,
supportsBrowserLocalRun,
- type BrowserCaseRun,
- type BrowserCompileOutcome,
} from "$lib/services/browser-local-run";
import type { ProblemDetail, TestCaseView, TestRunResult } from "$lib/types";
import {
@@ -90,21 +77,8 @@ function messageForSubmitError(code: string | null): string {
return m.editor_clientTestCustomImage();
case "client_test_language":
return m.editor_clientTestLanguage();
- case "client_test_interactive_language":
- return m.editor_testInteractiveLanguage();
- case "client_test_no_interactive_samples":
- return m.editor_testNoInteractiveSamples();
- case "test_judge_busy":
- return m.editor_testJudgeBusy();
- case "test_request_too_large":
- return m.editor_testTooLarge();
- case "judge_program_build_failed":
- return m.editor_testJudgeProgramBuildFailed();
- case "test_judge_unavailable":
- case "judge_program_unsupported":
+ case "client_test_judge_program":
return m.editor_testUnavailableForProblem();
- case "test_rejected":
- return m.editor_runFailed();
case "browser_toolchain_unavailable":
return m.editor_toolchainUnavailable();
case "invalid_source":
@@ -128,16 +102,7 @@ function messageForSubmitError(code: string | null): string {
}
}
-const TEST_DISABLING_CODES = new Set([
- "judge_program_build_failed",
- "judge_program_unsupported",
-]);
-
-const UNJUDGED_SAMPLE_CODES = new Set(["test_judge_busy", "test_judge_unavailable"]);
-
-function interactiveSampleIndices(samples: ProblemDetail["samples"]): number[] {
- return samples.flatMap((sample, index) => (sample.interactorInput?.trim() ? [index] : []));
-}
+const TEST_DISABLING_CODES = new Set(["client_test_judge_program"]);
function initialRunCases(
samples: ProblemDetail["samples"],
@@ -168,124 +133,6 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
let destroyed = false;
let abortController: AbortController | null = null;
- async function runCheckerTest(
- request: SubmissionRequest,
- runCases: SubmissionRunCase[],
- signal: AbortSignal,
- ): Promise {
- let runs: BrowserCaseRun[];
- try {
- const build = await compileBrowserLocally(request, args.problemId, signal);
- if (!build.ok) return build.result;
- runs = await runBrowserCases(
- build.artifact,
- runCases,
- {
- language: request.language,
- timeLimitMs: args.timeLimitMs,
- memoryLimitMb: args.memoryLimitMb,
- env: args.judgeConfig().runtime?.env ?? {},
- },
- signal,
- );
- } catch (error) {
- return signal.aborted ? null : browserLocalErrorResult(error);
- }
- const judgedCases = new Map();
- for (const [index, runCase] of runCases.entries()) {
- const sampleIndex = args.initialSamples.findIndex(
- (sample) => sample.input === runCase.input,
- );
- if (sampleIndex >= 0 && runs[index]?.verdict === "AC" && !judgedCases.has(sampleIndex)) {
- judgedCases.set(sampleIndex, index);
- }
- }
- const judgements = new Map();
- let serverNotice: string | undefined;
- if (judgedCases.size > 0) {
- runStatus = m.editor_judgingOnServer();
- try {
- const response = await requestTestJudge(
- args.problemId,
- {
- kind: "checker",
- context: args.context(),
- cases: [...judgedCases].map(([sampleIndex, index]) => ({
- sampleIndex,
- output: (runs[index]?.stdout ?? "").slice(0, MAX_CASE_STDOUT_BYTES),
- })),
- },
- signal,
- );
- if (!response) return null;
- for (const [position, index] of [...judgedCases.values()].entries()) {
- const judgement = response.cases[position];
- if (judgement) judgements.set(index, judgement);
- }
- } catch (error) {
- if (
- !(error instanceof SubmissionRequestError) ||
- !UNJUDGED_SAMPLE_CODES.has(error.code ?? "")
- )
- throw error;
- serverNotice = messageForSubmitError(error.code);
- }
- }
- const caseResults = runs.map((run, index): TestCaseView => {
- const view = browserCaseResult(run, undefined, undefined, index);
- const judgement = judgements.get(index);
- if (!judgement) return view.verdict === "AC" ? { ...view, executionOnly: true } : view;
- return {
- ...view,
- verdict: judgement.verdict,
- serverJudged: true,
- ...(judgement.teamMessage ? { teamMessage: judgement.teamMessage } : {}),
- };
- });
- return {
- ...browserLocalSubmissionResult(caseResults),
- caseResults,
- ...(serverNotice ? { serverNotice } : {}),
- };
- }
-
- async function runInteractiveTest(
- request: SubmissionRequest,
- sampleIndices: number[],
- signal: AbortSignal,
- ): Promise {
- let build: BrowserCompileOutcome;
- try {
- build = await compileBrowserLocally(request, args.problemId, signal);
- } catch (error) {
- return signal.aborted ? null : browserLocalErrorResult(error);
- }
- if (!build.ok) return build.result;
- runStatus = m.editor_judgingOnServer();
- const response = await requestTestJudge(
- args.problemId,
- {
- kind: "interactive",
- context: args.context(),
- language: request.language,
- artifact: serialiseBuildArtifact(build.artifact),
- cases: sampleIndices.map((sampleIndex) => ({ sampleIndex })),
- },
- signal,
- );
- if (!response) return null;
- const caseResults = response.cases.map((judgement, index): TestCaseView => ({
- index,
- verdict: judgement.verdict,
- serverJudged: true,
- timeMs: judgement.timeMs ?? 0,
- ...(judgement.contestantStderr ? { stderr: judgement.contestantStderr } : {}),
- ...(judgement.teamMessage ? { teamMessage: judgement.teamMessage } : {}),
- ...(judgement.transcript ? { transcript: judgement.transcript } : {}),
- }));
- return { ...browserLocalSubmissionResult(caseResults), caseResults };
- }
-
async function runSubmission(): Promise {
if (args.isSpecialEnv())
throw new SubmissionRequestError(
@@ -293,28 +140,19 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
"client_test_custom_image",
null,
);
- const judgeType = args.judgeType();
- const language = args.language();
- const interactive = judgeType === "interactive";
- if (interactive && !interactiveContestantSupported(language))
- throw new SubmissionRequestError(
- "Interactive Test does not support this language.",
- "client_test_interactive_language",
- null,
- );
- const sampleIndices = interactive ? interactiveSampleIndices(args.initialSamples) : [];
- if (interactive && sampleIndices.length === 0)
+ if (args.judgeType() !== "standard")
throw new SubmissionRequestError(
- "No sample has an interactor input.",
- "client_test_no_interactive_samples",
+ "Test can't run this problem's judge program in the browser yet.",
+ "client_test_judge_program",
null,
);
+ const language = args.language();
abortController = new AbortController();
const { signal } = abortController;
- const runCases = interactive ? [] : projectRunCasesForRequest(panelRunCases);
- if (!interactive && runCases.length === 0)
+ const runCases = projectRunCasesForRequest(panelRunCases);
+ if (runCases.length === 0)
throw new SubmissionRequestError("No testcases provided.", "invalid_run_cases", null);
const request = buildSubmissionRequest({
@@ -326,7 +164,7 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
sampleOnly: true,
workspaceDrafts: args.workspaceDrafts(),
workspaceFiles: args.workspaceFiles(),
- ...(interactive ? {} : { runCases }),
+ runCases,
});
const validationError = submissionRequestValidationError(request);
@@ -357,31 +195,23 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
}
if (signal.aborted) return null;
runStatus = m.editor_running();
- const browserRequest = projectBrowserSubmission(request, args.workspaceFiles());
- let result: TestRunResult | null;
- if (judgeType === "checker") {
- result = await runCheckerTest(browserRequest, runCases, signal);
- } else if (interactive) {
- result = await runInteractiveTest(browserRequest, sampleIndices, signal);
- } else {
- const local = await runBrowserLocally({
- request: browserRequest,
- cases: runCases,
- judgeConfig: args.judgeConfig(),
- problemId: args.problemId,
- timeLimitMs: args.timeLimitMs,
- memoryLimitMb: args.memoryLimitMb,
- signal,
- });
- result = local && {
- ...local,
- caseResults: local.caseResults?.map((view, index): TestCaseView =>
- view.verdict === "AC" && runCases[index]?.expectedOutput === undefined
- ? { ...view, executionOnly: true }
- : view,
- ),
- };
- }
+ const local = await runBrowserLocally({
+ request: projectBrowserSubmission(request, args.workspaceFiles()),
+ cases: runCases,
+ judgeConfig: args.judgeConfig(),
+ problemId: args.problemId,
+ timeLimitMs: args.timeLimitMs,
+ memoryLimitMb: args.memoryLimitMb,
+ signal,
+ });
+ const result = local && {
+ ...local,
+ caseResults: local.caseResults?.map((view, index): TestCaseView =>
+ view.verdict === "AC" && runCases[index]?.expectedOutput === undefined
+ ? { ...view, executionOnly: true }
+ : view,
+ ),
+ };
return destroyed ? null : result;
}
diff --git a/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte b/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
index e5cdf8778..65858f352 100644
--- a/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
+++ b/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
@@ -2,13 +2,11 @@
import { untrack } from "svelte";
import { invalidateAll } from "$app/navigation";
import type { ProblemDetail } from "$lib/types";
- import type { testJudgeDomain } from "@nojv/application";
import type { JudgeScriptLanguage, JudgeType } from "@nojv/core";
import { inputClassName } from "$lib/utils/css";
import { m } from "$lib/paraglide/messages.js";
import MonacoScriptEditor from "$lib/components/primitives/ui/MonacoScriptEditor.svelte";
import UploadDropZone from "$lib/components/features/problem/admin/UploadDropZone.svelte";
- import JudgeProgramTestStatus from "./judge/JudgeProgramTestStatus.svelte";
import { toasts } from "$lib/stores/toast";
import { submitFormAction } from "$lib/utils/actions";
import {
@@ -21,11 +19,10 @@
interface Props {
problem: ProblemDetail;
validatorScripts: { checkerScript: string; interactorScript: string };
- judgeProgramStatus: testJudgeDomain.JudgeProgramStatus | null;
ondirtychange?: (dirty: boolean) => void;
}
- let { problem, validatorScripts, judgeProgramStatus, ondirtychange }: Props = $props();
+ let { problem, validatorScripts, ondirtychange }: Props = $props();
const cfg = untrack(() => problem.judgeConfig ?? {});
@@ -69,10 +66,9 @@
let initialConfig = $state(dirtySnapshot());
let saving = $state(false);
let saveMessage = $state("");
- let dirty = $derived(dirtySnapshot() !== initialConfig);
$effect(() => {
- ondirtychange?.(dirty);
+ ondirtychange?.(dirtySnapshot() !== initialConfig);
});
export function save() {
@@ -306,20 +302,6 @@
/>
{/if}
-
- {#if judgeProgramStatus && judgeType !== "standard" && judgeType === problem.judgeType}
-
- {/if}
{/if}
{/if}
diff --git a/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts b/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
index 21108b3fa..db0644b38 100644
--- a/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
+++ b/apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts
@@ -24,6 +24,7 @@ import {
preloadBrowserToolchain,
runBrowserCases,
runBrowserChecker,
+ runBrowserInteraction,
runBrowserLocally,
supportsBrowserLocalRun,
} from "$lib/services/browser-local-run";
@@ -71,8 +72,6 @@ export interface EditorRunController {
readonly runSource: "local" | null;
readonly runStatus: string | null;
readonly runError: string | null;
- readonly testDisabledReason: string | null;
- readonly customCasesAllowed: boolean;
readonly cooldownUntil: number | null;
panelRunCases: SubmissionRunCase[];
setBottomTab: (tab: "testcase" | "result") => void;
@@ -87,8 +86,6 @@ function messageForSubmitError(code: string | null): string {
return m.editor_clientTestCustomImage();
case "client_test_language":
return m.editor_clientTestLanguage();
- case "client_test_judge_program":
- return m.editor_testUnavailableForProblem();
case "browser_toolchain_unavailable":
return m.editor_toolchainUnavailable();
case "invalid_source":
@@ -112,8 +109,6 @@ function messageForSubmitError(code: string | null): string {
}
}
-const TEST_DISABLING_CODES = new Set(["client_test_judge_program"]);
-
function initialRunCases(
samples: ProblemDetail["samples"],
judgeType: JudgeType,
@@ -134,7 +129,6 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
let runSource = $state<"local" | null>(null);
let runStatus = $state(null);
let runError = $state(null);
- let testDisabledReason = $state(null);
let cooldownUntil = $state(null);
let panelRunCases = $state(
initialRunCases(args.initialSamples, args.judgeType()),
@@ -143,6 +137,15 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
let destroyed = false;
let abortController: AbortController | null = null;
+ function runLimits(language: Language) {
+ return {
+ language,
+ timeLimitMs: args.timeLimitMs,
+ memoryLimitMb: args.memoryLimitMb,
+ env: args.judgeConfig().runtime?.env ?? {},
+ };
+ }
+
async function runCheckerTest(
request: SubmissionRequest,
runCases: SubmissionRunCase[],
@@ -155,12 +158,7 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
const runs = await runBrowserCases(
build.artifact,
runCases,
- {
- language: request.language,
- timeLimitMs: args.timeLimitMs,
- memoryLimitMb: args.memoryLimitMb,
- env: args.judgeConfig().runtime?.env ?? {},
- },
+ runLimits(request.language),
signal,
);
const caseResults: TestCaseView[] = [];
@@ -194,6 +192,31 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
}
}
+ async function runInteractiveTest(
+ request: SubmissionRequest,
+ runCases: SubmissionRunCase[],
+ interactor: BuildArtifact,
+ signal: AbortSignal,
+ ): Promise {
+ try {
+ const build = await compileBrowserLocally(request, args.problemId, signal);
+ if (!build.ok) return build.result;
+ const caseResults: TestCaseView[] = [];
+ for (const [index, testCase] of runCases.entries()) {
+ const interaction = await runBrowserInteraction(
+ build.artifact,
+ interactor,
+ { interactorInput: testCase.input, limits: runLimits(request.language) },
+ signal,
+ );
+ caseResults.push({ index, ...interaction, judged: true });
+ }
+ return { ...browserLocalSubmissionResult(caseResults), caseResults };
+ } catch (error) {
+ return signal.aborted ? null : browserLocalErrorResult(error);
+ }
+ }
+
async function runSubmission(): Promise {
if (args.isSpecialEnv())
throw new SubmissionRequestError(
@@ -201,12 +224,6 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
"client_test_custom_image",
null,
);
- if (args.judgeType() === "interactive")
- throw new SubmissionRequestError(
- "Test can't run this problem's judge program in the browser yet.",
- "client_test_judge_program",
- null,
- );
const language = args.language();
abortController = new AbortController();
@@ -256,12 +273,18 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
}
if (signal.aborted) return null;
const browserRequest = projectBrowserSubmission(request, args.workspaceFiles());
- if (args.judgeType() === "checker") {
- runStatus = m.editor_checkerPreparing();
- const checker = await args.judgeProgram();
- if (!checker.ok) return null;
+ if (args.judgeType() !== "standard") {
+ runStatus =
+ args.judgeType() === "interactive"
+ ? m.editor_interactorPreparing()
+ : m.editor_checkerPreparing();
+ const judgeProgram = await args.judgeProgram();
+ if (!judgeProgram.ok) return null;
runStatus = m.editor_running();
- const judged = await runCheckerTest(browserRequest, runCases, checker.artifact, signal);
+ const judged =
+ args.judgeType() === "interactive"
+ ? await runInteractiveTest(browserRequest, runCases, judgeProgram.artifact, signal)
+ : await runCheckerTest(browserRequest, runCases, judgeProgram.artifact, signal);
return destroyed ? null : judged;
}
runStatus = m.editor_running();
@@ -302,9 +325,6 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
err instanceof SubmissionRequestError
? messageForSubmitError(err.code)
: m.editor_runFailed();
- if (err instanceof SubmissionRequestError && TEST_DISABLING_CODES.has(err.code ?? "")) {
- testDisabledReason = message;
- }
runError = message;
toasts.error(message);
runStatus = null;
@@ -396,12 +416,6 @@ export function createEditorRunController(args: EditorRunArgs): EditorRunControl
get runError() {
return runError;
},
- get testDisabledReason() {
- return testDisabledReason;
- },
- get customCasesAllowed() {
- return args.judgeType() !== "interactive";
- },
get cooldownUntil() {
return cooldownUntil;
},
diff --git a/apps/web/src/lib/services/browser-local-run.ts b/apps/web/src/lib/services/browser-local-run.ts
index 9b2fb023d..a5fd08c66 100644
--- a/apps/web/src/lib/services/browser-local-run.ts
+++ b/apps/web/src/lib/services/browser-local-run.ts
@@ -1,12 +1,18 @@
import {
+ DEFAULT_MAX_MEMORY_MB,
+ DEFAULT_MEMORY_HEADROOM_MB,
+ MAX_CASE_STDERR_BYTES,
MAX_EXECUTION_OUTPUT_BYTES,
checkerCaseVerdict,
compareStandard,
entryFileNameFor,
effectiveTimeLimitMs,
executionWallTimeLimitMs,
+ interactiveCaseVerdict,
isBrowserLocalLanguage,
judgeProgramCompileInput,
+ resolveContainerMemoryMb,
+ truncateUtf8,
validatorTimeoutMs,
wasmOjTerminationVerdict,
withCppPlatformHeaders,
@@ -48,8 +54,11 @@ const BROWSER_TOOLCHAINS = [
rustSource(BROWSER_TOOLCHAIN_BASE_URL),
];
const PRELOAD_RETRY_DELAYS_MS = [2_000, 5_000];
+const JUDGE_PROGRAM_ARGS = ["/judge/input", "/judge/answer", "/judge/feedback"];
const CHECKER_MEMORY_LIMIT_BYTES = 512 * 1024 * 1024;
const CHECKER_TEAM_MESSAGE_PATH = "/judge/feedback/teammessage.txt";
+const MIN_INTERACTIVE_WALL_LIMIT_MS = 3_000;
+const INTERACTION_TRANSCRIPT_BYTES = 64 * 1024;
const RUN_OUTPUT_LIMITS = {
outputLimitBytes: MAX_EXECUTION_OUTPUT_BYTES,
filesystemWriteLimitBytes: 64 * 1024 * 1024,
@@ -243,6 +252,13 @@ interface BrowserRunLimits {
env: Record;
}
+interface BrowserInteraction {
+ verdict: ReturnType["verdict"];
+ timeMs: number;
+ transcript: { toInteractor: string; toContestant: string };
+ stderr?: string;
+}
+
type BrowserCompileOutcome =
{ ok: true; artifact: BuildArtifact } | { ok: false; result: SubmissionResult };
@@ -460,7 +476,7 @@ export async function runBrowserChecker(
return withBrowserEngine(signal, async (browserEngine) => {
const timeoutMs = validatorTimeoutMs(timeLimitMs);
const run = await browserEngine.run(artifact, {
- args: ["/judge/input", "/judge/answer", "/judge/feedback"],
+ args: JUDGE_PROGRAM_ARGS,
stdin: output,
files: {
"/judge/input": input,
@@ -484,6 +500,66 @@ export async function runBrowserChecker(
});
}
+export async function runBrowserInteraction(
+ contestant: BuildArtifact,
+ interactor: BuildArtifact,
+ { interactorInput, limits }: { interactorInput: string; limits: BrowserRunLimits },
+ signal: AbortSignal,
+): Promise {
+ return withBrowserEngine(signal, async (browserEngine) => {
+ const timeLimitMs = effectiveTimeLimitMs(limits.timeLimitMs, limits.language);
+ const wallTimeLimitMs = Math.max(MIN_INTERACTIVE_WALL_LIMIT_MS, 3 * timeLimitMs);
+ const interactorMemoryMb = resolveContainerMemoryMb(limits.memoryLimitMb, {
+ defaultMemoryMb: limits.memoryLimitMb,
+ headroomMb: DEFAULT_MEMORY_HEADROOM_MB,
+ maxMemoryMb: DEFAULT_MAX_MEMORY_MB,
+ });
+ const run = await browserEngine.interact(contestant, interactor, {
+ contestant: {
+ env: limits.env,
+ resources: {
+ logicalTimeLimitMs: timeLimitMs,
+ memoryLimitBytes: limits.memoryLimitMb * 1024 * 1024,
+ wallTimeLimitMs,
+ ...RUN_OUTPUT_LIMITS,
+ },
+ },
+ interactor: {
+ args: JUDGE_PROGRAM_ARGS,
+ files: {
+ "/judge/input": interactorInput,
+ "/judge/answer": "",
+ "/judge/feedback/.keep": "",
+ },
+ resources: {
+ logicalTimeLimitMs: validatorTimeoutMs(timeLimitMs),
+ memoryLimitBytes: interactorMemoryMb * 1024 * 1024,
+ wallTimeLimitMs,
+ ...RUN_OUTPUT_LIMITS,
+ },
+ },
+ });
+ signal.throwIfAborted();
+ const contestantStop = wasmOjTerminationVerdict(
+ run.contestant.termination,
+ run.contestant.code,
+ );
+ const stderr = truncateUtf8(run.contestant.stderr, MAX_CASE_STDERR_BYTES);
+ return {
+ verdict:
+ contestantStop === "TLE" || contestantStop === "MLE"
+ ? contestantStop
+ : interactiveCaseVerdict(run).verdict,
+ timeMs: Math.max(0, Math.ceil((run.contestant.metrics.logicalTimeNs ?? 0) / 1_000_000)),
+ transcript: {
+ toInteractor: truncateUtf8(run.contestantToInteractor, INTERACTION_TRANSCRIPT_BYTES),
+ toContestant: truncateUtf8(run.interactorToContestant, INTERACTION_TRANSCRIPT_BYTES),
+ },
+ ...(stderr ? { stderr } : {}),
+ };
+ });
+}
+
export async function runBrowserLocally(args: {
request: SubmissionRequest;
cases: SubmissionRunCase[];
diff --git a/tests/component/web/editor-output-comparison.test.ts b/tests/component/web/editor-output-comparison.test.ts
index bc70b41cb..8e89107f6 100644
--- a/tests/component/web/editor-output-comparison.test.ts
+++ b/tests/component/web/editor-output-comparison.test.ts
@@ -82,29 +82,64 @@ function result(caseResults: TestCaseView[]): TestRunResult {
};
}
-it("shows interactive samples read-only with the interaction notes", async () => {
- await mountPanel({
- judgeType: "interactive",
- customCasesAllowed: false,
- interactionFormat: "Read **n** then guess.",
- runCases: [{ input: "1 100\n42\n" }, { input: "1 10\n7\n" }],
+it("edits interactive cases as interactor inputs next to the interaction notes", async () => {
+ let runCases = [{ input: "1 100\n42\n" }, { input: "1 10\n7\n" }];
+ target = document.createElement("div");
+ document.body.append(target);
+ component = mount(Panel, {
+ target,
+ props: {
+ get runCases() {
+ return runCases;
+ },
+ set runCases(next) {
+ runCases = next;
+ },
+ judgeType: "interactive",
+ interactionFormat: "Read **n** then guess.",
+ tab: "testcase",
+ runResult: null,
+ runStatus: null,
+ runError: null,
+ ontabchange: () => {},
+ },
});
- expect(target.querySelector("textarea")).toBeNull();
- expect(target.querySelector('input[type="checkbox"]')).toBeNull();
- expect(target.querySelector(`button[aria-label="${m.editor_testcase()}"]`)).toBeNull();
- expect(
- target.querySelector(`button[aria-label="${m.editor_removeCase({ index: 1 })}"]`),
- ).toBeNull();
+ await tick();
+
expect(target.textContent).toContain(m.problemDetail_interactionFormat());
expect(target.querySelector("strong")?.textContent).toBe("n");
expect(target.textContent).toContain(m.problemDetail_interactorInput());
- expect(target.querySelector("pre")?.textContent).toBe("1 100\n42\n");
+ expect(target.textContent).not.toContain(m.editor_input());
expect(target.textContent).toContain(m.editor_interactiveTestNote());
+ expect(target.querySelector('input[type="checkbox"]')).toBeNull();
+ expect(target.querySelector(`textarea[aria-label="${m.editor_expectLabel()}"]`)).toBeNull();
+ const input = target.querySelector("textarea")!;
+ expect(input.value).toBe("1 100\n42\n");
+
+ target
+ .querySelector(`button[aria-label="${m.editor_testcase()}"]`)!
+ .click();
+ await tick();
+ expect(runCases).toHaveLength(3);
+ expect(input.value).toBe("");
+ input.value = "1 1000\n999\n";
+ input.dispatchEvent(new Event("input", { bubbles: true }));
+ await tick();
+ expect(runCases[2]).toEqual({ input: "1 1000\n999\n" });
+
+ target
+ .querySelector(
+ `button[aria-label="${m.editor_removeCase({ index: 1 })}"]`,
+ )!
+ .click();
+ await tick();
+ expect(runCases).toEqual([{ input: "1 10\n7\n" }, { input: "1 1000\n999\n" }]);
});
-it("explains that an interactive problem without interactor samples has nothing to test", async () => {
- await mountPanel({ judgeType: "interactive", customCasesAllowed: false, runCases: [] });
- expect(target.textContent).toContain(m.editor_testNoInteractiveSamples());
+it("lets an interactive problem without interactor samples start from an empty case list", async () => {
+ await mountPanel({ judgeType: "interactive", runCases: [] });
+ expect(target.querySelector(`button[aria-label="${m.editor_testcase()}"]`)).not.toBeNull();
+ expect(target.textContent).toContain(m.problemDetail_interactorInput());
});
it("hides output comparison on checker problems and explains that only samples are judged", async () => {
@@ -130,7 +165,6 @@ it("renders the interactive transcript instead of an empty output block", async
await mountPanel({
tab: "result",
judgeType: "interactive",
- customCasesAllowed: false,
runSource: "local",
runCases: [{ input: "1 100\n42\n" }],
runResult: result([
@@ -221,6 +255,29 @@ it("explains a judge error on a case the checker judged", async () => {
expect(target.textContent).toContain(m.editor_judgeSystemError());
});
+it("explains a judge error and shows the program's stderr on an interactive case", async () => {
+ await mountPanel({
+ tab: "result",
+ judgeType: "interactive",
+ runSource: "local",
+ runCases: [{ input: "1 100\n42\n" }],
+ runResult: result([
+ {
+ index: 0,
+ verdict: "SE",
+ judged: true,
+ timeMs: 2,
+ stderr: "debug: guessing 50",
+ transcript: { toInteractor: "50\n", toContestant: "" },
+ },
+ ]),
+ });
+ expect(target.textContent).toContain(m.editor_judgeSystemError());
+ expect(target.textContent).toContain(m.submissionDetail_stderr());
+ expect(target.textContent).toContain("debug: guessing 50");
+ expect(target.textContent).toContain(m.editor_transcript());
+});
+
it("does not blame the judge for a system error before any judging", async () => {
await mountPanel({
tab: "result",
diff --git a/tests/component/web/editor-shortcuts.test.ts b/tests/component/web/editor-shortcuts.test.ts
index a2fb24cba..2cd89390e 100644
--- a/tests/component/web/editor-shortcuts.test.ts
+++ b/tests/component/web/editor-shortcuts.test.ts
@@ -10,7 +10,6 @@ const mocks = vi.hoisted(() => ({
run: {
isSubmitting: false,
panelRunCases: [],
- testDisabledReason: null,
markDestroyed: vi.fn(),
},
}));
diff --git a/tests/component/web/editor-test-button-state.test.ts b/tests/component/web/editor-test-button-state.test.ts
index 4ceee435c..cf4fe94b1 100644
--- a/tests/component/web/editor-test-button-state.test.ts
+++ b/tests/component/web/editor-test-button-state.test.ts
@@ -13,7 +13,6 @@ const mocks = vi.hoisted(() => ({
preload: vi.fn(() => Promise.resolve()),
prepare:
vi.fn<(scope: unknown, onProgress: (progress: Progress) => void) => Promise>(),
- controllerReason: null as string | null,
}));
vi.mock("$lib/services/browser-local-run", () => ({
supportsBrowserLocalRun: () => true,
@@ -35,9 +34,6 @@ vi.mock("$lib/components/features/problem/editors/use-editor-run.svelte", () =>
isRunning: false,
isSubmitting: false,
panelRunCases: [],
- get testDisabledReason() {
- return mocks.controllerReason;
- },
markDestroyed: vi.fn(),
setBottomTab: vi.fn(),
submit: vi.fn(),
@@ -70,7 +66,6 @@ afterEach(async () => {
if (component) await unmount(component);
component = undefined;
target.remove();
- mocks.controllerReason = null;
vi.clearAllMocks();
});
@@ -139,17 +134,6 @@ describe("Test button state", () => {
expect(mocks.preload).not.toHaveBeenCalled();
});
- it("is disabled on an interactive problem in C++", async () => {
- const button = await renderTestButton({
- judgeType: "interactive",
- samples: interactiveSamples,
- });
- expect(button.disabled).toBe(true);
- expect(visibleReason(button)).toBe(m.editor_testUnavailableForProblem());
- expect(mocks.preload).not.toHaveBeenCalled();
- expect(mocks.prepare).not.toHaveBeenCalled();
- });
-
it("says the checker is preparing and keeps Test clickable while it builds", async () => {
mocks.prepare.mockImplementation((_scope, onProgress) => {
onProgress({ phase: "build" });
@@ -201,23 +185,84 @@ describe("Test button state", () => {
expect(button.textContent).toContain(m.editor_run());
});
- it("is disabled for an interactive problem in JavaScript", async () => {
+ it("says the interactor is preparing and keeps Test clickable while it builds", async () => {
+ mocks.prepare.mockImplementation((_scope, onProgress) => {
+ onProgress({ phase: "build" });
+ return new Promise(() => undefined);
+ });
const button = await renderTestButton({
judgeType: "interactive",
- language: "javascript",
samples: interactiveSamples,
});
- expect(button.disabled).toBe(true);
- expect(visibleReason(button)).toBe(m.editor_testInteractiveLanguage());
+ await vi.waitFor(() =>
+ expect(button.textContent).toContain(m.editor_interactorPreparing()),
+ );
+ expect(mocks.preload).toHaveBeenCalledWith("cpp", expect.any(Function));
+ expect(button.disabled).toBe(false);
+ expect(button.getAttribute("aria-busy")).toBe("true");
+ expect(visibleReason(button)).toBeUndefined();
});
- it("is disabled for an interactive problem without interactor samples", async () => {
+ it("is enabled once the interactor is built", async () => {
+ mocks.prepare.mockResolvedValue({ ...checkerReady, role: "interactor" });
+ const button = await renderTestButton({
+ judgeType: "interactive",
+ samples: interactiveSamples,
+ });
+ await vi.waitFor(() => expect(mocks.prepare).toHaveBeenCalledOnce());
+ await tick();
+ expect(button.textContent).toContain(m.editor_run());
+ expect(button.disabled).toBe(false);
+ expect(visibleReason(button)).toBeUndefined();
+ });
+
+ it.each([
+ [
+ "fails to build",
+ { ok: false, reason: "build_failed", diagnostics: "error" } as const,
+ () => m.editor_interactorBuildFailed(),
+ ],
+ [
+ "can't be loaded",
+ { ok: false, reason: "load_failed" } as const,
+ () => m.editor_interactorLoadFailed(),
+ ],
+ [
+ "is refused",
+ { ok: false, reason: "unavailable" } as const,
+ () => m.editor_interactorUnavailable(),
+ ],
+ ])("is disabled when the interactor %s", async (_label, prepared, reason) => {
+ mocks.prepare.mockResolvedValue(prepared);
+ const button = await renderTestButton({
+ judgeType: "interactive",
+ samples: interactiveSamples,
+ });
+ await vi.waitFor(() => expect(button.disabled).toBe(true));
+ expect(visibleReason(button)).toBe(reason());
+ });
+
+ it("is enabled for an interactive problem without interactor samples", async () => {
+ mocks.prepare.mockResolvedValue({ ...checkerReady, role: "interactor" });
const button = await renderTestButton({
judgeType: "interactive",
samples: [{ input: "1", output: "1" }],
});
+ await vi.waitFor(() => expect(mocks.prepare).toHaveBeenCalledOnce());
+ expect(button.disabled).toBe(false);
+ expect(visibleReason(button)).toBeUndefined();
+ });
+
+ it("is disabled for an interactive problem in JavaScript without preparing anything", async () => {
+ const button = await renderTestButton({
+ judgeType: "interactive",
+ language: "javascript",
+ samples: interactiveSamples,
+ });
expect(button.disabled).toBe(true);
- expect(visibleReason(button)).toBe(m.editor_testNoInteractiveSamples());
+ expect(visibleReason(button)).toBe(m.editor_testInteractiveLanguage());
+ expect(mocks.preload).not.toHaveBeenCalled();
+ expect(mocks.prepare).not.toHaveBeenCalled();
});
it("is enabled on a standard problem and preloads the toolchain", async () => {
@@ -230,12 +275,4 @@ describe("Test button state", () => {
);
expect(mocks.prepare).not.toHaveBeenCalled();
});
-
- it("is disabled with the controller's reason", async () => {
- mocks.controllerReason = m.editor_testUnavailableForProblem();
- const button = await renderTestButton({});
- expect(button.disabled).toBe(true);
- expect(visibleReason(button)).toBe(m.editor_testUnavailableForProblem());
- expect(mocks.preload).not.toHaveBeenCalled();
- });
});
diff --git a/tests/component/web/editor-toolchain-preload.test.ts b/tests/component/web/editor-toolchain-preload.test.ts
index 015c51c78..2ae342977 100644
--- a/tests/component/web/editor-toolchain-preload.test.ts
+++ b/tests/component/web/editor-toolchain-preload.test.ts
@@ -23,7 +23,6 @@ const mocks = vi.hoisted(() => ({
run: {
isSubmitting: false,
panelRunCases: [],
- testDisabledReason: null,
markDestroyed: vi.fn(),
},
}));
diff --git a/tests/unit/web/browser-local-execution.test.ts b/tests/unit/web/browser-local-execution.test.ts
index db056a39b..9c7008b81 100644
--- a/tests/unit/web/browser-local-execution.test.ts
+++ b/tests/unit/web/browser-local-execution.test.ts
@@ -2,6 +2,7 @@ import { expect, it, vi } from "vitest";
import {
runBrowserCases,
runBrowserChecker,
+ runBrowserInteraction,
runBrowserLocally,
} from "$lib/services/browser-local-run";
@@ -18,6 +19,7 @@ const engine = vi.hoisted(() => ({
durationMs: 1,
metrics: { logicalTimeNs: 1, memoryBytes: 1024 },
}),
+ interact: vi.fn(),
cancel: vi.fn(),
}));
vi.mock("../../../apps/web/node_modules/@wasm-oj/browser", async (importOriginal) => ({
@@ -470,3 +472,175 @@ it.each([
),
).resolves.toEqual({ verdict: "SE" });
});
+
+function side(termination: string, code: number, stderr = "", logicalTimeNs = 1) {
+ return { termination, code, stderr, metrics: { logicalTimeNs, memoryBytes: 1024 } };
+}
+
+function interaction(overrides: Record = {}) {
+ return {
+ contestant: side("exited", 0),
+ interactor: side("exited", 42),
+ contestantToInteractor: "50\n",
+ interactorToContestant: "1 100\n",
+ durationMs: 1,
+ ...overrides,
+ };
+}
+
+const interactionLimits = {
+ language: "python",
+ timeLimitMs: 1000,
+ memoryLimitMb: 256,
+ env: { MODE: "strict" },
+} as const;
+
+it("runs the contestant against the interactor with the sample's input and official limits", async () => {
+ const contestant = { id: "contestant" } as never;
+ const interactor = { id: "interactor" } as never;
+ engine.interact.mockResolvedValueOnce(
+ interaction({ contestant: side("exited", 0, "", 12_300_000) }),
+ );
+
+ const result = await runBrowserInteraction(
+ contestant,
+ interactor,
+ { interactorInput: "1 100\n42\n", limits: interactionLimits },
+ new AbortController().signal,
+ );
+
+ expect(result).toEqual({
+ verdict: "AC",
+ timeMs: 13,
+ transcript: { toInteractor: "50\n", toContestant: "1 100\n" },
+ });
+ expect(engine.interact).toHaveBeenLastCalledWith(contestant, interactor, {
+ contestant: {
+ env: { MODE: "strict" },
+ resources: {
+ logicalTimeLimitMs: 3000,
+ memoryLimitBytes: 256 * 1024 * 1024,
+ wallTimeLimitMs: 9000,
+ outputLimitBytes: 16 * 1024 * 1024,
+ filesystemWriteLimitBytes: 64 * 1024 * 1024,
+ filesystemEntryLimit: 4096,
+ },
+ },
+ interactor: {
+ args: ["/judge/input", "/judge/answer", "/judge/feedback"],
+ files: {
+ "/judge/input": "1 100\n42\n",
+ "/judge/answer": "",
+ "/judge/feedback/.keep": "",
+ },
+ resources: {
+ logicalTimeLimitMs: 30_000,
+ memoryLimitBytes: 320 * 1024 * 1024,
+ wallTimeLimitMs: 9000,
+ outputLimitBytes: 16 * 1024 * 1024,
+ filesystemWriteLimitBytes: 64 * 1024 * 1024,
+ filesystemEntryLimit: 4096,
+ },
+ },
+ });
+});
+
+it("keeps short limits at a 3 s wall stop and caps the interactor's memory headroom", async () => {
+ engine.interact.mockResolvedValueOnce(interaction());
+
+ await runBrowserInteraction(
+ { id: "contestant" } as never,
+ { id: "interactor" } as never,
+ {
+ interactorInput: "",
+ limits: { language: "cpp", timeLimitMs: 200, memoryLimitMb: 1500, env: {} },
+ },
+ new AbortController().signal,
+ );
+
+ const config = engine.interact.mock.calls.at(-1)?.[2];
+ expect(config.contestant.resources).toMatchObject({
+ logicalTimeLimitMs: 200,
+ memoryLimitBytes: 1500 * 1024 * 1024,
+ wallTimeLimitMs: 3000,
+ });
+ expect(config.interactor.resources).toMatchObject({
+ logicalTimeLimitMs: 30_000,
+ memoryLimitBytes: 1536 * 1024 * 1024,
+ wallTimeLimitMs: 3000,
+ });
+});
+
+it.each([
+ ["the interactor rejects the replies", interaction({ interactor: side("exited", 43) }), "WA"],
+ ["the interactor crashes", interaction({ interactor: side("trap", 1) }), "SE"],
+ [
+ "the contestant runs out of instructions",
+ interaction({ contestant: side("instruction-limit", 0) }),
+ "TLE",
+ ],
+ [
+ "the contestant exceeds its memory",
+ interaction({ contestant: side("memory-limit", 0) }),
+ "MLE",
+ ],
+ ["the contestant exits with an error", interaction({ contestant: side("exited", 3) }), "RE"],
+ [
+ "the contestant runs out of instructions and the interactor dies on the closed pipe",
+ interaction({
+ contestant: side("instruction-limit", 137),
+ interactor: side("exited", 120),
+ }),
+ "TLE",
+ ],
+ [
+ "the contestant exceeds its memory and the interactor dies on the closed pipe",
+ interaction({ contestant: side("memory-limit", 0), interactor: side("exited", 120) }),
+ "MLE",
+ ],
+ [
+ "the contestant exits with an error and the interactor crashes",
+ interaction({ contestant: side("exited", 3), interactor: side("exited", 120) }),
+ "SE",
+ ],
+ [
+ "both sides hit the wall stop",
+ interaction({
+ contestant: side("wall-time-limit", 0),
+ interactor: side("wall-time-limit", 0),
+ }),
+ "TLE",
+ ],
+])("maps an interaction where %s", async (_label, run, verdict) => {
+ engine.interact.mockResolvedValueOnce(run);
+
+ const result = await runBrowserInteraction(
+ { id: "contestant" } as never,
+ { id: "interactor" } as never,
+ { interactorInput: "", limits: interactionLimits },
+ new AbortController().signal,
+ );
+
+ expect(result.verdict).toBe(verdict);
+});
+
+it("caps each transcript direction at 64 KiB and the contestant's stderr", async () => {
+ engine.interact.mockResolvedValueOnce(
+ interaction({
+ contestant: side("exited", 0, "e".repeat(200_000)),
+ contestantToInteractor: "q".repeat(70_000),
+ interactorToContestant: `${"a".repeat(65_535)}中`,
+ }),
+ );
+
+ const result = await runBrowserInteraction(
+ { id: "contestant" } as never,
+ { id: "interactor" } as never,
+ { interactorInput: "", limits: interactionLimits },
+ new AbortController().signal,
+ );
+
+ expect(result.transcript.toInteractor).toBe("q".repeat(64 * 1024));
+ expect(result.transcript.toContestant).toBe("a".repeat(65_535));
+ expect(result.stderr).toBe("e".repeat(100_000));
+});
diff --git a/tests/unit/web/editor-client-test.test.ts b/tests/unit/web/editor-client-test.test.ts
index 51ad6736d..85867dfe2 100644
--- a/tests/unit/web/editor-client-test.test.ts
+++ b/tests/unit/web/editor-client-test.test.ts
@@ -11,6 +11,7 @@ const mocks = vi.hoisted(() => ({
compile: vi.fn(),
runCases: vi.fn(),
check: vi.fn(),
+ interact: vi.fn(),
toast: vi.fn(),
}));
vi.mock("$lib/services/submission-service", async (importOriginal) => ({
@@ -24,6 +25,7 @@ vi.mock("$lib/services/browser-local-run", async (importOriginal) => ({
compileBrowserLocally: mocks.compile,
runBrowserCases: mocks.runCases,
runBrowserChecker: mocks.check,
+ runBrowserInteraction: mocks.interact,
}));
vi.mock("$lib/stores/toast", () => ({ toasts: { error: mocks.toast } }));
import { createEditorRunController } from "$lib/components/features/problem/editors/use-editor-run.svelte";
@@ -193,16 +195,118 @@ const interactiveSamples = [
{ input: "? 7", output: "=", interactorInput: "7" },
];
-it("reports interactive Test as unavailable and disables it without compiling", async () => {
- const run = controller("interactive", { samples: interactiveSamples });
+const interactorReady: PreparedJudgeProgram = {
+ ok: true,
+ role: "interactor",
+ language: "python",
+ artifact: checkerArtifact,
+};
+
+function interacted(verdict: string, toInteractor: string, toContestant: string) {
+ return { verdict, timeMs: 4, transcript: { toInteractor, toContestant } };
+}
+
+it("runs interactive samples and custom cases through the prepared interactor", async () => {
+ mocks.compile.mockResolvedValue({ ok: true, artifact: { id: "contestant" } });
+ mocks.interact
+ .mockResolvedValueOnce(interacted("AC", "50\n42\n", "lower\ncorrect\n"))
+ .mockResolvedValueOnce({ ...interacted("TLE", "", ""), stderr: "spinning" })
+ .mockResolvedValueOnce(interacted("WA", "1\n", "higher\n"));
+ const run = controller("interactive", {
+ samples: interactiveSamples,
+ judgeProgram: () => Promise.resolve(interactorReady),
+ });
+ run.panelRunCases = [...run.panelRunCases, { input: "99" }];
+
await run.run();
- expect(mocks.preload).not.toHaveBeenCalled();
+
+ expect(mocks.compile).toHaveBeenCalledOnce();
expect(mocks.run).not.toHaveBeenCalled();
- expect(mocks.execute).not.toHaveBeenCalled();
+ expect(mocks.runCases).not.toHaveBeenCalled();
+ expect(mocks.check).not.toHaveBeenCalled();
+ expect(
+ mocks.interact.mock.calls.map(([contestant, interactor, data]) => [
+ contestant,
+ interactor,
+ data,
+ ]),
+ ).toEqual(
+ ["42", "7", "99"].map((interactorInput) => [
+ { id: "contestant" },
+ checkerArtifact,
+ {
+ interactorInput,
+ limits: { language: "cpp", timeLimitMs: 1000, memoryLimitMb: 128, env: {} },
+ },
+ ]),
+ );
+ expect(run.runResult?.verdict).toBe("time_limit_exceeded");
+ expect(run.runResult?.caseResults).toEqual([
+ {
+ index: 0,
+ verdict: "AC",
+ timeMs: 4,
+ judged: true,
+ transcript: { toInteractor: "50\n42\n", toContestant: "lower\ncorrect\n" },
+ },
+ expect.objectContaining({ index: 1, verdict: "TLE", judged: true, stderr: "spinning" }),
+ expect.objectContaining({ index: 2, verdict: "WA", judged: true }),
+ ]);
+});
+
+it("waits for the interactor and stops when it can't be loaded", async () => {
+ let finishPreparing!: (prepared: PreparedJudgeProgram) => void;
+ const run = controller("interactive", {
+ samples: interactiveSamples,
+ judgeProgram: () =>
+ new Promise((resolve) => (finishPreparing = resolve)),
+ });
+
+ const pending = run.run();
+ await vi.waitFor(() => expect(run.runStatus).toBe(m.editor_interactorPreparing()));
+ expect(mocks.compile).not.toHaveBeenCalled();
+
+ finishPreparing({ ok: false, reason: "load_failed" });
+ await pending;
+ expect(mocks.compile).not.toHaveBeenCalled();
+ expect(mocks.interact).not.toHaveBeenCalled();
expect(run.runResult).toBeNull();
- expect(run.runError).toBe(m.editor_testUnavailableForProblem());
- expect(mocks.toast).toHaveBeenCalledWith(m.editor_testUnavailableForProblem());
- expect(run.testDisabledReason).toBe(m.editor_testUnavailableForProblem());
+ expect(run.runError).toBeNull();
+});
+
+it("shows the contestant's compile error without starting an interaction", async () => {
+ const compileError = {
+ accepted: false,
+ caseResults: [],
+ feedback: "main.cpp:1: error",
+ runtimeMs: 0,
+ score: 0,
+ verdict: "compile_error",
+ };
+ mocks.compile.mockResolvedValue({ ok: false, result: compileError });
+ const run = controller("interactive", {
+ samples: interactiveSamples,
+ judgeProgram: () => Promise.resolve(interactorReady),
+ });
+
+ await run.run();
+
+ expect(run.runResult).toEqual(compileError);
+ expect(mocks.interact).not.toHaveBeenCalled();
+});
+
+it("reports an engine failure during an interaction as a system error", async () => {
+ mocks.compile.mockResolvedValue({ ok: true, artifact: { id: "contestant" } });
+ mocks.interact.mockRejectedValueOnce(new Error("Worker crashed"));
+ const run = controller("interactive", {
+ samples: interactiveSamples,
+ judgeProgram: () => Promise.resolve(interactorReady),
+ });
+
+ await run.run();
+
+ expect(run.runResult).toMatchObject({ verdict: "system_error", caseResults: [] });
+ expect(run.runResult?.feedback).toContain("Worker crashed");
});
function exited(stdout: string) {
@@ -301,11 +405,9 @@ it("waits for the checker before compiling and stops when it failed to build", a
expect(mocks.toast).not.toHaveBeenCalled();
});
-it("starts interactive cases from the samples' interactor inputs and allows no custom cases", () => {
+it("starts interactive cases from the samples' interactor inputs", () => {
const run = controller("interactive", { samples: interactiveSamples });
expect(run.panelRunCases).toEqual([{ input: "42" }, { input: "7" }]);
- expect(run.customCasesAllowed).toBe(false);
- expect(controller("checker", { samples: checkerSamples }).customCasesAllowed).toBe(true);
});
it("ignores a second Test press while one is running", async () => {
From 1ea07423767738b2b3a4c594e0a03a28b0d523a4 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 19:53:58 +0800
Subject: [PATCH 12/22] fix(web): merge interactive Test verdicts like official
judging
Official interactive judging (resolveInteractiveStage) checks the
interactor first: an interactor that exits with anything but 42 or 43,
or is stopped, is SE even when the contestant hit a limit; only then does
the contestant's TLE, MLE or RE win over the interactor's AC or WA.
Core's interactiveCaseVerdict already follows that order, so browser
Test now uses it as is and drops its own rule that kept a contestant's
TLE or MLE over a failed interactor.
Official judging never breaks the interactor's pipe: its channel keeps
reading the interactor's output after the contestant ends, so the
wrapper's read() sees EOF and exits 43, and the case stays TLE. The
browser engine closes the pipe when the contestant stops, so a Python
interactor that writes after that exits 120 and the case shows SE in
Test, as does an interaction where both sides reach the shared wall stop.
Co-Authored-By: Claude Opus 5.5
---
.../web/src/lib/services/browser-local-run.ts | 9 +------
tests/unit/core/test-judge-verdict.test.ts | 27 ++++++++++++++-----
.../unit/web/browser-local-execution.test.ts | 14 +++++-----
3 files changed, 29 insertions(+), 21 deletions(-)
diff --git a/apps/web/src/lib/services/browser-local-run.ts b/apps/web/src/lib/services/browser-local-run.ts
index a5fd08c66..e2e44a99e 100644
--- a/apps/web/src/lib/services/browser-local-run.ts
+++ b/apps/web/src/lib/services/browser-local-run.ts
@@ -540,16 +540,9 @@ export async function runBrowserInteraction(
},
});
signal.throwIfAborted();
- const contestantStop = wasmOjTerminationVerdict(
- run.contestant.termination,
- run.contestant.code,
- );
const stderr = truncateUtf8(run.contestant.stderr, MAX_CASE_STDERR_BYTES);
return {
- verdict:
- contestantStop === "TLE" || contestantStop === "MLE"
- ? contestantStop
- : interactiveCaseVerdict(run).verdict,
+ verdict: interactiveCaseVerdict(run).verdict,
timeMs: Math.max(0, Math.ceil((run.contestant.metrics.logicalTimeNs ?? 0) / 1_000_000)),
transcript: {
toInteractor: truncateUtf8(run.contestantToInteractor, INTERACTION_TRANSCRIPT_BYTES),
diff --git a/tests/unit/core/test-judge-verdict.test.ts b/tests/unit/core/test-judge-verdict.test.ts
index 266486dcf..d19306cd3 100644
--- a/tests/unit/core/test-judge-verdict.test.ts
+++ b/tests/unit/core/test-judge-verdict.test.ts
@@ -95,14 +95,29 @@ describe("interactiveCaseVerdict", () => {
).toEqual({ verdict: "SE" });
});
- it("lets an interactor protocol failure win over a contestant failure", () => {
+ it("keeps a contestant limit when the interactor then reads EOF and rejects", () => {
expect(
- interactiveCaseVerdict({
- contestant: { termination: "memory-limit", code: 0 },
- interactor: exited(1),
- }),
- ).toEqual({ verdict: "SE" });
+ interactiveCaseVerdict(
+ {
+ contestant: { termination: "logical-time-limit", code: 0 },
+ interactor: exited(43),
+ },
+ "solution closed its output early",
+ ),
+ ).toEqual({ verdict: "TLE" });
});
+
+ it.each(["memory-limit", "instruction-limit", "wall-time-limit"])(
+ "lets an interactor failure win over a contestant %s, as official judging does",
+ (termination) => {
+ expect(
+ interactiveCaseVerdict({
+ contestant: { termination, code: 0 },
+ interactor: exited(120),
+ }),
+ ).toEqual({ verdict: "SE" });
+ },
+ );
});
describe("truncateUtf8", () => {
diff --git a/tests/unit/web/browser-local-execution.test.ts b/tests/unit/web/browser-local-execution.test.ts
index 9c7008b81..2fc7ec8df 100644
--- a/tests/unit/web/browser-local-execution.test.ts
+++ b/tests/unit/web/browser-local-execution.test.ts
@@ -585,18 +585,18 @@ it.each([
"MLE",
],
["the contestant exits with an error", interaction({ contestant: side("exited", 3) }), "RE"],
+ [
+ "the contestant hits the wall stop and the interactor rejects the closed input",
+ interaction({ contestant: side("wall-time-limit", 0), interactor: side("exited", 43) }),
+ "TLE",
+ ],
[
"the contestant runs out of instructions and the interactor dies on the closed pipe",
interaction({
contestant: side("instruction-limit", 137),
interactor: side("exited", 120),
}),
- "TLE",
- ],
- [
- "the contestant exceeds its memory and the interactor dies on the closed pipe",
- interaction({ contestant: side("memory-limit", 0), interactor: side("exited", 120) }),
- "MLE",
+ "SE",
],
[
"the contestant exits with an error and the interactor crashes",
@@ -609,7 +609,7 @@ it.each([
contestant: side("wall-time-limit", 0),
interactor: side("wall-time-limit", 0),
}),
- "TLE",
+ "SE",
],
])("maps an interaction where %s", async (_label, run, verdict) => {
engine.interact.mockResolvedValueOnce(run);
From 9d40836313463544b54ee4d9b0421f416e58be75 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:08:10 +0800
Subject: [PATCH 13/22] docs(decisions): Test runs entirely in the browser,
judge programs included
JDG-15 now records that Test runs the problem's checker or interactor in
the student's browser, that students may read judge programs and authors
own their robustness, that the server compiles and executes nothing for
Test, and that Test never receives non-sample testcase data. It rejects
server-side judge programs (#641), server compilation, execution-only
Test and exposure toggles, keeps the earlier rejections that still hold,
and drops the rejections of shipping judge programs and compiled judge
Wasm to the browser. Generic runtime fixes land in wasm-oj/forge and are
consumed as pinned releases.
JDG-26, SEC-15 and OPS-21 are withdrawn and point to JDG-15. JDG-05 is
scoped to official judging. JDG-03 keeps the byte-identical wrapper rule
for the browser copy. JDG-12, PRB-22 and OPS-18 return to their text
before #641; PRB-03 and WEB-05 describe browser Test and the
judge-program endpoint's view access. PRB-01 and PRB-09 drop the hidden
visibility, and PRB-01 rejects it.
Co-Authored-By: Claude Opus 5.5
---
docs/decisions/README.md | 10 ++---
docs/decisions/judge.md | 80 ++++++++++++++++----------------------
docs/decisions/platform.md | 14 ++-----
docs/decisions/problems.md | 22 +++++------
docs/decisions/security.md | 12 ++----
docs/decisions/web.md | 4 +-
6 files changed, 58 insertions(+), 84 deletions(-)
diff --git a/docs/decisions/README.md b/docs/decisions/README.md
index 6fe3bd547..c4bd8befd 100644
--- a/docs/decisions/README.md
+++ b/docs/decisions/README.md
@@ -15,7 +15,7 @@ Durable decisions for NOJV, grouped by area. Each entry records what was decided
- JDG-02 Standard compare is DOMjudge token comparison with two knobs
- JDG-03 DOMjudge validator protocol for checkers and interactors, AC/WA only
- JDG-04 Subtasks score all-or-nothing in every context
-- JDG-05 Run/check separation: untrusted code never sees answers or validators
+- JDG-05 Run/check separation: in official judging untrusted code never sees answers or validators
- JDG-06 One sandbox per stage; per-process accounting via nojv-exec
- JDG-07 Per-language time factor applied once
- JDG-08 Memory ceiling above the problem limit; admission rejections are terminal
@@ -25,7 +25,7 @@ Durable decisions for NOJV, grouped by area. Each entry records what was decided
- JDG-12 Judge queue is Temporal priority and fairness, not a coordinator
- JDG-13 Load-aware judge slots follow node load from /proc
- JDG-14 One canonical toolchain manifest with exact pins
-- JDG-15 Test runs the contestant in the browser; judge programs run only on the server
+- JDG-15 Test runs entirely in the browser, judge programs included
- JDG-16 Advanced Mode is a platform-orchestrated run/grade split
- JDG-17 Advanced network is none or service; answer-bearing containers have no egress
- JDG-18 sandbox-runner depends only on core
@@ -36,7 +36,7 @@ Durable decisions for NOJV, grouped by area. Each entry records what was decided
- JDG-23 Testcase payloads are a content-addressed ConfigMap cache
- JDG-24 The judge worker sweeps orphaned payloads and guards its own memory
- JDG-25 Stage results are read at container exit; cleanup starts at the terminal Pod
-- JDG-26 Test judging runs on its own queue and worker, and web awaits it within a fixed budget
+- JDG-26 Withdrawn: Test judging on its own queue and worker
## [Problems and submissions](problems.md)
@@ -108,7 +108,7 @@ Durable decisions for NOJV, grouped by area. Each entry records what was decided
- SEC-12 Graded testcase data never reaches non-staff
- SEC-13 Rejudge control accepts only rejudge workflows owned by the caller or an admin
- SEC-14 Advanced-mode `/output` capture never dereferences student paths
-- SEC-15 Server-judged Test runs only problem samples, from server-side data
+- SEC-15 Withdrawn: server-judged Test ran only problem samples, from server-side data
## [Web application](web.md)
@@ -190,7 +190,7 @@ Durable decisions for NOJV, grouped by area. Each entry records what was decided
- OPS-18 Renovate is the only dependency update bot
- OPS-19 Single-machine Temporal is one pod per role, reproduced from the repo
- OPS-20 The web image ships production dependencies only
-- OPS-21 The worker image carries the WASM-OJ runtime as stable layers; upgrades are manual
+- OPS-21 Withdrawn: the worker image carried the WASM-OJ runtime as stable layers
## [Engineering practice](engineering.md)
diff --git a/docs/decisions/judge.md b/docs/decisions/judge.md
index aeb40cc01..87df95c86 100644
--- a/docs/decisions/judge.md
+++ b/docs/decisions/judge.md
@@ -27,14 +27,14 @@ The Standard Mode pipeline is implicit and fixed, not a list of configurable sta
### JDG-03 DOMjudge validator protocol for checkers and interactors, AC/WA only
-**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-13-judge-config-simplification-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-13-judge-config-simplification-design.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-06-13-domjudge-alignment](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-06-13-domjudge-alignment.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-13-judge-config-simplification-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-13-judge-config-simplification-design.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-06-13-domjudge-alignment](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-06-13-domjudge-alignment.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
Checkers and interactors are written in `python` or `cpp` only and follow the DOMjudge/Kattis protocol: `validator `, team output on stdin, exit 42 = AC, 43 = WA, anything else = SE; `teammessage.txt` reaches students, `judgemessage.txt` is staff-only. The product owner chose DOMjudge for standards compliance; undocumented argv/exit protocols forced boilerplate.
- Rejected: testlib (bundled or forked; removed entirely); DMOJ-style `process_output`/partial wrapper; `score.txt` partial credit (removed 2026-06); bash/node/C checkers; a presentation-error verdict.
- Rule: verdicts stay AC/WA/TLE/MLE/RE/CE/SE; PE will not be added.
- Rule: protocol changes are hard breaks; existing validators are re-authored, never auto-translated.
-- Rule: the Python wrappers exist twice, as the sandbox runner's assets and in core for the test judge (JDG-26); a unit test keeps the copies byte-identical, so a wrapper change edits both.
+- Rule: the Python wrappers exist twice, as the sandbox runner's assets and in core for browser Test (JDG-15); a unit test keeps the copies byte-identical, so a wrapper change edits both.
- Code: `packages/core/src/judge/validator.ts`, `apps/sandbox-runner/assets/wrappers/`, `packages/core/src/judge/python-judge-wrappers.ts`, `tests/unit/core/judge-program-sources.test.ts`
### JDG-04 Subtasks score all-or-nothing in every context
@@ -48,16 +48,16 @@ A TestcaseSet earns its full weight only if every case is AC, else 0, in practic
- Rule: no subtask early exit; students see every case's result.
- Code: `packages/application/src/submission/scoring.ts`
-### JDG-05 Run/check separation: untrusted code never sees answers or validators
+### JDG-05 Run/check separation: in official judging untrusted code never sees answers or validators
-**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-02-judge-pipeline-spec](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-02-judge-pipeline-spec.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-09-23-judge-single-sandbox-per-stage](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-judge-single-sandbox-per-stage.md), [PR #624](https://github.com/NOJV-TW/NOJV/pull/624), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-02-judge-pipeline-spec](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-02-judge-pipeline-spec.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-09-23-judge-single-sandbox-per-stage](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-judge-single-sandbox-per-stage.md), [PR #624](https://github.com/NOJV-TW/NOJV/pull/624), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
-The container that runs student code never mounts expected answers, validator/interactor source or secret interactor input; checking happens in a separate container that starts only after the run container exits (interactive pairs a solution side with an interactor side that alone holds the secret data). A shared mount namespace once let programs read `expected.txt` and always get AC. Isolation must need no extra privileges and work on Docker and K8s.
+In official judging, the container that runs student code never mounts expected answers, validator/interactor source or secret interactor input; checking happens in a separate container that starts only after the run container exits (interactive pairs a solution side with an interactor side that alone holds the secret data). A shared mount namespace once let programs read `expected.txt` and always get AC. Isolation must need no extra privileges and work on Docker and K8s.
- Rejected: in-container namespaces/unshare (blocked by cap-drop/no-new-privileges; user namespaces unreliable on GKE); privileged supervisors (isolate, nsjail); starting the judge container alongside run; earlier worker-side comparison and separate validator Jobs.
- Rule: no student process may be alive while answers exist in the Pod; output hashes are verified before comparison.
- Rule: unit tests pin the no-leak property of run payloads.
-- Rule: Test never sends judge programs, answers beyond the public samples, or hidden input to clients; the test judge reads sample data on the server and runs judge programs only there (JDG-15, SEC-15).
+- Rule: the separation covers official judging only. Browser Test runs the problem's checker or interactor beside the student's program in the student's own browser, on sample data and the student's own cases (JDG-15); answers and inputs of non-sample testcases never leave the server (SEC-12).
- Code: `apps/sandbox-runner/src/judges/run-stage.ts`, `apps/sandbox-runner/src/judges/judge-stage.ts`, `apps/worker/src/sandbox/kubernetes/job-manifests.ts`
### JDG-06 One sandbox per stage; per-process accounting via nojv-exec
@@ -131,13 +131,12 @@ Sandbox pipeline failures store bounded SE diagnostics for the admin submissions
### JDG-12 Judge queue is Temporal priority and fairness, not a coordinator
-**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-09-21-judge-capacity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-21-judge-capacity.md), [2026-09-22-temporal-native-judge-queue](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-22-temporal-native-judge-queue.md), [PR #597](https://github.com/NOJV-TW/NOJV/pull/597), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-09 · **Source:** [2026-09-21-judge-capacity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-21-judge-capacity.md), [2026-09-22-temporal-native-judge-queue](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-22-temporal-native-judge-queue.md), [PR #597](https://github.com/NOJV-TW/NOJV/pull/597)
Each `JudgeExecution` runs `durableJudgeWorkflow` on the `judge` queue with `priorityKey` (exam 1, contest 2, practice/assignment 3, recovered submission 4, rejudge 5) and `fairnessKey = studentId`; a student has at most one dispatched non-terminal execution and completion dispatches the next. Capacity is judge worker activity slots, with the sandbox ResourceQuota as hard safety net. A 789-execution rejudge collapsed the workflow-based coordinator.
- Rejected: `judgeAdmissionWorkflow` coordinator with FIFO/permit loops and waves (removed with `JudgeAdmission`, `capacityStrategy`); Kueue (quota-based; revisit only for shared multi-node clusters); HPA/KEDA on one node.
-- Rule: self-hosted Temporal needs `matching.enableFairness` and one partition per NOJV queue (`judge`, `judge-state`, `judge-cleanup`, `platform`, `test-judge`).
-- Rule: Test never runs on `judge` or in a stage Job; server-judged Test has its own queue and worker (JDG-26).
+- Rule: self-hosted Temporal needs `matching.enableFairness` and one partition per NOJV queue.
- Rule: stage and bookkeeping activities carry priority; bookkeeping runs on `judge-state` so it never queues behind Jobs; unmapped paths degrade to priority 3.
- Rule: priority derives from origin (`operationId` marks a rejudge, `recoveryEpoch` a recovery), never from `queueClass`, which only orders a student's own executions; recovered live submissions dispatch ahead of bulk rejudges.
- Rule: rollback re-dispatches execution rows via the reconciler; never replay new histories with old worker code.
@@ -169,28 +168,32 @@ With `WORKER_MIN_CONCURRENCY` set, the judge worker's activity slots come from a
- Rule: TypeScript type errors are compile errors.
- Code: `packages/core/src/judge-environment.json`, `scripts/check-doc-drift.mjs`
-### JDG-15 Test runs the contestant in the browser; judge programs run only on the server
+### JDG-15 Test runs entirely in the browser, judge programs included
-**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-08-18-forge-judge-spike-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-08-18-forge-judge-spike-design.md), [2026-08-21-browser-local-run-npm-migration](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-21-browser-local-run-npm-migration.md), [2026-09-08-test-submit-parity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-08-test-submit-parity.md), [2026-09-09-test-reliability](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-09-test-reliability.md), [#639](https://github.com/NOJV-TW/NOJV/pull/639), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-08-18-forge-judge-spike-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-08-18-forge-judge-spike-design.md), [2026-08-21-browser-local-run-npm-migration](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-21-browser-local-run-npm-migration.md), [2026-09-08-test-submit-parity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-08-test-submit-parity.md), [2026-09-09-test-reliability](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-09-test-reliability.md), [#639](https://github.com/NOJV-TW/NOJV/pull/639), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
-Test never creates a submission. The contestant program compiles and runs client-side in all eight languages (a firm requirement to keep compile cost off the server) using pinned `@wasm-oj/browser` and `@wasm-oj/toolchain-*`, same-origin assets verified by digest. Standard problems compare in the browser with the shared comparator. Checker and interactive problems are judged by the problem's own checker or interactor, which runs only on the server, as WASM in the test judge (JDG-26) and only on samples (SEC-15): a checker problem sends the stdout of each sample that exited normally, an interactive problem sends the browser-compiled contestant, which the worker pairs with the interactor. Submit and Advanced stay on the server. Exact stdin bytes are preserved on every path. Until 2026-10 Test failed on checker and interactive problems with a request for a "public Test judge program" that no author could provide. Anything executed client-side is readable in DevTools: a checker for a multiple-answer problem often computes the optimum and leaks the reference algorithm, readable source exposes checker holes that also game official verdicts, an interactor that derives its secret from a PRNG or the case number becomes predictable, and staff-only `judgemessage` text shows. Problems are reused across semesters and forks, so a leak is permanent, and exam page lock does not block DevTools.
+Test never creates a submission and runs entirely in the student's browser. The contestant program compiles and runs client-side in all eight languages (a firm requirement to keep cost off the server) using pinned `@wasm-oj/browser` and `@wasm-oj/toolchain-*`, same-origin assets verified by digest. Standard problems compare with the shared comparator. On checker and interactive problems the editor fetches the problem's checker or interactor source when it opens, builds it in the browser, and Test runs it: the checker on each sample case the program finished normally, the interactor against the program on every case through `interact`. Submit and Advanced stay on the server. Exact stdin bytes are preserved on both paths. #641 judged checker and interactive samples on a server test worker, which ran TA-authored judge programs, and on interactive problems the student's Wasm, with only the WASM-OJ runtime between them and the container's object-storage keys (every problem's hidden testcases) and Redis URL. Test is the student's own run, so the product owner moved all of it to the student's machine. Judge programs become readable by students: bugs such as a missing validity check or an off-by-one query limit get easier to find, but they exist whether or not the source is public, and authors own them.
-- Rejected: shipping checker or interactor source to the browser (above).
-- Rejected: compiling the judge program to WASM and shipping the binary. It raises the bar but is not a boundary: strings survive, the module can be probed as a black-box oracle, and Python judge programs would still ship as source or `.pyc`.
-- Rejected: a per-problem "allow browser Test (publishes source)" toggle (the owner does not want authors deciding exposure); separate public Test judge programs (double the authoring work, and problems without one stay untestable).
+- Rejected: running judge programs on the server for Test (#641, withdrawn before release; JDG-26, SEC-15, OPS-21).
+- Rejected: compiling judge programs on the server and shipping the build; it keeps a server WASM-OJ runtime and a worker for a preview feature.
+- Rejected: execution-only Test on checker and interactive problems; students could not tell whether an answer or an interaction is accepted.
+- Rejected: exposure toggles: a per-problem "allow browser Test (publishes source)" switch (the owner does not want authors deciding exposure) and separate public Test judge programs (double the authoring work, and problems without one stay untestable).
- Rejected: full server-side Test through stage Jobs. Each click would cost about 7 s of a judge slot, and the 2026-10-05 exam stress test showed judging already CPU-bound at about 20 submissions a minute.
-- Rejected: judging client-supplied custom cases with the private checker or interactor; it turns the judge program into an oracle (SEC-15).
- Rejected: splitting an interactive run across the network (contestant in the browser, interactor on the server); every turn would be a round trip and the server would hold state.
-- Rejected: a server fallback that compiles or runs contestant code when the browser cannot (unsupported language, failed toolchain download). Compilation never happens on the server for Test; the server runs a contestant only in interactive Test, already compiled by the browser and paired with the interactor.
+- Rejected: a server fallback that compiles or runs code when the browser cannot (unsupported language, failed toolchain download).
- Rejected: relaxing CSP (`unsafe-eval`, worker exceptions, relaxed-CSP origin); forking or patching Forge in NOJV; host-clock time; appending LF to stdin; browser-executed official Submit; mapping WASM metrics onto CPU/RSS limits before calibration; the monolithic `@wasm-oj/forge` adapter.
-- Rule: generic runtime fixes land upstream and are consumed as pinned releases; keep document CSP at `wasm-unsafe-eval`. Python interactors and JS/TS contestants on interactive problems stay unavailable until an upstream release supports them (runtime-bundle interactors, streaming QuickJS stdin).
-- Rule: Test results are previews on Forge logical time, server-judged ones included; do not claim resource or toolchain-version equivalence; fix sample data, not engine input.
-- Rule: source diagnostics are CE, toolchain/infrastructure faults SE. Custom cases without expected output are execution-only; on checker problems every custom case is execution-only, and interactive problems have no custom cases.
-- Rule: no judge program, `judgemessage`, interactor stderr or judge-program build diagnostic reaches a student; a server-judged case returns only its verdict, the checker's `teammessage`, contestant stderr, the interaction transcript and time.
-- Rule: Test is exempt from the submit cooldown (PRB-22); server-judged Test is limited per user (30 requests a minute, one in flight) and never queues indefinitely.
-- Rule: the editor preloads the selected language's toolchain through `prefetchBrowserToolchain` (Worker URLs, so the HTTP cache serves the build) and Test waits for it; download failures are reported without building and never point students to Submit. A first build otherwise spends its 60 s boundary downloading on slow exam networks and fails the same way on every retry.
-- Rule: add the admitted libc++ PCH only when a C++ source includes `` and ships no own copy; forcing standard headers into other sources would let Test accept code Submit rejects. Judge programs built by the test judge follow the same rule.
-- Code: `apps/web/src/lib/services/browser-local-run.ts`, `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts`, `packages/core/src/judge/test-capability.ts`, `packages/core/src/schemas/test-judge.ts`, `packages/application/src/test-judge/index.ts`, `apps/web/package.json`, `apps/web/svelte.config.js`
+- Rule: students may read the checker or interactor source, sample inputs, outputs and interactor inputs, and `readonly` workspace files; authors own judge programs that hold up when read. A checker only verifies and reads the optimum from `judge_answer` (the testcase's expected output); an interactor reads its secret from `judge_input`.
+- Rule: the server neither compiles nor executes anything for Test. Its only Test role is `/api/problems/[id]/judge-program`, which returns the source to anyone with the problem page's view access for the context.
+- Rule: Test never receives data from non-sample testcases (SEC-12).
+- Rule: generic runtime fixes land in wasm-oj/forge and are consumed as pinned releases; NOJV never forks or patches forge. Keep document CSP at `wasm-unsafe-eval`. JavaScript and TypeScript contestants on interactive problems stay unavailable until upstream streams QuickJS stdin.
+- Rule: browser results are previews on Forge logical time; do not claim resource or toolchain-version equivalence; fix sample data, not engine input.
+- Rule: verdicts map as in official judging through core's `checkerCaseVerdict` and `interactiveCaseVerdict`, the one place for that rule; Test shows `teammessage`, the transcript and contestant stderr, never `judgemessage` or interactor stderr.
+- Rule: source diagnostics are CE, toolchain/infrastructure faults SE. Custom cases without expected output are execution-only; on a checker problem only a case whose input is a sample's input is checked, with that sample's output as the answer, while an interactive custom case supplies the interactor's input and is judged.
+- Rule: Test is exempt from the submit cooldown (PRB-22).
+- Rule: the editor preloads the selected language's toolchain through `prefetchBrowserToolchain` (Worker URLs, so the HTTP cache serves the build) and, on checker and interactive problems, prepares the judge program, and Test waits for both; download failures are reported without building and never point students to Submit. A first build otherwise spends its 60 s boundary downloading on slow exam networks and fails the same way on every retry. A judge program that fails to build disables Test and shows its diagnostics.
+- Rule: every browser engine operation goes through one queue, because the engine runs one foreground compile at a time.
+- Rule: add the admitted libc++ PCH only when a C++ source includes `` and ships no own copy; forcing standard headers into other sources would let Test accept code Submit rejects. Judge programs built in the browser follow the same rule.
+- Code: `apps/web/src/lib/services/browser-local-run.ts`, `apps/web/src/lib/services/judge-program.ts`, `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts`, `packages/application/src/problem/judge-program.ts`, `packages/core/src/judge/test-judge-program.ts`, `packages/core/src/judge/test-judge-verdict.ts`, `apps/web/package.json`, `apps/web/svelte.config.js`
### JDG-16 Advanced Mode is a platform-orchestrated run/grade split
@@ -312,25 +315,8 @@ A standard Kubernetes stage returns once its result is read and saved, freeing i
- Rule: the deferred path runs behind `patched("deferred-stage-cleanup-v1")`; a verdict published before a failed cleanup is not published again on the reconcile path.
- Code: `apps/worker/src/workflows/durable-judge.ts`, `apps/worker/src/activities/judge-execution.ts`, `apps/worker/src/sandbox/kubernetes/job-watch.ts`, `apps/worker/src/sandbox/kubernetes/job-state.ts`, `apps/worker/src/sandbox/kubernetes/standard-executor.ts`, `apps/worker/src/sandbox/kubernetes/interactive-executor.ts`, `apps/worker/src/sandbox/kubernetes/termination.ts`, `apps/worker/src/sandbox/kubernetes/job-manifests.ts`
-### JDG-26 Test judging runs on its own queue and worker, and web awaits it within a fixed budget
-
-**Decided:** 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641)
-
-Checker and interactive Test run on the `test-judge` Temporal queue, served only by `WORKER_MODE=test` workers (`nojv-worker-test`). Web stores the request in object storage, executes `testJudgeWorkflow` and awaits its result under a 30 s execution timeout; it is the only workflow whose result web awaits. The worker holds `TEST_JUDGE_SLOTS` `@wasm-oj/server` engines, one per activity slot because an engine runs one operation at a time, and gives a request 24 s from the moment its activity was scheduled. A request still queued when the workflow or its activity times out answers `test_judge_busy` instead of waiting longer, and so does one whose 24 s run out before every case is judged: running out of time is a capacity problem, and per-case SE told students the judge had failed. Judge programs compile once per input into an object-storage cache; saving a judge configuration starts a best-effort build and a cache miss at Test time builds on demand. Official judging is CPU-bound during exams, so Test must never take a judge slot or a stage Job, and a separate Deployment keeps untrusted contestant WASM out of the judge worker, which holds the sandbox service-account token. A Python checker cost about 1.1 s per case on the 2026-10-06 spike machine, so five samples fit the budget. Each engine builds and runs a Python and a C++ judge program before the worker polls: with two engines per process, a few in 100 cold engines stalled in their first Python runtime preparation or C++ compile until the budget ran out, which was never seen after a warm run; cancelling a stalled warm-up and retrying it moves that cost to startup (40 cold engines on a development machine: one stalled warm-up, recovered by its retry, and no stall afterwards).
-
-- Rejected: Test activities on `judge` or inside the judge worker (they would compete with official judging for slots and run untrusted WASM next to the Kubernetes credentials).
-- Rejected: queueing a request until an engine frees; a student would wait on a spinner exactly when an exam is busiest.
-- Rejected: request bytes in the workflow input (DAT-16; an interactive request carries up to 8 MiB of contestant Wasm).
-- Rejected: a database column for build status; the cached record (artifact or diagnostics) is the status, needs no migration and is keyed so that an upgrade rebuilds by itself.
-- Rejected: upstream `runTrusted`; its profiles exclude Python judge programs.
-- Rule: the cache key is the SHA-256 of `WASM_OJ_SERVER_IDENTITY` and the exact compile input (wrapper, `bits/stdc++.h` shim and source); a record is written only if absent, a failed build is cached like a success, and changing the identity, the wrapper or the shim rebuilds without author action.
-- Rule: the build dispatch runs after the save commits and never fails it; a lost dispatch costs one on-demand build.
-- Rule: on `test-judge` a Test workflow runs at priority 1 and a build workflow at priority 5, each passing its priority to its activity, so builds never delay a waiting Test; a build workflow ends after 15 minutes.
-- Rule: a compile that reaches the engine's own limit is cached as a failed build ("Compilation exceeded the time limit."), so every later Test answers `judge_program_build_failed` at once instead of spending its budget compiling; a compile cancelled at a request's deadline is never cached. The key is content-based, so an author whose program only timed out under load edits the source to rebuild.
-- Rule: an interactive artifact decodes to at most 8 MiB and the request body to 12 MiB, because web holds and parses the body in memory. Release builds of a small stdin program measured 4.8 MB for Java (`java.util.Scanner`), 2.7–2.8 MB for Go, 0.6 MB for C++ with ``, 0.17 MB for Rust, 4 KB for C and a 1.6 KB Python bundle (2026-10-06, `@wasm-oj` 0.2.3 toolchains). Web refuses a larger artifact, or one whose kind is not its language's (a Python runtime bundle, Wasm otherwise), before storing the request.
-- Rule: the worker reads a judge program only through its verified pointer, refuses request keys outside `test-judge-requests/`, deletes the request after reading it and returns at most 1 MiB of JSON.
-- Rule: a case gets at most the remaining budget as its wall stop; a case cut short by the budget makes the whole request `test_judge_busy`, while an interactive contestant stopped at its full wall stop, max(3 s, 3 × the language-factored limit), is TLE, because upstream `interact` does not end a CPU-bound contestant at its logical-time budget.
-- Rule: `nojv-worker-test` mounts no service-account token, has no database credentials, never polls `judge` and stays up through the release window.
-- Rule: a checker gets 512 MiB, the official judge container's floor, since it runs alone; an interactor gets 256 MiB because it runs beside a contestant of up to 1 GiB, and the test worker's 3 GiB limit must hold two such pairs plus Node. Official judging gives an interactor the problem's limit plus 64 MiB, so an interactor that needs more than 256 MiB is SE in Test only.
-- Rule: the test worker requests 100m CPU with a 2-CPU limit; on single-machine the node's CPU request budget bounds judge capacity (JDG-13), so Test borrows idle CPU instead of reserving it. That CPU still counts in the node load the judge worker's slot supplier reads (JDG-13), so heavy Test use can slow judge slot growth.
-- Code: `apps/worker/src/activities/test-judge.ts`, `apps/worker/src/test-judge/`, `apps/worker/src/workflows/test-judge.ts`, `apps/worker/src/worker-app.ts`, `packages/temporal/src/dispatch.ts`, `packages/core/src/judge/test-judge-program.ts`, `infra/charts/nojv/templates/worker-test.deployment.yaml`
+### JDG-26 Withdrawn: Test judging on its own queue and worker
+
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+
+Withdrawn 2026-10 (#TBD): Test no longer runs anything on the server (JDG-15), so there is no Test queue, test worker or judge-program build cache. The separate worker kept Test off the CPU-bound judge queue, but it still ran TA judge programs and student Wasm next to Redis, object-storage and Temporal credentials with only the WASM-OJ runtime in between.
diff --git a/docs/decisions/platform.md b/docs/decisions/platform.md
index 71bdaba77..795da1a78 100644
--- a/docs/decisions/platform.md
+++ b/docs/decisions/platform.md
@@ -199,14 +199,13 @@ Web exposes a public exact-path `/api/release` returning only `{ version, source
### OPS-18 Renovate is the only dependency update bot
-**Decided:** 2026-09, revised 2026-10 · **Source:** [#533](https://github.com/NOJV-TW/NOJV/pull/533), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-09 · **Source:** [#533](https://github.com/NOJV-TW/NOJV/pull/533)
Renovate (`.github/renovate.json`) updates npm packages and pnpm catalog/overrides, GitHub Actions, Dockerfile and Compose images, the digest-pinned images in the chart values, the CloudNativePG operator manifest and the Temporal Helm chart pinned in the runbooks. Dependabot covered only the first four, so the CNPG operator reached its end of support unnoticed.
- Rejected: running Dependabot and Renovate side by side (two bots opening overlapping PRs); a custom version-watch workflow.
- Rule: every version installed outside `package.json` or a Dockerfile is written where a Renovate custom manager reads it (`--version` on Helm installs, a versioned manifest URL, `image: repo:tag@sha256:…` in values); a unit test fails when a manager stops matching.
- Rule: majors, and CNPG or Temporal minors, open only after approval on the dependency dashboard.
-- Rule: `@wasm-oj/*` packages and the `rust` image that builds the WASM-OJ runtime are excluded; they move by hand with a forge upgrade (OPS-21).
- Code: `.github/renovate.json`, `tests/unit/infra/renovate-coverage.test.ts`
### OPS-19 Single-machine Temporal is one pod per role, managed by Flux
@@ -235,13 +234,8 @@ Production pulls images over a ~0.5 MB/s uplink, and the web image was 1.1 GB co
- Rule: a package the server needs at runtime must not arrive only through `optionalDependencies`; the `prod-deps` install skips them.
- Code: `infra/docker/web.Dockerfile`, `apps/web/package.json`
-### OPS-21 The worker image carries the WASM-OJ runtime as stable layers; upgrades are manual
+### OPS-21 Withdrawn: the worker image carried the WASM-OJ runtime as stable layers
-**Decided:** 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
-The test judge (JDG-26) runs from the worker image. Two layers that come before every app layer hold the WASM-OJ native runtime (`wasm-oj-compiler` and `wasm-oj-runner`, built from the forge source at a pinned tag whose commit the build verifies) and the server toolchains (`npm ci` from `infra/docker/wasm-oj-toolchains/` and its lockfile, outside the app's `node_modules`). Both normalise file timestamps, so they stay byte-identical across releases and only a forge upgrade replaces them. Production pulls images over a ~0.5 MB/s uplink (OPS-20) and the two layers are about 82 MB compressed; re-pulling them with every release would add minutes to each rollout.
-
-- Rejected: a separate test-worker image (another image to build, attest, pre-pull and pin on `deploy`); installing the server toolchains in the app's `node_modules` (every release would re-pull them); Renovate bumps (OPS-18), since one upgrade must move the forge tag and commit, the toolchain lockfile, the web and worker `@wasm-oj/*` pins and the server identity together, and each bump replaces the layer.
-- Rule: a forge upgrade bumps `WASM_OJ_FORGE_TAG` and `WASM_OJ_FORGE_COMMIT`, the toolchain lockfile, the `@wasm-oj/*` pins in `apps/web` and `apps/worker` (with their `minimumReleaseAgeExclude` entries) and `WASM_OJ_SERVER_VERSIONS` in one change; `tests/unit/infra/wasm-oj-pins.test.ts` fails when they disagree.
-- Rule: nothing that changes per release goes into the two WASM-OJ stages.
-- Code: `infra/docker/worker.Dockerfile`, `infra/docker/wasm-oj-toolchains/`, `packages/core/src/judge/test-judge-program.ts`, `.github/renovate.json`, `tests/unit/infra/wasm-oj-pins.test.ts`
+Withdrawn 2026-10 (#TBD): with server-side Test gone (JDG-15) the worker image carries no WASM-OJ runtime or server toolchains, and Renovate no longer skips `@wasm-oj/*` or a `rust` builder image. The layers were pinned and timestamp-normalised because production pulls images over a ~0.5 MB/s uplink (OPS-20) and they cost about 82 MB compressed.
diff --git a/docs/decisions/problems.md b/docs/decisions/problems.md
index ffa4a9f10..071e4bafb 100644
--- a/docs/decisions/problems.md
+++ b/docs/decisions/problems.md
@@ -4,17 +4,17 @@ Durable decisions for the problem model, authoring, publication, ownership, and
### PRB-01 Three problem types; workspace files instead of templates
-**Decided:** 2026-04 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [2026-04-12-codebase-cleanup-audit](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-12-codebase-cleanup-audit.md), [2026-05-12-full-source-system-templates-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-12-full-source-system-templates-design.md), [2026-04-01-cp-problem-judge-mapping](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-01-cp-problem-judge-mapping.md), [#628](https://github.com/NOJV-TW/NOJV/pull/628), [#629](https://github.com/NOJV-TW/NOJV/pull/629)
+**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [2026-04-12-codebase-cleanup-audit](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-12-codebase-cleanup-audit.md), [2026-05-12-full-source-system-templates-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-12-full-source-system-templates-design.md), [2026-04-01-cp-problem-judge-mapping](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-01-cp-problem-judge-mapping.md), [#628](https://github.com/NOJV-TW/NOJV/pull/628), [#629](https://github.com/NOJV-TW/NOJV/pull/629), #TBD
-Problem types are `full_source`, `multi_file` and `special_env`. `multi_file` problems use `ProblemWorkspaceFile` (problem, language, path) with whole-file visibility `editable`/`readonly`/`hidden`; the server merges the student's editable files with the rest and judges the whole tree. `full_source` accepts every supported language with system `LANGUAGE_TEMPLATES` starters and no teacher starters. One model covers single-file, fill-in-function, library and multi-file problems without hidden wrapping code.
+Problem types are `full_source`, `multi_file` and `special_env`. `multi_file` problems use `ProblemWorkspaceFile` (problem, language, path) with whole-file visibility `editable`/`readonly`; the server merges the student's editable files with the rest and judges the whole tree. `full_source` accepts every supported language with system `LANGUAGE_TEMPLATES` starters and no teacher starters. One model covers single-file, fill-in-function, library and multi-file problems without hidden wrapping code.
- Rejected: a LeetCode-style `function` type with `driverCode`, insertion markers, `editableRegions` or `assembleSource` templates (use `multi_file` plus a readonly driver); per-file editable regions; letting workspace files decide `full_source` languages; function-mode, custom-script and score-stage judge kits. Nondeterministic or subjective course tasks get deterministic statements or manual grading instead of new judge modes.
- Rule: no insertion markers or driver injection; students submit whole editable files.
- Rule: workspace-file requirements and language filtering apply only when `type === "multi_file"`.
- Rule: a `multi_file` language is allowed exactly when it ships an editable `main.`; there is no allowed-languages column, so the editor derives its ticks from entry files and drops unticked languages' files on save.
- Rule: published problems never change type through any path, and the workspace action refuses `special_env` problems, whose limits and config have their own guarded actions.
-- Rule: `hidden` controls student editor/API presentation for helpers, drivers or an implementation behind an assumed API. The judge still supplies these files to compilation and execution, so student code can inspect them; this provides no runtime confidentiality. Do not put secrets or testcase answers in workspace files.
-- Rejected: treating `hidden` as a runtime-secret guarantee or isolating hidden workspace code from student execution. Files that must run with student code remain part of that execution's trust boundary.
+- Rule: workspace files give no confidentiality: students read every one in the editor or through Test, and student code can read them during judging. Do not put secrets or testcase answers in workspace files.
+- Rejected: a `hidden` visibility (removed 2026-10, #TBD). It gave no confidentiality, since official judging and student code read the file and browser Test would have to ship it; production had no hidden files on 2026-10-07; and authors took it for a secret. Earlier: isolating hidden workspace code from student execution; files that must run with student code stay inside that execution's trust boundary.
- Code: `packages/core/src/types.ts`, `packages/application/src/problem/details.ts`, `packages/core/src/language-templates.ts`
### PRB-02 Judge settings live in one validated `judgeConfig` JSON column
@@ -29,16 +29,16 @@ Eight scattered judge columns became one Zod-validated `Problem.judgeConfig` (ty
### PRB-03 Samples are presentation data, not testcases
-**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
Sample input/output pairs live in `Problem.samples` (JSON); every `TestcaseSet` is a judged subtask with weight ≥ 0. Samples are problem presentation, not grading data. A teacher may add a 0-point set (for example the sample cases) so every submission is judged on it without it adding points; a failing 0-point set still shows in the verdict. Publishing requires the subtask weights to total more than 0, and a problem that is published or used in an activity cannot have its set weights or sets changed so that the total drops to 0; an unused draft may pass through 0 while its subtasks are being built.
- Rejected: samples as a flagged or hidden `TestcaseSet` (`isHidden` was removed). Earlier: every `TestcaseSet` weight > 0 (teachers need judged 0-point sets, 2026-10).
- Rule: do not reintroduce sample flags on `TestcaseSet`; sample-only runs read `Problem.samples`, never 0-point sets.
- Rule: on an interactive problem a sample's `input`/`output` are the two sides of its transcript, and `interactorInput` is the interactor's input file for that sample; saving samples on an interactive problem requires a non-blank `interactorInput` on each, which students see and Test feeds to the interactor (JDG-15). Switching an existing problem to interactive does not check samples; samples without one are left out of Test. `ProblemStatement.interactionFormat` (Markdown) describes the interactor's input and behaviour and is shown only on interactive problems.
-- Rule: server-judged Test on a checker problem uses `sample.output` as the answer file. Authors check that the checker accepts each sample output with the edit page's sample check; a rejected sample shows WA in students' Test.
+- Rule: browser Test on a checker problem uses `sample.output` as the answer file; a sample output the checker rejects shows WA in students' Test, so authors press Test on their own problem (JDG-15).
- Rejected: a separate sample answer field for Test.
-- Code: `packages/db/prisma/schema/problem.prisma`, `packages/core/src/schemas/problem.ts`, `packages/application/src/problem/mutations/records.ts`, `packages/application/src/problem/subtask-points.ts`, `packages/application/src/test-judge/index.ts`
+- Code: `packages/db/prisma/schema/problem.prisma`, `packages/core/src/schemas/problem.ts`, `packages/application/src/problem/mutations/records.ts`, `packages/application/src/problem/subtask-points.ts`, `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts`
### PRB-04 Testcase and workspace content live in object storage behind versioned pointers
@@ -101,13 +101,13 @@ New problems start as `draft` (students may only create private drafts) so autho
### PRB-09 Publication requires a private, current reference solution
-**Decided:** 2026-08 · **Source:** [2026-08-08-reference-solution-validation](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-08-reference-solution-validation.md), [2026-08-15-reference-validation-editor-form](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-15-reference-validation-editor-form.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629)
+**Decided:** 2026-08, revised 2026-10 · **Source:** [2026-08-08-reference-solution-validation](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-08-reference-solution-validation.md), [2026-08-15-reference-validation-editor-form](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-15-reference-validation-editor-form.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), #TBD
A standard problem publishes only with an accepted reference solution for the current judge configuration, authored in the editor's "Reference solution" section. It is an ordinary practice submission flagged `isReferenceSolution` against the full testcase set, pointed to by `Problem.referenceSolutionSubmissionId`, and tied to the problem's storage generation so any judge-affecting edit invalidates it. It validates the testcase and judge contract, not correctness.
- Rejected: a separate route or modal; ZIP upload; auto-generated editorials; review queues or approval states; schema defaults for time/memory limits (required fields).
- Rule: reference source is never public and never appears in lists or history; only the owner, an admin, or course staff with PRB-10 content read access (including staff of an archived course that shares the problem) may read it directly.
-- Rule: only authorized publishers submit with the reference purpose; hidden workspace files are never exposed in the section.
+- Rule: only authorized publishers submit with the reference purpose.
- Rule: invalidation covers testcases, subtask weights, workspace files, judge config/checker/interactor, languages/type, limits and advanced config; subtask descriptions are presentation and do not invalidate.
- Code: `packages/application/src/problem/mutations/publishing.ts`, `packages/db/prisma/schema/submission.prisma`
@@ -244,9 +244,9 @@ Tracking belongs to the authenticated session (SSE wakeups, 5 s visible polling
### PRB-22 Every non-sample submission waits a per-problem cooldown with a platform minimum
-**Decided:** 2026-10 · **Source:** [PR #634](https://github.com/NOJV-TW/NOJV/pull/634), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-10 · **Source:** [PR #634](https://github.com/NOJV-TW/NOJV/pull/634)
-In every context (practice, assignment, exam, contest, virtual) a user's non-sample submission to a problem must come at least `max(activity submitCooldownSec, SUBMIT_COOLDOWN_MIN_SEC)` seconds after their previous one to the same problem in the same context; assignments, practice and virtual contests use the platform minimum alone. One env value is both the platform cooldown and the minimum for exam and contest settings, so teachers and students learn one rule: "wait N seconds before resubmitting this problem". Test never creates a submission and is exempt; server-judged Test has its own per-user limit (JDG-15). The Data Structures exams of 2026-10 had 29–58% of submissions within 60 s of the same student's previous one on the same problem.
+In every context (practice, assignment, exam, contest, virtual) a user's non-sample submission to a problem must come at least `max(activity submitCooldownSec, SUBMIT_COOLDOWN_MIN_SEC)` seconds after their previous one to the same problem in the same context; assignments, practice and virtual contests use the platform minimum alone. One env value is both the platform cooldown and the minimum for exam and contest settings, so teachers and students learn one rule: "wait N seconds before resubmitting this problem". Test never creates a submission and is exempt (JDG-15). The Data Structures exams of 2026-10 had 29–58% of submissions within 60 s of the same student's previous one on the same problem.
- Rejected: a platform floor across all problems (the first 2026-10-05 draft; a second cooldown scope would mean two rules and two settings); capping each student's concurrently queued submissions (students cannot tell why they are blocked); a Redis cooldown key (DAT-10); answering 429 (clients already key on `code: "submit_cooldown"` with 403); relaxing the cooldown near an exam's end.
- Rule: the server enforces the maximum at submit time; form `min` attributes and the settings display only help teachers. A one-time migration raised stored exam and contest settings below 30 to 30.
diff --git a/docs/decisions/security.md b/docs/decisions/security.md
index 06df092d9..b08913b37 100644
--- a/docs/decisions/security.md
+++ b/docs/decisions/security.md
@@ -155,14 +155,8 @@ Run output is copied host-side by `safeCopyTree` (lstat first, drop symlinks and
- Rule: Docker and Kubernetes gates stay behavior-identical (parity test); watchdogs count every filesystem entry as well as regular-file bytes.
- Code: `apps/worker/src/sandbox/docker/advanced-mode-executor.ts`, `apps/worker/src/sandbox/kubernetes/advanced-executor.ts`
-### SEC-15 Server-judged Test runs only problem samples, from server-side data
+### SEC-15 Withdrawn: server-judged Test ran only problem samples, from server-side data
-**Decided:** 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
-A test-judge request names samples by index, each at most once; the server reads every sample's input, answer (`output`) and interactor input from `Problem.samples` and never takes them from the request. A checker request adds the contestant's stdout for each sample, an interactive request the browser-compiled contestant. The checker and interactor are private and the same programs judge official submissions. If students could post their own input/answer pairs or interactor inputs and read back verdicts and `teammessage`, the judge program would be an oracle: they could map the checker's acceptance rule or the interactor's behaviour and exploit it in official judging.
-
-- Rejected: judging client-supplied custom cases with the private checker or interactor; trusting sample data sent by the client.
-- Rule: custom cases are execution-only on checker problems and unavailable on interactive problems; a sample without an interactor input cannot be judged.
-- Rule: requests use the read-only problem-context check shared with code drafts (`assertProblemContextAllowed`): the active-exam context lock and proctoring gate, assignment membership, contest participation inside the contest window (managers exempt), the virtual-contest timer and practice view access. It is looser than Submit (no close or language check) because Test creates no submission.
-- Rule: the response never carries `judgemessage`, interactor stderr, build diagnostics or data from non-sample testcases (SEC-12); judge-program diagnostics are shown only to the problem's editors.
-- Code: `packages/application/src/test-judge/index.ts`, `packages/application/src/code-draft.ts`, `packages/core/src/schemas/test-judge.ts`, `apps/web/src/routes/api/problems/[id]/test-judge/+server.ts`
+Withdrawn 2026-10 (#TBD): Test no longer judges anything on the server (JDG-15). The rule kept private judge programs from becoming an oracle for client-supplied cases; judge programs are now readable by students, so the browser judges a student's interactive cases with the real interactor. Test still never receives non-sample testcase data (SEC-12).
diff --git a/docs/decisions/web.md b/docs/decisions/web.md
index 4b8aa8735..e93250062 100644
--- a/docs/decisions/web.md
+++ b/docs/decisions/web.md
@@ -49,14 +49,14 @@ A `Notification` row is a persistent review-later event behind the navbar bell;
### WEB-05 The server holds the code draft of record, keyed by context, problem and language
-**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-05-11-code-draft-autosave-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-11-code-draft-autosave-design.md), [2026-09-23-server-code-drafts](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-server-code-drafts.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641)
+**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-05-11-code-draft-autosave-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-11-code-draft-autosave-design.md), [2026-09-23-server-code-drafts](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-server-code-drafts.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
`CodeDraft` keyed by (user, contextKey, problem, language) holds unsubmitted code with autosave; the browser keeps only unacknowledged edits in a v2 local cache sealed with a per-user AES-GCM key and deletes them once acknowledged. Contexts (`practice`, `assignment:`, `exam:`, `contest:`, `virtual:`) never share drafts. Students lost code on reload, and plain localStorage lost exam code on shared lab PCs and could leak it to the next user.
- Rejected: server-side encryption at rest (access control is the boundary); staff visibility of drafts; TTL expiry. Earlier: localStorage-only drafts saved on Ctrl+S with no server sync (v1, 2026-05), then local autosave (2026-06) — replaced by server `CodeDraft` rows.
- Rule: draft keys always include the context so drafts never leak across contexts.
- Rule: owner-only access; exam drafts need an active exam session on a published, not-ended exam containing the problem, and an active exam session sees only that exam's drafts.
-- Rule: contest drafts need the contest to be running and the user to participate; contest managers are exempt. Server-judged Test shares this check (SEC-15).
+- Rule: contest drafts need the contest to be running and the user to participate; contest managers are exempt. Test's judge-program source endpoint does not use this check; it grants the problem page's view access for the context (JDG-15).
- Rule: drafts use their own rate limiter, never the submission budget; no pushes after a failed initial load; last write wins.
- Rule: when local storage is full, evict the oldest drafts; no scheduled expiry.
- Code: `packages/application/src/code-draft.ts`, `packages/db/prisma/schema/submission.prisma`, `apps/web/src/routes/api/drafts/+server.ts`, `apps/web/src/lib/services/draft-sync.ts`, `apps/web/src/lib/stores/code-draft.ts`
From a4080a782b8d1ab2b5a68e366eadf952c6afa05f Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:08:18 +0800
Subject: [PATCH 14/22] docs: describe browser-run checker and interactive Test
The living docs drop the server test judge: the test-judge queue, worker,
workflows, route, limiter, Redis lock, storage keys, chart values,
WASM-OJ image layers and their upgrade steps, the local test-judge setup
and its runbook section, and the Test failure mode in Reliability.
The Judge Pipeline's Browser Test section now covers the judge-program
endpoint and its view access, preparation and preload when the editor
opens, the checker's args, files and limits, interactive limits and
custom interactor inputs, the engine queue, verdict mapping, the
transcript cap, and where Test differs from official judging after a
broken pipe. Architecture gets a browser Test flow; Frontend, Security,
the Threat Model and Product Sense describe readable judge programs as an
accepted risk owned by authors, with hidden testcases never leaving the
server. Workspace visibility is editable or readonly everywhere.
The Problem Test spec is rewritten as Given/When/Then for checker and
interactive Test, custom cases, build failures, JavaScript and
TypeScript on interactive problems and special_env. The Quality Ledger
drops the server-Test items, lists the forge release NOJV needs and the
upstream gaps, and adds a Wasm fast path for official judging to
evaluate.
Co-Authored-By: Claude Opus 5.5
---
docs/architecture/ARCHITECTURE.md | 81 ++++----
docs/architecture/DATABASE.md | 6 +-
docs/architecture/FRONTEND.md | 56 ++---
docs/architecture/JUDGE_PIPELINE.md | 298 ++++++++++++---------------
docs/architecture/REDIS.md | 8 -
docs/features/contests.md | 3 +-
docs/features/problem-test.md | 81 ++++----
docs/operations/DEPLOYMENT.md | 111 +++-------
docs/operations/QUALITY_SCORE.md | 10 +-
docs/operations/RELIABILITY.md | 12 +-
docs/operations/SECURITY.md | 49 +++--
docs/operations/THREAT_MODEL.md | 39 ++--
docs/product/PRODUCT_SENSE.md | 4 +-
docs/runbooks/getting-started.md | 25 +--
docs/runbooks/judge-queue.md | 33 +--
docs/runbooks/observability-setup.md | 2 +-
docs/runbooks/testing.md | 1 -
17 files changed, 324 insertions(+), 495 deletions(-)
diff --git a/docs/architecture/ARCHITECTURE.md b/docs/architecture/ARCHITECTURE.md
index d5c103ccd..9504a6805 100644
--- a/docs/architecture/ARCHITECTURE.md
+++ b/docs/architecture/ARCHITECTURE.md
@@ -10,7 +10,7 @@ flows. Judge internals live in [Judge Pipeline](./JUDGE_PIPELINE.md); schema in
- `apps/web/` — SvelteKit BFF; `src/lib/server/domain-orchestration.ts` wires Temporal into the application port
- `apps/worker/src/worker-app.ts` — worker boot, task-queue registration, startup singletons
- `apps/worker/src/workflows/index.ts` — every registered workflow
-- `apps/worker/src/activities/{judge-bundle,platform-bundle,test-judge-bundle}.ts` — activities per queue
+- `apps/worker/src/activities/{judge-bundle,platform-bundle}.ts` — activities per queue
- `apps/worker/src/activities/durable-work-registry.ts` — outbox work kinds and handlers
- `packages/temporal/src/{dispatch,task-queues,orchestration-adapter}.ts` — start/query helpers, queue names, port adapter
- `packages/application/src/shared/orchestration.ts` — `DomainOrchestrationAdapter` port
@@ -84,10 +84,8 @@ against route drift by `tests/unit/openapi-contract.test.ts` (ENG-05).
### Worker modes and task queues
`WORKER_MODE` selects which Temporal workers a process runs. Helm deploys
-`nojv-worker` (`judge`), `nojv-worker-platform` (`platform`) and, when
-`worker.test.enabled`, `nojv-worker-test` (`test`); `all` is for development and
-serves `test-judge` only when `WASM_OJ_RUNTIME_DIR` and `WASM_OJ_TOOLCHAIN_DIR` are
-set.
+`nojv-worker` (`judge`) and `nojv-worker-platform` (`platform`); `all` is for
+development.
| Queue | Served in mode | Worker shape | Work |
| --------------- | ----------------- | -------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- |
@@ -95,18 +93,15 @@ set.
| `judge-state` | `judge`, `all` | Activity-only, 16 fixed slots | Judge bookkeeping activities (state, verdict commit, scoreboard nudge) |
| `judge-cleanup` | `judge`, `all` | Activity-only, 16 fixed slots | Deferred standard stage cleanup (`cleanupJudgeStage`) |
| `platform` | `platform`, `all` | Workflows + activities; `WORKER_CONCURRENCY` slots | Lifecycle timers, plagiarism, durable work, sweeper, score effects, notifications |
-| `test-judge` | `test`, `all` | Workflows + activities; `TEST_JUDGE_SLOTS` slots, one WASM-OJ engine each | Server-judged browser Test and judge-program builds (JDG-26) |
Judge and platform workers cache at most 32 workflows and run at most 8 workflow
-tasks concurrently; the test worker caches 16 and runs 8. Queue capacity and priority: see
+tasks concurrently. Queue capacity and priority: see
[Judge Pipeline](./JUDGE_PIPELINE.md) and JDG-12/13.
On `platform`/`all` startup the worker ensures the three cron singletons, then
runs `sweepStaleSubmissions()` and `recoverSystemErrorSubmissions()` once.
`judge` mode with `EXECUTION_BACKEND=kubernetes` refuses to start unless the
-sandbox runtime and NetworkPolicy probes pass (JDG-20). `test` mode refuses to start
-without `WASM_OJ_RUNTIME_DIR` and `WASM_OJ_TOOLCHAIN_DIR`, and starts its engines
-before it polls; it never creates sandbox Jobs.
+sandbox runtime and NetworkPolicy probes pass (JDG-20).
## Temporal orchestration
@@ -114,20 +109,18 @@ Rules: all async work runs in Temporal (DAT-13); workflow inputs carry IDs, not
blobs (DAT-16); workflow code changes use `patched()` (DAT-15); cron processors
are a cron parent awaiting a continue-as-new child (DAT-19).
-| Workflow | Queue | Workflow ID | Start / notes |
-| -------------------------------------- | ------------ | ----------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
-| `durableJudgeWorkflow` | `judge` | `judge-execution-{executionId}-{recoveryEpoch}` | `dispatchJudgeExecution`; `REJECT_DUPLICATE`; carries priority/fairness keys; state lives in `JudgeExecution` rows |
-| `judgeCleanupWorkflow` | `judge` | `judge-cleanup-{leaseToken}` | `dispatchJudgeCleanup`; `ALLOW_DUPLICATE_FAILED_ONLY` |
-| `contestLifecycleWorkflow` | `platform` | `contest-lifecycle-{contestId}` | `ensure/replace/cancelContestLifecycle`; publishes `contest:starting` / `contest:ending` |
-| `examAutoCloseWorkflow` | `platform` | `exam-auto-close-{examId}` | `ensure/replace/cancelExamAutoClose`; closes active sessions at `endsAt` |
-| `assignmentDueSoonWorkflow` | `platform` | `assignment-due-soon-{assignmentId}` | `ensure/replace/cancelAssignmentDueSoon`; lead-day reminders (DAT-18) |
-| `plagiarismCheckWorkflow` | `platform` | `plagiarism-{targetType}-{targetId}` | `dispatchPlagiarismCheck`; `TERMINATE_EXISTING` on conflict (ASM-23) |
-| `registryGarbageCollectWorkflow` | `platform` | `registry-gc` | `dispatchRegistryGarbageCollect`; singleton, reports `alreadyRunning` (OPS-10) |
-| `submissionSweeperWorkflow` | `platform` | `submission-pending-sweeper` | Cron `* * * * *`; ensured by the platform worker |
-| `durableWorkProcessorWorkflow` | `platform` | `durable-work-processor` | Cron `* * * * *`; runs `durableWorkWorkflow` child |
-| `lifecycleReconcilerProcessorWorkflow` | `platform` | `lifecycle-timer-reconciler` | Cron `*/5 * * * *`; runs `lifecycleReconcilerWorkflow` child; re-ensures timers and missed judge dispatch |
-| `testJudgeWorkflow` | `test-judge` | `test-judge-{uuid}` | `runTestJudgeWorkflow`; web executes it and awaits the result under a 30 s execution timeout; a timeout is `test_judge_busy` |
-| `testJudgeProgramBuildWorkflow` | `test-judge` | `test-judge-build-{role}-{language}-{sha256}` | `dispatchTestJudgeProgramBuild` after a judge-config save, best effort; `USE_EXISTING`, `ALLOW_DUPLICATE` |
+| Workflow | Queue | Workflow ID | Start / notes |
+| -------------------------------------- | ---------- | ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ |
+| `durableJudgeWorkflow` | `judge` | `judge-execution-{executionId}-{recoveryEpoch}` | `dispatchJudgeExecution`; `REJECT_DUPLICATE`; carries priority/fairness keys; state lives in `JudgeExecution` rows |
+| `judgeCleanupWorkflow` | `judge` | `judge-cleanup-{leaseToken}` | `dispatchJudgeCleanup`; `ALLOW_DUPLICATE_FAILED_ONLY` |
+| `contestLifecycleWorkflow` | `platform` | `contest-lifecycle-{contestId}` | `ensure/replace/cancelContestLifecycle`; publishes `contest:starting` / `contest:ending` |
+| `examAutoCloseWorkflow` | `platform` | `exam-auto-close-{examId}` | `ensure/replace/cancelExamAutoClose`; closes active sessions at `endsAt` |
+| `assignmentDueSoonWorkflow` | `platform` | `assignment-due-soon-{assignmentId}` | `ensure/replace/cancelAssignmentDueSoon`; lead-day reminders (DAT-18) |
+| `plagiarismCheckWorkflow` | `platform` | `plagiarism-{targetType}-{targetId}` | `dispatchPlagiarismCheck`; `TERMINATE_EXISTING` on conflict (ASM-23) |
+| `registryGarbageCollectWorkflow` | `platform` | `registry-gc` | `dispatchRegistryGarbageCollect`; singleton, reports `alreadyRunning` (OPS-10) |
+| `submissionSweeperWorkflow` | `platform` | `submission-pending-sweeper` | Cron `* * * * *`; ensured by the platform worker |
+| `durableWorkProcessorWorkflow` | `platform` | `durable-work-processor` | Cron `* * * * *`; runs `durableWorkWorkflow` child |
+| `lifecycleReconcilerProcessorWorkflow` | `platform` | `lifecycle-timer-reconciler` | Cron `*/5 * * * *`; runs `lifecycleReconcilerWorkflow` child; re-ensures timers and missed judge dispatch |
Lifecycle timers (`contest`, `exam`, `assignment`) are reconciled, not blindly
restarted: each start carries `scheduleRevision` and `timerFingerprint` in the
@@ -193,33 +186,31 @@ sequenceDiagram
### Test judging
-Checker and interactive Test (JDG-15, JDG-26): the browser compiles and runs the
-contestant, and the server runs only the judge program, on samples.
+Test runs entirely in the student's browser, judge programs included (JDG-15). On a
+checker or interactive problem the editor fetches the judge program's source when
+it opens; the server only authorises and reads it.
```mermaid
sequenceDiagram
participant Browser
+ participant Engine as WASM-OJ (browser Workers)
participant Web
- participant Redis
participant Postgres
participant Storage
- participant Temporal
- participant TestWorker as worker-test
-
- Browser->>Web: POST /api/problems/{id}/test-judge (sample indices + stdout or contestant Wasm)
- Web->>Redis: rl:test-judge, SET NX in-flight lock
- Web->>Postgres: authorise context; read samples, judgeConfig, judge pointer
- Web->>Storage: put test-judge-requests/{uuid}.json
- Web->>Temporal: execute testJudgeWorkflow (30 s timeout)
- Temporal->>TestWorker: runTestJudge(requestKey)
- TestWorker->>Storage: read and delete request; read judge source; read or write test-judge-programs/v1/{key}.json
- TestWorker-->>Temporal: per-case verdicts (at most 1 MiB)
- Temporal-->>Web: result
- Web->>Storage: delete request
- Web-->>Browser: { cases } or { code }
+
+ Browser->>Web: GET /api/problems/{id}/judge-program?context=…
+ Web->>Postgres: problem view access in that context; judgeConfig; script pointer
+ Web->>Storage: read checker or interactor source (verified pointer)
+ Web-->>Browser: { role, language, source, sha256 }
+ Browser->>Engine: build the judge program (kept per sha256 for the page session)
+ Note over Browser,Engine: student presses Test
+ Browser->>Engine: compile the student's program, run each case
+ Browser->>Engine: checker on each sample that exited normally, or interact(contestant, interactor)
+ Engine-->>Browser: verdicts, teammessage, transcript
```
-Details, limits and error codes: [Judge Pipeline](./JUDGE_PIPELINE.md#browser-test).
+Standard problems skip the fetch. Details, limits and verdict mapping:
+[Judge Pipeline](./JUDGE_PIPELINE.md#browser-test).
### Exam session
@@ -249,10 +240,8 @@ Cache and lease details: [Redis](./REDIS.md); rationale DAT-11.
`@nojv/storage` (S3-compatible: MinIO locally, GCS/R2/S3 in production) holds
submission sources and verdict detail, testcases, workspace files,
checker/interactor programs, judge snapshots and stage results, and images.
-Keys come from `packages/storage/src/keys.ts`, including the test judge's transient
-requests (`test-judge-requests/`) and judge-program build cache
-(`test-judge-programs/v1/`); rows store verified pointers (size + SHA-256).
-Images are served same-origin through
+Keys come only from `packages/storage/src/keys.ts`; rows store verified
+pointers (size + SHA-256). Images are served same-origin through
`/api/storage/{problem-images,user-content-images,avatars}/…`. Env:
`S3_ENDPOINT`, `S3_ACCESS_KEY`, `S3_SECRET_KEY`, `S3_BUCKET`, `S3_REGION`
(`packages/storage/src/env.ts`); deployment values in
diff --git a/docs/architecture/DATABASE.md b/docs/architecture/DATABASE.md
index 8200c0b49..39cb719b6 100644
--- a/docs/architecture/DATABASE.md
+++ b/docs/architecture/DATABASE.md
@@ -151,10 +151,8 @@ erDiagram
`Problem.samples`, not testcases (PRB-03). Samples saved on an interactive
problem each need a non-blank `interactorInput`. `Testcase` and
`ProblemWorkspaceFile` bodies are storage pointers (PRB-04). Workspace
- `visibility` is `editable` / `readonly` / `hidden`; submitted contents cannot
- override readonly or hidden files at merge time. Hidden content is omitted from
- student editor/API reads (metadata can remain), but compilation and execution
- still receive it. Visibility provides no runtime confidentiality; workspace
+ `visibility` is `editable` / `readonly`; submitted contents cannot override
+ readonly files at merge time. Visibility provides no confidentiality; workspace
files must not hold secrets or testcase answers (PRB-01).
- `special_env` problems use `advancedConfig` and `advancedRequiredPaths` and no
testcase rows (PRB-12, PRB-13).
diff --git a/docs/architecture/FRONTEND.md b/docs/architecture/FRONTEND.md
index 3b42306dc..550095be4 100644
--- a/docs/architecture/FRONTEND.md
+++ b/docs/architecture/FRONTEND.md
@@ -10,7 +10,7 @@
- `apps/web/src/lib/server/hooks/` — `request-security.ts` (CSRF, security headers), `route-paths.ts`
- `apps/web/src/lib/server/exam-lock.ts`, `step-up.ts`, `problem-solve.ts`, `domain-orchestration.ts`
- `apps/web/src/lib/server/openapi/` — public / internal OpenAPI documents
-- `apps/web/src/lib/services/` — client submission tracker, draft sync, `fetchWithCsrf`, browser-local run
+- `apps/web/src/lib/services/` — client submission tracker, draft sync, `fetchWithCsrf`, browser-local run, judge-program preparation
- `apps/web/src/lib/stores/` — toast, SSE, notifications, clarifications, theme, code drafts
- `apps/web/src/lib/components/{primitives,features/}/` — domain-agnostic vs domain UI
- `apps/web/svelte.config.js` — CSP, CSRF origin setting; `apps/web/messages/{en,zh-TW}.json` — UI strings
@@ -87,31 +87,31 @@ Solve pages all render `ProblemSolveView` via `loadProblemSolveData` in `lib/ser
The HTTP API reference is the OpenAPI document (`/api/openapi.public.json`, `/api/openapi.internal.json`, rendered at `/docs`); `tests/unit/openapi-contract.test.ts` fails when a route is undocumented or a documented path has no handler (ENG-05). Business rules for each endpoint live in the owning `@nojv/application` domain.
-| Family | Notes |
-| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------ |
-| `/api/auth/[...path]` | better-auth catch-all; auth and sign-in rate limits applied in the hook |
-| `/api/livez` | Process probe; with `/api/readyz` (Postgres + Redis) and `/api/release` bypasses the pipeline (OPS-15) |
-| `/api/admin/healthz` | Admin-only per-subsystem detail |
-| `/api/submissions` | POST create + dispatch; GET history or workspace cursor pages; `status`, `pending`, `[id]`, `[id]/source`, `[id]/rejudge` |
-| `/api/rejudges` | Batch rejudge POST, active rejudges GET; `[workflowId]` progress and `[workflowId]/cancel` |
-| `/api/drafts` | Server code drafts GET/PUT (`draftApiHandler`, WEB-05) |
-| `/api/problems` | List / create; `[id]` delete, bundle, checker, interactor, testcases + ZIP, images, posts, fork, bookmark, storage usage, test-judge |
-| `/api/problems/advanced-scaffold` | Advanced Mode starter templates |
-| `/api/posts/[id]` | Posts, votes, comments, reports; `/api/comments/[id]` delete and reports |
-| `/api/clarifications` | List / create; `[id]` answer, dismiss, delete; `[id]/replies` |
-| `/api/overrides` | Score overrides; `/api/feedback` grading feedback (writes gated post-close) |
-| `/api/plagiarism/[assignmentId]/reports` | Reports and detection trigger; `sources/...` pair sources; `/api/plagiarism-flags` curation |
-| `/api/exams/[examId]/ip-violations` | Proctoring IP log for managers |
-| `/api/contests/[id]/scoreboard` | Built from Postgres on read (DAT-11); `chart` sub-route |
-| `/api/notifications` | List, bulk mark-read / clear; `[id]`; `unread-count` |
-| `/api/events/stream` | SSE per user |
-| `/api/images/proxy` | SSRF-safe third-party Markdown and avatar image proxy (SEC-11) |
-| `/api/uploads/image` | Generic image upload; `/api/account/avatar` avatar PUT/DELETE |
-| `/api/storage/avatars/[userId]/[filename]` | Object-storage reads (also `problem-images`, `user-content-images`) |
-| `/api/admin-mode` | Enter / exit admin mode (may return `verificationRequired`) |
-| `/api/api-token-access` | Whether the token page needs factor setup or step-up |
-| `/api/account/onboarding-tour` | Claim the one-time onboarding tour (UI-22) |
-| `/api/registry/token` | Docker registry token endpoint (Basic credentials, not cookies) |
+| Family | Notes |
+| ------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------- |
+| `/api/auth/[...path]` | better-auth catch-all; auth and sign-in rate limits applied in the hook |
+| `/api/livez` | Process probe; with `/api/readyz` (Postgres + Redis) and `/api/release` bypasses the pipeline (OPS-15) |
+| `/api/admin/healthz` | Admin-only per-subsystem detail |
+| `/api/submissions` | POST create + dispatch; GET history or workspace cursor pages; `status`, `pending`, `[id]`, `[id]/source`, `[id]/rejudge` |
+| `/api/rejudges` | Batch rejudge POST, active rejudges GET; `[workflowId]` progress and `[workflowId]/cancel` |
+| `/api/drafts` | Server code drafts GET/PUT (`draftApiHandler`, WEB-05) |
+| `/api/problems` | List / create; `[id]` delete, bundle, checker, interactor, testcases + ZIP, images, posts, fork, bookmark, storage usage, judge-program |
+| `/api/problems/advanced-scaffold` | Advanced Mode starter templates |
+| `/api/posts/[id]` | Posts, votes, comments, reports; `/api/comments/[id]` delete and reports |
+| `/api/clarifications` | List / create; `[id]` answer, dismiss, delete; `[id]/replies` |
+| `/api/overrides` | Score overrides; `/api/feedback` grading feedback (writes gated post-close) |
+| `/api/plagiarism/[assignmentId]/reports` | Reports and detection trigger; `sources/...` pair sources; `/api/plagiarism-flags` curation |
+| `/api/exams/[examId]/ip-violations` | Proctoring IP log for managers |
+| `/api/contests/[id]/scoreboard` | Built from Postgres on read (DAT-11); `chart` sub-route |
+| `/api/notifications` | List, bulk mark-read / clear; `[id]`; `unread-count` |
+| `/api/events/stream` | SSE per user |
+| `/api/images/proxy` | SSRF-safe third-party Markdown and avatar image proxy (SEC-11) |
+| `/api/uploads/image` | Generic image upload; `/api/account/avatar` avatar PUT/DELETE |
+| `/api/storage/avatars/[userId]/[filename]` | Object-storage reads (also `problem-images`, `user-content-images`) |
+| `/api/admin-mode` | Enter / exit admin mode (may return `verificationRequired`) |
+| `/api/api-token-access` | Whether the token page needs factor setup or step-up |
+| `/api/account/onboarding-tour` | Claim the one-time onboarding tour (UI-22) |
+| `/api/registry/token` | Docker registry token endpoint (Basic credentials, not cookies) |
## Request pipeline
@@ -137,14 +137,14 @@ Security headers, CSP and exam rules are specified in [Security Requirements](..
- Auth: `requireAuth(event)` for pages (redirects), `requireApiAuth(event)` for APIs (throws `HttpError`); `requirePlatformRole(actor, ...roles)`; `getCoursePermissionRole`, `isCourseManager`, `isCourseMember`, `canCreateCourse`. Application errors and security helpers are imported from `@nojv/application` directly.
- Business logic and data access go through `@nojv/application`; `@nojv/db` is imported only for auth wiring (ENG-02).
- Workflow dispatch: routes call application orchestration functions; `lib/server/domain-orchestration.ts` binds them to `@nojv/temporal` at startup. Routes never import raw Temporal helpers.
-- Wrappers: `apiHandler` / `writeApiHandler` / `draftApiHandler` / `registryTokenApiHandler` / `testJudgeApiHandler` rate-limit and map errors for API routes; form actions use `withAction` / `withRateLimit` / `withRateLimitActions` (WEB-02); loaders use `handleLoad` (see [Domain error handling](DESIGN.md#domain-error-handling)).
+- Wrappers: `apiHandler` / `writeApiHandler` / `draftApiHandler` / `registryTokenApiHandler` rate-limit and map errors for API routes; form actions use `withAction` / `withRateLimit` / `withRateLimitActions` (WEB-02); loaders use `handleLoad` (see [Domain error handling](DESIGN.md#domain-error-handling)).
- JSON bodies go through `readJsonBody` / `assertJsonBodyWithinLimit` (1 MiB default).
- Client IP comes only from `getClientIp(event)` (SEC-09).
### Client (`+page.svelte`)
- Editor: Monaco via `features/problem/editors/Editor.svelte` and `primitives/ui/MonacoScriptEditor.svelte`; Advanced Mode uses `AdvancedModeWorkspace.svelte`. No solving workspace below `md` (`MobileWorkspaceBlocker`, UI-13).
-- Browser Test compiles and runs locally through WASM-OJ (`lib/services/browser-local-run.ts`); checker and interactive problems send samples to `/api/problems/[id]/test-judge` (`features/problem/editors/use-editor-run.svelte.ts`, JDG-15).
+- Browser Test compiles and runs locally through WASM-OJ (`lib/services/browser-local-run.ts`, one engine queue for every compile, run and interaction). On checker and interactive problems the editor fetches the judge program from `/api/problems/[id]/judge-program` when it opens and builds it in the browser (`lib/services/judge-program.ts`); Test then runs the checker or interactor there too (`features/problem/editors/use-editor-run.svelte.ts`, JDG-15).
- Forms: `sveltekit-superforms` with `@nojv/core` Zod schemas, created through `appSuperForm` (`lib/utils/super-form.ts`) so a result without a form (rate limit, unexpected error) still shows the form error (WEB-02); errors inline and translated. Plain `fetch` posts to form actions go through `submitFormAction` / `postProblemAction` (`lib/utils/actions.ts`), which `deserialize` the result and throw the server's error unless it is `success`, since action failures arrive as HTTP 200 (PRB-08).
- Problem editor sections save only the fields they own: Basic info never sends judge config or type, sends limits only when it shows them, and sends visibility or admin consent only when the owner changes them (PRB-11); the Judge section refreshes page data after a save or script upload; the Workspace section adopts each saved payload as its persisted state and rebuilds from fresh page data when remounted, and a workspace file upload fills the editor until the next save.
- Markdown: `MarkdownRenderer` → `lib/utils/markdown.ts` (`marked` + KaTeX + DOMPurify); remote HTTPS images rewrite to `/api/images/proxy` at render time.
diff --git a/docs/architecture/JUDGE_PIPELINE.md b/docs/architecture/JUDGE_PIPELINE.md
index b4055d20d..9b11a0d89 100644
--- a/docs/architecture/JUDGE_PIPELINE.md
+++ b/docs/architecture/JUDGE_PIPELINE.md
@@ -10,27 +10,27 @@ fixed **Standard Mode** (`standard` / `checker` / `interactive`, JDG-01) or
## Key code
-| Area | Path |
-| ------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Durable workflow, cleanup workflow | `apps/worker/src/workflows/durable-judge.ts` |
-| Stage / journal activities | `apps/worker/src/activities/judge-execution.ts` |
-| Pinned request, workspace merge, time factor | `apps/worker/src/activities/judge-request.ts`, `judge.ts` (`mergeSandboxSources`) |
-| Worker bootstrap, slots, queues | `apps/worker/src/worker-app.ts`, `apps/worker/src/judge-slot-supplier.ts`, `apps/worker/src/env.ts` |
-| Docker backend | `apps/worker/src/sandbox/docker/` (`args.ts` is the hardened-args builder, JDG-19) |
-| Kubernetes backend | `apps/worker/src/sandbox/kubernetes/` (standard, interactive, advanced executors; manifests; watch; cleanup) |
-| Shared plan, payloads, log parsing, result merge | `apps/worker/src/sandbox/shared/` |
-| In-container runner and phases | `apps/sandbox-runner/src/` (`index.ts`, `judges/`, `payload-materializer.ts`) |
-| Execution helper | `apps/sandbox-runner/native/nojv-exec.c` |
-| DOMjudge Python wrappers | `apps/sandbox-runner/assets/wrappers/`; test-judge copy `packages/core/src/judge/python-judge-wrappers.ts` |
-| Execution creation, state, stages, rejudge | `packages/application/src/submission/judge-execution.ts`, `rejudge-control.ts` |
-| Dispatch gate, reconciliation | `packages/application/src/submission/judge-recovery.ts`, `sweep.ts` |
-| Scoring, adjustments | `packages/application/src/submission/scoring.ts`, `adjustments.ts` |
-| Priority key, stage size, states, limits | `packages/core/src/judge-execution.ts`, `packages/core/src/sandbox.ts` |
-| Comparator, time factor, toolchain manifest | `packages/core/src/judge/compare.ts`, `judge/time-factor.ts`, `judge-environment.json` |
-| Schemas (`judgeConfig`, advanced, output, adjust) | `packages/core/src/schemas/judge-config.ts`, `advanced-mode.ts`, `sandbox-output.ts`, `assessment-adjustments.ts` |
-| Dispatch API, task queues | `packages/temporal/src/dispatch.ts`, `task-queues.ts` |
-| Browser Test | `apps/web/src/lib/services/browser-local-run.ts`, `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts` |
-| Test judge (server) | `packages/application/src/test-judge/index.ts`, `apps/worker/src/activities/test-judge.ts`, `apps/worker/src/test-judge/`, `packages/core/src/schemas/test-judge.ts`, `packages/core/src/judge/test-*.ts` |
+| Area | Path |
+| ------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| Durable workflow, cleanup workflow | `apps/worker/src/workflows/durable-judge.ts` |
+| Stage / journal activities | `apps/worker/src/activities/judge-execution.ts` |
+| Pinned request, workspace merge, time factor | `apps/worker/src/activities/judge-request.ts`, `judge.ts` (`mergeSandboxSources`) |
+| Worker bootstrap, slots, queues | `apps/worker/src/worker-app.ts`, `apps/worker/src/judge-slot-supplier.ts`, `apps/worker/src/env.ts` |
+| Docker backend | `apps/worker/src/sandbox/docker/` (`args.ts` is the hardened-args builder, JDG-19) |
+| Kubernetes backend | `apps/worker/src/sandbox/kubernetes/` (standard, interactive, advanced executors; manifests; watch; cleanup) |
+| Shared plan, payloads, log parsing, result merge | `apps/worker/src/sandbox/shared/` |
+| In-container runner and phases | `apps/sandbox-runner/src/` (`index.ts`, `judges/`, `payload-materializer.ts`) |
+| Execution helper | `apps/sandbox-runner/native/nojv-exec.c` |
+| DOMjudge Python wrappers | `apps/sandbox-runner/assets/wrappers/`; browser Test copy `packages/core/src/judge/python-judge-wrappers.ts` |
+| Execution creation, state, stages, rejudge | `packages/application/src/submission/judge-execution.ts`, `rejudge-control.ts` |
+| Dispatch gate, reconciliation | `packages/application/src/submission/judge-recovery.ts`, `sweep.ts` |
+| Scoring, adjustments | `packages/application/src/submission/scoring.ts`, `adjustments.ts` |
+| Priority key, stage size, states, limits | `packages/core/src/judge-execution.ts`, `packages/core/src/sandbox.ts` |
+| Comparator, time factor, toolchain manifest | `packages/core/src/judge/compare.ts`, `judge/time-factor.ts`, `judge-environment.json` |
+| Schemas (`judgeConfig`, advanced, output, adjust) | `packages/core/src/schemas/judge-config.ts`, `advanced-mode.ts`, `sandbox-output.ts`, `assessment-adjustments.ts` |
+| Dispatch API, task queues | `packages/temporal/src/dispatch.ts`, `task-queues.ts` |
+| Browser Test | `apps/web/src/lib/services/browser-local-run.ts`, `apps/web/src/lib/services/judge-program.ts`, `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts` |
+| Judge program for Test | `packages/application/src/problem/judge-program.ts`, `apps/web/src/routes/api/problems/[id]/judge-program/+server.ts`, `packages/core/src/judge/test-judge-program.ts`, `test-judge-verdict.ts` |
## Durable execution and recovery
@@ -200,8 +200,8 @@ Ordering is Temporal task-queue priority and fairness, not an in-house scheduler
slot is one sandbox Job. Bookkeeping runs on `judge-state` and deferred stage
cleanup on `judge-cleanup` (activity-only workers in the same process, 16 fixed
slots each) so verdicts never queue behind Jobs or teardown.
-- `judge`, `judge-state`, `judge-cleanup`, `platform` and `test-judge` must each use one
- task-queue partition, and fairness needs `matching.enableFairness` (see runbook).
+- `judge`, `judge-state`, `judge-cleanup` and `platform` must each use one task-queue
+ partition, and fairness needs `matching.enableFairness` (see runbook).
- Judge and platform workers cache at most 32 workflows and run at most 8 workflow
tasks concurrently.
@@ -257,7 +257,7 @@ limits or type invalidates it, while subtask description edits do not (PRB-09).
### Workspace merge and payloads
- `mergeSandboxSources()` rebuilds the workspace from `ProblemWorkspaceFile` rows plus
- the student's contents for `editable` files only; `readonly` and `hidden` files
+ the student's contents for `editable` files only; `readonly` files
always win (see [Workspace files](#problem-types-and-workspace-files)).
- Run payload: `testcase-{i}-input.txt` (inputs only) and opaque
`source-file-{n}` keys; `sourceFileMap` restores original paths inside the runner's
@@ -469,20 +469,19 @@ Anything token comparison cannot express needs a checker; no compare modes.
## Browser Test
-**Test** never creates a submission; **Submit** always uses the server pipeline
-above (JDG-15). The browser compiles and runs the contestant program through
-pinned `@wasm-oj/browser`. Standard problems are judged in the browser; checker and
-interactive problems are judged by the [test judge](#test-judge), which runs the
-problem's checker or interactor on the server against its samples. `special_env`
-problems have no Test. Acceptance behaviour: [Problem Test](../features/problem-test.md).
+**Test** never creates a submission and runs entirely in the student's browser
+through pinned `@wasm-oj/browser`, the problem's checker or interactor included
+(JDG-15). The server compiles and executes nothing for Test; it only serves the
+judge program's source. **Submit** always uses the server pipeline above.
+`special_env` problems have no Test. Acceptance behaviour:
+[Problem Test](../features/problem-test.md).
- Uses the problem limits and `judgeConfig.runtime.env` with the
language time factor; the problem time limit sets Forge's logical-time budget.
Instruction or logical-time exhaustion is TLE.
- Shares the comparator and workspace merge rules with the worker; only editable
- files count toward the submission size limit, public teacher files are added
- locally, hidden file contents never reach the browser.
-- Custom cases without an expected answer report execution success only.
+ files count toward the submission size limit, and `readonly` teacher files are
+ added locally.
- Shares the 16 MiB combined output limit.
- Results are previews: WASI toolchains, logical time, linear memory and filesystem
caps differ from native judging; passing samples implies nothing about hidden tests.
@@ -496,139 +495,110 @@ problems have no Test. Acceptance behaviour: [Problem Test](../features/problem-
standard headers.
- Engine failures stay SE; build-boundary timeouts, toolchain download failures and
missing cross-origin isolation add a localized hint above the engine message.
+- Every engine operation (the student's compile and runs, judge-program builds,
+ checker runs and interactions) goes through one module-level FIFO queue on one
+ shared engine, which runs one foreground compile at a time. A caller that aborts
+ while waiting leaves the queue, and aborting cancels the engine only while that
+ caller's own operation runs. A judge-program build nobody waits for any more keeps
+ running, because the page-session memo wants its result.
### Availability
-The problem page's `ProblemDetail.testCapability` (`staticTestCapability`) and the
-selected language decide the Test button before the first click:
-
-| Case | Test |
-| ------------------------------------------------------------------------------ | ---------------------------------------------- |
-| `standard` | Available |
-| `special_env` | Disabled ("doesn't support Test") |
-| Checker or interactive while web has `TEST_JUDGE_ENABLED` off | Disabled ("isn't available") |
-| Python interactor | Disabled until WASM-OJ runs Python interactors |
-| Interactive with a JavaScript or TypeScript contestant | Disabled for that language |
-| Interactive with no sample that has an `interactorInput` | Disabled |
-| A Test response of `judge_program_build_failed` or `judge_program_unsupported` | Disabled for the rest of the editor session |
+| Case | Test |
+| -------------------------------------------------------- | -------------------------------------------- |
+| `standard` | Available |
+| `checker` or `interactive` | Available once the judge program is prepared |
+| `special_env` | Disabled ("doesn't support Test") |
+| Interactive with a JavaScript or TypeScript contestant | Disabled for that language |
+| The judge program failed to build or could not be loaded | Disabled for the rest of the editor session |
+
+### Judge programs
+
+`GET /api/problems/{id}/judge-program?context=…` (`problemDomain.getJudgeProgramSource`,
+`apiHandler` with the standard API limiter, `exam-scoped` in the exam-confinement
+allowlist) returns `{ role, language, source, sha256 }` for a checker or interactive
+problem, read through the verified script pointer.
+
+- Access is the problem page's view access for that context. A context grants its
+ problem while its page would render: course staff and contest organisers always;
+ students in an open assignment, a running published contest they joined, their own
+ running virtual contest, or a running exam session that passes the proctoring gate.
+ Otherwise the practice view rules apply, so the program stays readable after an exam
+ or contest ends to students who can still view the problem. A page-locked exam
+ session keeps the request inside that exam's problems.
+- Hidden testcases never leave the server; Test uses only the problem's samples and
+ the student's own cases.
+
+When the editor opens on a checker or interactive problem and Test is not disabled
+for the selected language, `prepareJudgeProgram`
+(`apps/web/src/lib/services/judge-program.ts`):
+
+1. fetches the source; a 5xx, 429 or network failure is retried after 2 s and 5 s;
+2. preloads the judge program's toolchain (clang for C++, the Python runtime for
+ Python) next to the editor language's;
+3. builds it with core's `judgeProgramCompileInput`: Python gets the DOMjudge wrapper
+ (`python-judge-wrappers.ts`, byte-identical to the sandbox runner's) and is
+ packaged without a compile; C++ gets the platform `bits/stdc++.h` shim and the
+ libc++ PCH header under the rule above;
+4. keeps the built program for the page session per problem, language and `sha256`.
+ Each editor mount refetches the source, so it rebuilds only when the source changed.
+
+While it prepares, Test stays clickable and shows the toolchain download or
+"Preparing checker..." / "Preparing interactor..."; pressing Test waits for it.
+
+| Outcome | Test |
+| ---------------------------------------------------------- | ------------------------------------------------------------------------------------- |
+| Build failure | Disabled ("failed to build"); the Test Result panel opens on the compiler diagnostics |
+| 403 or 404 (no access, or no checker or interactor stored) | Disabled ("Couldn't load this problem's checker/interactor") |
+| Any other response, fetch, toolchain or engine failure | Disabled, with a hint to reload the page |
### Checker problems
-1. The browser runs every case as for a standard problem; TLE, MLE and RE are
- decided there.
-2. Each case whose input is exactly a sample's input and that exited normally posts
- `{ sampleIndex, output }` (stdout cut to 1,000,000 characters) in one request to
- `POST /api/problems/{id}/test-judge`. The server reads that sample's `input` and
- `output` (the answer file) from `Problem.samples` (SEC-15).
-3. The checker runs under the DOMjudge protocol (JDG-03) and returns AC, WA or SE
- plus `teammessage` (up to 10,000 characters).
-4. Every other case, custom or edited, is execution-only. When the request returns
- `test_judge_busy` or `test_judge_unavailable`, the samples are shown
- execution-only with a notice.
+1. The browser compiles and runs every selected case as for a standard problem; TLE,
+ MLE and RE are decided there.
+2. Each case that exited normally and whose input is exactly a sample's `input` is
+ checked under the DOMjudge protocol (JDG-03): args
+ `/judge/input /judge/answer /judge/feedback`, `/judge/input` holding the sample's
+ `input`, `/judge/answer` its `output`, an empty `/judge/feedback/`, and the
+ student's stdout as stdin.
+3. `checkerCaseVerdict` maps exit 42 to AC and 43 to WA; any other exit, or a checker
+ stopped by a limit, is SE. `/judge/feedback/teammessage.txt` is shown, trimmed and
+ cut to 10,000 characters.
+4. Every other case, custom or edited, is execution-only: a normal exit shows
+ "Executed".
### Interactive problems
-1. The browser compiles the contestant and posts the serialised build artifact
- (at most 8 MiB decoded) with the indices of the samples that have an
- `interactorInput`. The case panel shows those inputs read-only; there are no
- custom cases. JavaScript and TypeScript cannot run interactively.
-2. The worker runs WASM-OJ `interact(contestant, interactor)` once per sample. The
- interactor gets the server's copy of the sample's `interactorInput` as its input
- file and an empty answer file.
-3. Verdicts merge as in official judging: an interactor failure is SE, otherwise a
- contestant TLE, MLE or RE wins, otherwise the interactor's AC or WA stands.
-4. Each case returns the verdict, contestant stderr (100,000 bytes), the
- transcript in both directions (64 KiB each) and logical time; the interactor's
- `teammessage` is not returned.
-
-### Test judge
-
-Web side (`testJudgeDomain.runTestJudge`):
-
-- The route takes the per-user in-flight lock and `rl:test-judge` limit
- ([Redis](REDIS.md#rate-limiting)), caps the body at 12 MiB, then authorises the
- request with `assertProblemContextAllowed`, the read-only context check shared
- with code drafts (SEC-15). The request kind must match `judgeConfig.type`.
-- An interactive artifact must be built for the request's language, as Wasm for
- C, C++, Go, Java and Rust or as a runtime bundle for Python, and decode to at
- most 8 MiB (`TEST_JUDGE_MAX_ARTIFACT_BYTES`), measured from its base64 length
- before anything is stored.
-- It builds the stored request from server-side sample data and the judge program's
- storage pointer, writes it to `test-judge-requests/.json`, executes
- `testJudgeWorkflow` (`test-judge-`, 30 s execution timeout) on `test-judge`
- and deletes the request afterwards.
-
-Worker side (`nojv-worker-test`, `WORKER_MODE=test`, JDG-26):
-
-- At startup every engine builds and runs a Python and a C++ judge program before
- the worker polls `test-judge`. An attempt still running after 20 s is cancelled
- and retried; a third failure fails startup, so the pod restarts. A startup probe
- allows the pod two minutes for this.
-- `runTestJudge` runs once (no retry, 28 s schedule-to-close). Its budget is 24 s
- from when the activity was scheduled; it leases one of `TEST_JUDGE_SLOTS` WASM-OJ
- engines, loads or builds the judge program, then judges the cases in order. When
- the budget runs out or the activity is cancelled before every case is judged,
- including a case cut short by a wall stop the budget shortened, the whole request
- answers `test_judge_busy`; an engine error on a case stays SE for that case.
-- Per-case limits:
-
- | Limit | Checker | Interactive contestant | Interactor |
- | ------------ | ----------------------------- | ------------------------------------------------------ | --------------------------- |
- | Logical time | `max(30 s, limit)` | Limit × language factor | `max(30 s, factored limit)` |
- | Memory | 512 MiB | Problem limit | 256 MiB |
- | Wall stop | `min(10 s, remaining budget)` | `min(max(3 s, 3 × factored limit), remaining budget)` | Same as the contestant |
- | Other | — | 16 MiB output, 64 MiB and 4,096 entries of file writes | — |
-
-- A checker that does not exit normally is SE. A contestant stopped at its full wall
- stop is TLE (upstream `interact` does not end a CPU-bound contestant at its
- logical-time budget); one stopped earlier by the request budget makes the request
- busy.
-- The response is at most 1 MiB of JSON; when the texts exceed it each is cut to a
- fair share. It never contains `judgemessage`, interactor stderr, build diagnostics
- or the judge program.
-
-Judge program builds:
-
-- `getJudgeProgram` compiles with `@wasm-oj/server`. Python gets the DOMjudge
- wrapper (`python-judge-wrappers.ts`, byte-identical to the sandbox runner's);
- C++ gets the platform `bits/stdc++.h` shim and, only when the source includes
- ``, the libc++ PCH header, as in the browser.
-- The result, an artifact or at most 4 KiB of diagnostics, is stored at
- `test-judge-programs/v1/.json`, written only if absent. The key is the SHA-256
- of `WASM_OJ_SERVER_IDENTITY` and the exact compile input, so a WASM-OJ upgrade or a
- wrapper or shim change rebuilds without author action.
-- A compile that reaches the engine's own limit (60 s for C++, 120 s for Python) is
- stored as a failed build with the diagnostic "Compilation exceeded the time
- limit.", so later requests answer `judge_program_build_failed` at once; a compile
- cancelled because a request ran out of budget is not stored.
-- Saving a checker or interactor configuration that Test supports dispatches
- `testJudgeProgramBuildWorkflow` after the commit, best effort; a miss at Test time
- builds on demand. Builds run at priority 5 and `testJudgeWorkflow` at priority 1;
- each workflow passes its priority to its activity, so a queued Test request takes
- the next free engine before queued builds. A build workflow has a 15-minute
- execution timeout and its activity 5 minutes per attempt, three attempts.
-- The edit page shows authors the cached status (ready, failed with diagnostics,
- pending, or unavailable when the status cannot be read; only a pending status
- re-checks, every 3 s for a minute through
- `invalidate("problem:judge-program-status")`, skipping a check during
- navigation) and, for checker problems, a sample check that runs the checker with
- each sample's `output` as both the answer and the team output.
-
-Errors (`{ code, message }`):
-
-| Code | HTTP | Cause |
-| ---------------------------- | ---- | ----------------------------------------------------------------------------------------------------------------- |
-| `test_judge_busy` | 429 | The user already has a request in flight |
-| `test_judge_busy` | 503 | The workflow or its activity timed out, or the 24 s budget ran out before every case was judged |
-| `test_judge_unavailable` | 503 | `TEST_JUDGE_ENABLED` off, Redis, storage or Temporal failure, unreadable request, missing or corrupt judge source |
-| `judge_program_build_failed` | 409 | The judge program does not build under WASM-OJ |
-| `judge_program_unsupported` | 409 | Python interactor, JavaScript or TypeScript interactive contestant, or no judge program stored |
-| `test_rejected` | 4xx | Validation or authorisation failure, unknown sample, sample without an interactor input |
-| `test_rejected` | 413 | A contestant artifact over 8 MiB |
-
-A body over 12 MiB is 413 before parsing. An exhausted `rl:test-judge` budget is a
-429 without a code, which the editor reports as busy; any other response without a
-known code is reported as a failed run.
+1. The cases are the samples' `interactorInput`s (a sample without one is left out)
+ and the student's own cases, whose input is the interactor's input. JavaScript and
+ TypeScript contestants cannot run interactively.
+2. The browser compiles the contestant, then runs WASM-OJ
+ `interact(contestant, interactor)` once per case. The interactor gets the checker's
+ args, the case's input as `/judge/input`, an empty `/judge/answer` and an empty
+ `/judge/feedback/`.
+3. `interactiveCaseVerdict` merges as official judging does: an interactor that exits
+ with anything but 42 or 43, or is stopped by a limit, is SE; otherwise a contestant
+ TLE, MLE or RE wins; otherwise the interactor's AC or WA stands.
+4. Each case shows the verdict, contestant stderr (100,000 bytes), both transcript
+ directions (64 KiB each, cut at a UTF-8 boundary) and the contestant's logical
+ time. The interactor's `teammessage` and stderr are not shown.
+5. The engine closes the pipes of a side that stops, while official judging keeps
+ reading the interactor's output after the contestant ends and only gives it EOF. A
+ Python interactor that writes after the contestant stopped at a limit therefore
+ exits 120, and the case is SE in Test where Submit gives TLE or MLE; so is a case
+ where both sides reach the shared wall stop.
+
+### Limits
+
+| Limit | Checker | Interactive contestant | Interactor |
+| ---------------- | ------------------------------------------------------ | ------------------------------ | ----------------------------------------- |
+| Logical time | `max(30 s, limit)` | Limit × language factor | `max(30 s, factored limit)` |
+| Memory | 512 MiB | Problem limit | Problem limit + 64 MiB, at most 1,536 MiB |
+| Wall stop | 2 × logical time | `max(3 s, 3 × factored limit)` | Same as the contestant |
+| Output and files | 16 MiB output; 64 MiB and 4,096 entries of file writes | Same as the checker | Same as the checker |
+
+A system error on a case the browser judged explains that the judge, not the
+student's program, failed.
## Advanced Mode pipeline
@@ -848,15 +818,11 @@ which submitted contents `mergeSandboxSources()` accepts:
| ---------- | ----------- | ------------- | ---------- |
| `editable` | yes | yes | yes |
| `readonly` | greyed out | no | yes |
-| `hidden` | no | no | yes |
-
-Hidden files support helpers, drivers or an implementation behind an API the
-student is expected to use without an editor view. Student-facing problem/API
-reads omit their content while retaining metadata such as path and description.
-Official compilation and execution receive the complete merged workspace,
-including hidden files; student code can read or inspect that workspace. Hidden
-visibility provides no runtime confidentiality. Authors must not store secrets
-or testcase answers in any workspace file (PRB-01).
+
+Workspace files give no confidentiality: students read `readonly` files in the
+editor, browser Test runs them, and official compilation and execution receive the
+complete merged workspace, which student code can read. Authors must not store
+secrets or testcase answers in any workspace file (PRB-01).
## Adjustment rules
diff --git a/docs/architecture/REDIS.md b/docs/architecture/REDIS.md
index 999950cdb..967319884 100644
--- a/docs/architecture/REDIS.md
+++ b/docs/architecture/REDIS.md
@@ -15,7 +15,6 @@ All keys and channels come from one registry (DAT-09).
- `apps/web/src/lib/server/shared/{sse-hub,sse-response,sse-slot}.ts` — SSE plumbing
- `packages/application/src/contest/scoring.ts` — scoreboard cache and lease
- `packages/application/src/api-token/{step-up,security-settings}.ts` — security proof keys
-- `packages/application/src/test-judge/index.ts` — per-user test-judge in-flight lock
## Keys
@@ -30,7 +29,6 @@ All keyed state uses the `nojv:` prefix except rate-limiter keys (`rl:*`).
| `nojv:sb-chart-cache:{contestId}:{live\|public}:{topN}` | 10 s | Scoreboard chart cache |
| `nojv:sb-lock:{contestId}:{live\|public}` | 5 s (`SET NX`, token) | Scoreboard rebuild lease |
| `nojv:sb-throttle:{contestId}` | 10 s (`SET NX`) | Throttle for `scoreboard:update` publishes |
-| `nojv:test-judge:in-flight:{userId}` | 90 s (`SET NX`, token) | One in-flight test-judge request per user (see Rate limiting) |
| `nojv:apitoken:stepup:{sessionId}` | 600 s | API-token step-up proof, bound to `securityGeneration` |
| `nojv:apitoken:page-mfa:{sessionId}` | 3600 s | API-token page MFA proof |
| `nojv:stepup:handoff:{ticket}` | 60 s, `GETDEL` | One-shot step-up handoff ticket |
@@ -108,7 +106,6 @@ for 10 s on top of the scoreboard result and takes no lease.
| `writeApiRateLimiter` | `rl:write` | 10 / 60 s | `u:{userId}` or client IP | Write API handlers |
| `draftApiRateLimiter` | `rl:draft` | 60 / 60 s | `u:{userId}` or client IP | `/api/drafts` autosave |
| `registryTokenRateLimiter` | `rl:registry-token` | 60 / 60 s | `u:{userId}` or client IP | Registry token API handler |
-| `testJudgeApiRateLimiter` | `rl:test-judge` | 30 / 60 s | `u:{userId}` or client IP | `/api/problems/{id}/test-judge` |
| `formActionRateLimiter` | `rl:form` | 20 / 60 s | client IP | `withRateLimit` form actions |
| `authRateLimiter` | `rl:auth` | 60 / 60 s | client IP | All Auth API routes (`hooks.server.ts`) |
| `signInRateLimiter` | `rl:signin` | 5 / 15 min | client IP (`registry:{ip}` for registry) | Admin password sign-in, registry token |
@@ -125,11 +122,6 @@ for 10 s on top of the scoreboard result and takes no lease.
Redis errors return `unavailable` (HTTP 503) except `apiRateLimiter`, which
falls back to an in-process limiter (DAT-12).
- Development (`$app/environment` `dev`): in-memory limiters with 1000× points.
-- `/api/problems/{id}/test-judge` also holds `nojv:test-judge:in-flight:{userId}`
- from before the body is read until the response. A concurrent request from the
- same user gets 429 `{ code: "test_judge_busy" }`; a Redis error gets 503
- `test_judge_unavailable`. Release is a Lua compare-and-delete on the token;
- the TTL only frees the lock when the web process dies mid-request.
Submit cooldowns are not in Redis: `enforceSubmitCooldown`
(`packages/application/src/shared/submit-cooldown.ts`) checks the latest
diff --git a/docs/features/contests.md b/docs/features/contests.md
index a36bb664f..b35e06596 100644
--- a/docs/features/contests.md
+++ b/docs/features/contests.md
@@ -64,7 +64,8 @@ Out of scope: proctoring, course membership gating, score overrides and feedback
- `ensureContestParticipation` rejects before `startsAt` (`"Contest has not started yet."`) and at or after `endsAt` (`"Contest has ended."`). A non-manager without a participation row gets `"You must join the contest before submitting."`; managers and admins are exempt and auto-joined.
- On submit the participation is upserted to `active` with a composite-key upsert, so concurrent first submits converge on one row.
-- Contest code drafts and server-judged Test need a participation row and a running contest; managers and admins are exempt (WEB-05, SEC-15).
+- Contest code drafts need a participation row and a running contest; managers and admins are exempt (WEB-05).
+- Test loads a contest problem's checker or interactor for participants while the published contest runs and for managers while it is published; outside that the problem's practice view rules decide (JDG-15).
- The contest problem route redirects non-managers without participation, and non-managers before start, to `/contests/[id]`; after `endsAt` it redirects to `/problems/[problemId]`.
- A non-sample submission must come at least `max(submitCooldownSec, SUBMIT_COOLDOWN_MIN_SEC)` seconds after the user's previous contest submission to the same problem; otherwise `403 submit_cooldown` with `retryAfterSec` (PRB-22). Rejected requests leave no submission and no penalty.
diff --git a/docs/features/problem-test.md b/docs/features/problem-test.md
index 20976973b..bdf9368d9 100644
--- a/docs/features/problem-test.md
+++ b/docs/features/problem-test.md
@@ -1,72 +1,67 @@
# Feature: Problem Test
-Acceptance spec for the solving workspace's **Test** button: running the student's code on samples and their own cases without creating a submission. The browser compiles and runs the code; checker and interactive problems are judged by the problem's private checker or interactor on the server, on samples only. Also covers the author side: interactor inputs, interaction notes, the judge-program build status and the sample check. Mechanics and limits are in [Judge Pipeline](../architecture/JUDGE_PIPELINE.md#browser-test). Decisions: JDG-15, JDG-26, SEC-15, PRB-03.
+Acceptance spec for the solving workspace's **Test** button: running the student's code on samples and their own cases without creating a submission. Everything runs in the student's browser, including the problem's checker or interactor, whose source the browser fetches when the editor opens. Also covers the author side of interactive problems: interactor inputs and interaction notes. Mechanics and limits are in [Judge Pipeline](../architecture/JUDGE_PIPELINE.md#browser-test). Decisions: JDG-15, PRB-03.
## Key code
- `apps/web/src/lib/components/features/problem/editors/use-editor-run.svelte.ts` — Test flow per judge type, error messages
-- `apps/web/src/lib/components/features/problem/editors/Editor.svelte`, `EditorActionBar.svelte`, `EditorBottomPanel.svelte` — button state, case panel, transcript
-- `apps/web/src/lib/services/browser-local-run.ts` — browser compile and run; `apps/web/src/lib/services/submission-service.ts` — `requestTestJudge`
-- `apps/web/src/routes/api/problems/[id]/test-judge/+server.ts`
-- `packages/application/src/test-judge/index.ts` — `runTestJudge`, `checkSamplesWithChecker`, `getJudgeProgramStatus`, `withUserTestJudgeLock`
-- `packages/core/src/judge/test-capability.ts`, `packages/core/src/schemas/test-judge.ts`
-- Authoring: `apps/web/src/lib/components/features/problem/statement/SamplesEditor.svelte`, `tabs/BasicInfoTab.svelte`, `tabs/judge/JudgeProgramTestStatus.svelte`; edit page actions in `apps/web/src/routes/(app)/problems/[problemId]/edit/+page.server.ts`
-- Tests: `tests/unit/web/editor-client-test.test.ts`, `tests/component/web/editor-test-button-state.test.ts`, `editor-output-comparison.test.ts`, `judge-program-test-status.test.ts`, `samples-editor.test.ts`; `tests/integration/application/test-judge-domain.test.ts`, `test-judge-author.test.ts`; `tests/integration/http/test-judge.test.ts`; `tests/integration/judge/test-judge-runtime.test.ts`; `tests/unit/worker/test-judge-activity.test.ts`
+- `apps/web/src/lib/components/features/problem/editors/Editor.svelte`, `EditorActionBar.svelte`, `EditorBottomPanel.svelte` — preparation and button state, case panel, results and transcript
+- `apps/web/src/lib/services/browser-local-run.ts` — engine queue, compile, runs, `runBrowserChecker`, `runBrowserInteraction`
+- `apps/web/src/lib/services/judge-program.ts` — `prepareJudgeProgram`
+- `apps/web/src/routes/api/problems/[id]/judge-program/+server.ts`, `packages/application/src/problem/judge-program.ts` — `getJudgeProgramSource`
+- `packages/core/src/judge/test-judge-program.ts`, `test-judge-verdict.ts`, `python-judge-wrappers.ts`, `test-capability.ts`
+- Authoring and display: `apps/web/src/lib/components/features/problem/statement/SamplesEditor.svelte`, `tabs/BasicInfoTab.svelte`, `left-panel/ProblemDescriptionPanel.svelte`
+- Tests: `tests/unit/web/editor-client-test.test.ts`, `browser-local-execution.test.ts`, `browser-engine-queue.test.ts`, `judge-program.test.ts`, `tests/unit/core/test-judge-verdict.test.ts`; `tests/component/web/editor-test-button-state.test.ts`, `editor-judge-program.test.ts`, `editor-output-comparison.test.ts`, `samples-editor.test.ts`; `tests/integration/application/judge-program-source.test.ts`, `tests/integration/http/judge-program.test.ts`
-Out of scope: Test for `special_env` problems, official verdicts from Test, Test results in submission history, author-chosen exposure of judge programs, Python interactors and JavaScript/TypeScript interactive contestants (pending upstream WASM-OJ).
+Out of scope: Test for `special_env` problems, official verdicts from Test, Test results in submission history, hiding judge programs from students, JavaScript and TypeScript contestants on interactive problems (pending upstream WASM-OJ).
## API
-| Endpoint | Purpose |
-| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `POST /api/problems/[id]/test-judge` | Judge samples on the server. Body `{ kind: "checker", context, cases: [{ sampleIndex, output }] }` or `{ kind: "interactive", context, language, artifact, cases: [{ sampleIndex }] }`; returns `{ cases: [...] }` |
-| Edit page action `?/checkSamples` | Run the checker on every saved sample with its `output` as both answer and team output; problem editors only |
+| Endpoint | Purpose |
+| ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ |
+| `GET /api/problems/[id]/judge-program?context=` | The problem's checker or interactor as `{ role, language, source, sha256 }`; 404 for standard and `special_env` problems |
-The endpoint requires a session, at most 12 MiB of body, 1 to 5 distinct sample indices, a `kind` equal to the problem's judge type and, for an interactive request, an artifact built for its `language` that decodes to at most 8 MiB. Error codes are listed in [Judge Pipeline](../architecture/JUDGE_PIPELINE.md#test-judge).
+The endpoint requires a session and the problem page's view access for the context; it uses the standard API rate limit.
## Acceptance criteria
### Common
-- Test never creates a submission, consumes no attempt and is not subject to the submit cooldown.
-- The Test button state is known when the page loads. When Test is unavailable the button is disabled and the reason is shown next to it:
- - `special_env`: "This problem type doesn't support Test. Use Submit to have your code judged."
- - Checker or interactive problem while the test judge is off, or a Python interactor: "Test isn't available for this problem right now. Submit still judges your code normally."
- - Interactive problem with JavaScript or TypeScript selected: "Test can't run interactive problems in JavaScript or TypeScript yet…"; switching language re-enables it.
- - Interactive problem with no sample that has an interactor input: "This problem has no sample with an interactor input…"
-- The server answers a request only for a context the user may use: practice view access, assignment membership, an active exam session and the exam gate for that exam, contest participation while the contest runs (managers and admins exempt), or a running virtual contest. Otherwise the request is rejected and nothing is judged.
-- If the checker or interactor cannot be built for Test, the request fails with "This problem's checker or interactor can't run in Test. Submit still judges your code normally." and Test stays disabled for the rest of the editor session.
-- A second Test from the same user while one is in flight gets "Test is busy right now. Try again in a moment." A 31st request within a minute gets the same message, and so does a request the server could not finish judging within its time budget; no sample is reported as a judge failure for running out of time.
-- Results are previews: a server-judged AC does not imply an AC on Submit.
+- Given any problem, when the student presses Test, no submission is created, no attempt is consumed, the submit cooldown does not apply, and Test sends neither the code nor its output to the server.
+- Given a `special_env` problem, Test is disabled with "This problem type doesn't support Test. Use Submit to have your code judged."
+- Given a case that exits normally with nothing to compare against, it shows "Executed" ("This case ran without being judged. Check its output yourself.") instead of AC; a case that hits TLE, MLE or RE shows that verdict on every judge type.
+- Given any Test result, it is a preview: passing samples does not imply an AC on Submit.
### Standard problems
-- Samples and custom cases run in the browser; cases with an expected output get AC/WA from the shared comparator, cases without one show "Executed".
+- Given a standard problem, the case panel starts with each sample's input and expected output; cases with an expected output get AC or WA from the shared comparator, and cases without one show "Executed".
-### Checker problems
+### Judge program preparation
-- Given a checker problem with samples, when the student presses Test, every case runs in the browser and each case whose input is exactly a sample's input and that exited normally is judged by the problem's checker on the server. Those cases show AC or WA with the checker's `teammessage` as "Feedback", and the result is marked "Judged on the server".
-- The answer the checker sees is the sample's stored output on the server; editing the expected output in the case panel changes nothing.
-- A case whose input is not exactly a sample's input, or that hit TLE, MLE or RE in the browser, is never sent to the checker; a normally exiting one shows "Executed" ("This case ran without being judged. Check its output yourself."). The case panel says "This problem's checker judges the samples only…".
-- When the test judge is busy or unavailable, the browser results still show, with samples as "Executed" and the busy or unavailable message as a notice.
-- A checker that crashes, times out or exits with a code other than 42 or 43 is SE, shown as "The judge failed on this case. This is a problem on the judge's side, not with your program."
+- Given a checker or interactive problem, when the editor opens, the browser fetches the problem's checker or interactor and builds it while the student works; the Test button shows the toolchain download or "Preparing checker..." / "Preparing interactor...".
+- Given the judge program is still preparing, when the student presses Test, Test waits for it and then runs.
+- Given the same problem is reopened in the page session with an unchanged judge program, it is not rebuilt; given its source changed, it is rebuilt.
+- Given the judge program fails to build, Test is disabled with "This problem's checker failed to build." (or "interactor"), and the Test Result panel opens on the compiler output.
+- Given the source request is refused (403 or 404), Test is disabled with "Couldn't load this problem's checker." (or "interactor"); given it fails for any other reason (a server or network failure is retried twice first), the message adds "Reload the page to try again."
+- Given a student who may view the problem in the context (including after an exam or contest ends, while the problem is still viewable), the source loads; given an active page-locked exam session, only that exam's problems load.
-### Interactive problems
+### Checker problems
-- The case panel lists each sample that has an interactor input, read-only, and shows the problem's interaction notes; students cannot add cases.
-- When the student presses Test, the browser compiles the program, and the server runs it against the interactor once per listed sample, with that sample's interactor input as the interactor's input file and an empty answer file.
-- Each case shows its verdict, the student's stderr, the time, and a transcript with "From the interactor" and "From your program".
-- The interactor's failure is SE; otherwise the student's TLE, MLE or RE wins; otherwise the interactor's AC or WA. A student program still running at its wall stop (max(3 s, 3 × the language-factored time limit)) is TLE.
+- Given a checker problem, when the student presses Test, every case runs in the browser, and each case that exited normally and whose input is exactly a sample's input is judged by the problem's checker with that sample's stored output as the answer. Those cases show AC or WA, with the checker's `teammessage` as "Feedback".
+- Given the student edits a sample case's input or adds a case, that case is not judged; it shows "Executed" when it exits normally. The case panel says "This problem's checker judges the samples only…".
+- Given the checker crashes, runs out of time or memory, or exits with a code other than 42 or 43, the case is SE with "The judge failed on this case. This is a problem on the judge's side, not with your program."
-### Problem page
+### Interactive problems
-- On an interactive problem the statement shows the "Interaction" notes, and each sample shows its interactor input with a copy button, below the transcript-style input and output.
-- Checker and interactor source, `judgemessage`, interactor stderr and build diagnostics never appear on student pages or in Test responses.
+- Given an interactive problem, the case panel lists each sample's interactor input and shows the problem's interaction notes; a sample without an interactor input is left out.
+- Given the student adds a case, its input is an interactor input (for example the hidden number), and the real interactor judges it like a sample. The case panel says "…Cases you add are judged the same way."
+- When the student presses Test, the browser compiles the program and runs it against the interactor once per case. Each case shows its verdict, the student's stderr, the time, and a transcript with "From the interactor" and "From your program", each cut at 64 KiB.
+- Given the interactor fails (an exit code other than 42 or 43, or a limit), the case is SE; otherwise the student's TLE, MLE or RE wins; otherwise the interactor's AC or WA stands, as in official judging.
+- Given JavaScript or TypeScript is selected, Test is disabled with "Test can't run interactive problems in JavaScript or TypeScript yet…"; switching to another language enables it.
-### Authoring
+### Problem page and authoring
+- On an interactive problem the statement shows the "Interaction" notes, and each sample shows its interactor input with a copy button, above the transcript-style input and output.
- The basic info section shows "Interaction notes" (Markdown, up to 8,000 characters) only for interactive problems.
- On an interactive problem each sample has an "Interactor input" field, and its input and output are labelled as the two sides of the transcript. Saving samples on an interactive problem with any sample lacking a non-blank interactor input fails with "Every interactive sample needs an interactor input."
-- Saving a checker or interactor configuration that Test supports starts a background build; saving never fails because of it.
-- The judge section shows editors the build status for Test: "Test can run this judge program.", "Preparing this judge program for Test…", or "This judge program can't run in Test. Submissions are still judged normally." with the build output. While it says "Preparing", it re-checks every 3 seconds for up to a minute without a reload. When the status cannot be read it says "Couldn't check this judge program right now." and does not re-check. When the test judge is off it says "Test judging isn't enabled on this server."; otherwise a Python interactor shows "Test doesn't support Python interactors yet."
-- On a checker problem the judge section offers "Check samples with the checker", disabled with "Save your changes first…" while the section has unsaved changes. It lists each sample's verdict, the checker's message for rejected ones, and when any sample is rejected: "The checker must accept each sample's output, because students' Test uses it as the answer…".
+- Students can read the checker or interactor source; `judgemessage` and the interactor's stderr are not shown in Test.
diff --git a/docs/operations/DEPLOYMENT.md b/docs/operations/DEPLOYMENT.md
index d38efcd1f..539e10577 100644
--- a/docs/operations/DEPLOYMENT.md
+++ b/docs/operations/DEPLOYMENT.md
@@ -50,7 +50,6 @@ through the chart (OPS-12).
| `ADVANCED_IMAGE_ALLOWED_REGISTRIES` | major public registries | Registry hosts accepted for digest-pinned special_env image refs (`web.advancedImageAllowedRegistries`) |
| `SUBMIT_COOLDOWN_MIN_SEC` | `0`; chart `30` (`web.submitCooldownMinSec`) | Range 0–600. Seconds between two submissions to the same problem in the same context, and the minimum exam/contest cooldown |
| `DB_POOL_MAX` | `10` (`web.dbPoolMax`) | Prisma pool size per web pod |
-| `TEST_JUDGE_ENABLED` | `false`; chart follows `worker.test.enabled` | `true` enables checker and interactive Test through the test worker; standard Test never needs it |
| `REGISTRY_PUBLIC_HOST`, `REGISTRY_INTERNAL_URL`, `REGISTRY_TOKEN_ISSUER` | chart-derived when `registry.enabled` | Self-hosted registry token service |
| `REGISTRY_TOKEN_PRIVATE_KEY`, `REGISTRY_TOKEN_CERT`, `REGISTRY_PULL_PASSWORD_HASH` | empty | Runtime Secret; see [Self-hosted registry](#self-hosted-registry) |
@@ -89,30 +88,26 @@ reached over a private network.
`parseWorkerEnv` validates at boot and throws on a missing required key. The
schema is a union on `EXECUTION_BACKEND`, so each backend requires only the keys
-it uses. The test worker never creates sandbox Jobs, but the chart still gives it
-the Kubernetes sandbox keys so the schema accepts it.
-`tests/unit/infra/env-manifest-parity.test.ts` checks the rendered chart against it.
-
-| Variable | Required / default | Purpose |
-| ----------------------------------------------------------------------------------------------------- | ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `EXECUTION_BACKEND` | required: `docker`, `kubernetes` | Sandbox backend |
-| `PORT` | required | Health server (`/healthz`, `/readyz`) |
-| `REDIS_URL` | required | Redis connection |
-| `SANDBOX_IMAGE` | required | Standard sandbox image |
-| `WORKER_CONCURRENCY` | required, 1–64 | Activity slots per task queue |
-| `WORKER_MIN_CONCURRENCY` | unset | Judge only: slots float between this and `WORKER_CONCURRENCY` by node load (see [Judge Queue](../runbooks/judge-queue.md)) |
-| `WORKER_MODE` | `all` | `all`, `judge` (queues `judge`, `judge-state`, `judge-cleanup`), `platform` (queue `platform`), `test` (queue `test-judge`) |
-| `SANDBOX_MEMORY_HEADROOM_MB`, `SANDBOX_MAX_MEMORY_MB` | `64`, `1536` | Sandbox memory ceiling above the problem limit |
-| `SANDBOX_CPU_LIMIT`, `SANDBOX_MEMORY_MB`, `SANDBOX_PIDS_LIMIT` | required (Docker) | Per-sandbox limits |
-| `K8S_NAMESPACE`, `K8S_CPU_REQUEST`, `K8S_CPU_LIMIT`, `K8S_MEMORY_REQUEST`, `K8S_MEMORY_LIMIT` | required (Kubernetes) | Sandbox namespace and container resources (`worker.sandbox.*`) |
-| `K8S_RUN_PARALLELISM` | `1`, range 1–8 | Testcases one stage Pod runs at once; its run container requests half and is limited to this many CPUs |
-| `K8S_RUNTIME_CLASS_NAME` | required, must be `gvisor` | RuntimeClass for every sandbox Pod |
-| `K8S_IMAGE_PULL_SECRET` | unset | dockerconfigjson Secret in the sandbox namespace (`worker.sandbox.imagePullSecret`) |
-| `REGISTRY_GC_IMAGE`, `REGISTRY_GC_NAMESPACE`, `REGISTRY_GC_CONFIG_CONFIGMAP`, `REGISTRY_GC_S3_SECRET` | defaults match the chart | Registry garbage-collection Job |
-| `SUBMISSION_PENDING_TIMEOUT_MINUTES` | `10`, range 10–1440 | Platform worker: stale-submission cutoff; a running judge workflow is exempt |
-| `WASM_OJ_RUNTIME_DIR`, `WASM_OJ_TOOLCHAIN_DIR` | empty; chart `/opt/wasm-oj/bin`, `/opt/wasm-oj/toolchains` | Test worker: directory with `wasm-oj-compiler` and `wasm-oj-runner`; directory whose `node_modules` holds the server toolchains. Required in `test` mode; `all` serves `test-judge` only when both are set |
-| `WASM_OJ_CACHE_DIR` | `/tmp/wasm-oj`; chart `/var/cache/wasm-oj` (2Gi emptyDir) | Test worker engine cache, one subdirectory per slot |
-| `TEST_JUDGE_SLOTS` | `2`, range 1–8 (`worker.test.slots`) | Test worker engines and `test-judge` activity slots |
+it uses. `tests/unit/infra/env-manifest-parity.test.ts` checks the rendered
+chart against it.
+
+| Variable | Required / default | Purpose |
+| ----------------------------------------------------------------------------------------------------- | -------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| `EXECUTION_BACKEND` | required: `docker`, `kubernetes` | Sandbox backend |
+| `PORT` | required | Health server (`/healthz`, `/readyz`) |
+| `REDIS_URL` | required | Redis connection |
+| `SANDBOX_IMAGE` | required | Standard sandbox image |
+| `WORKER_CONCURRENCY` | required, 1–64 | Activity slots per task queue |
+| `WORKER_MIN_CONCURRENCY` | unset | Judge only: slots float between this and `WORKER_CONCURRENCY` by node load (see [Judge Queue](../runbooks/judge-queue.md)) |
+| `WORKER_MODE` | `all` | `all`, `judge` (queues `judge`, `judge-state`, `judge-cleanup`), `platform` (queue `platform`) |
+| `SANDBOX_MEMORY_HEADROOM_MB`, `SANDBOX_MAX_MEMORY_MB` | `64`, `1536` | Sandbox memory ceiling above the problem limit |
+| `SANDBOX_CPU_LIMIT`, `SANDBOX_MEMORY_MB`, `SANDBOX_PIDS_LIMIT` | required (Docker) | Per-sandbox limits |
+| `K8S_NAMESPACE`, `K8S_CPU_REQUEST`, `K8S_CPU_LIMIT`, `K8S_MEMORY_REQUEST`, `K8S_MEMORY_LIMIT` | required (Kubernetes) | Sandbox namespace and container resources (`worker.sandbox.*`) |
+| `K8S_RUN_PARALLELISM` | `1`, range 1–8 | Testcases one stage Pod runs at once; its run container requests half and is limited to this many CPUs |
+| `K8S_RUNTIME_CLASS_NAME` | required, must be `gvisor` | RuntimeClass for every sandbox Pod |
+| `K8S_IMAGE_PULL_SECRET` | unset | dockerconfigjson Secret in the sandbox namespace (`worker.sandbox.imagePullSecret`) |
+| `REGISTRY_GC_IMAGE`, `REGISTRY_GC_NAMESPACE`, `REGISTRY_GC_CONFIG_CONFIGMAP`, `REGISTRY_GC_S3_SECRET` | defaults match the chart | Registry garbage-collection Job |
+| `SUBMISSION_PENDING_TIMEOUT_MINUTES` | `10`, range 10–1440 | Platform worker: stale-submission cutoff; a running judge workflow is exempt |
### Object storage
@@ -272,12 +267,9 @@ unsupported.
### Verify
-1. `kubectl rollout status deploy/nojv-web deploy/nojv-worker deploy/nojv-worker-platform -n nojv`,
- plus `deploy/nojv-worker-test` where `worker.test.enabled`.
+1. `kubectl rollout status deploy/nojv-web deploy/nojv-worker deploy/nojv-worker-platform -n nojv`
2. Web `/api/livez` and `/api/readyz`; worker `/readyz`.
-3. Where the test worker runs, press Test on a C++ checker problem: its samples show
- "Judged on the server" verdicts.
-4. Watch logs for at least 15 minutes.
+3. Watch logs for at least 15 minutes.
Secrets are rotated out-of-band in the runtime Secret, then the affected
Deployments are restarted.
@@ -311,8 +303,7 @@ Production never runs these by hand. The migrator Job
`migrator.releaseWindow` (default `true`) is a render-time flag because Helm
cannot see what the hook will find. When true on an upgrade, web, judge and
-platform render with `replicas: 0` (the test worker has no database access and
-keeps its replicas), the HPA targets the maintenance Deployment,
+platform render with `replicas: 0`, the HPA targets the maintenance Deployment,
and the post-upgrade Job (`templates/web-maintenance.yaml`, also post-rollback)
starts and verifies the new workloads, then restores the HPA. The release
workflow sets it to `false` only when `packages/db/prisma/migrations` is
@@ -378,7 +369,6 @@ apply ad-hoc down migrations.
| web | HPA 1–3, CPU 70%, 384Mi / 1Gi memory | HPA 2–15, CPU 70%, 384Mi / 1Gi memory |
| judge | 1 replica, slots 2–8 by load, 1Gi / 3Gi memory | 2 replicas × 2 slots |
| platform | 1 replica | 2 replicas |
-| test | 1 replica, 2 slots, 100m / 2 CPU, 512Mi / 3Gi | off (`worker.test.enabled`), so no checker or interactive Test |
| registry | 1 replica | 2 replicas |
| sandbox | quota 16 pods / 6 CPU / 16Gi; judge container 300m | quota 10 pods / 10 CPU / 30Gi; one on-demand gVisor node plus Spot 0–4 |
| postgres | CNPG, 500m CPU, 2Gi memory request = limit | Cloud SQL, outside the chart |
@@ -392,11 +382,9 @@ capacity.
Single-machine Postgres has a memory limit equal to its request and no CPU
limit, so its usage never exceeds its request and kubelet node-pressure
eviction takes every pod above its request first. Its requests count against
-the 10 vCPU / 24 GiB node with the other platform pods, about 3.3 CPU of
-requests in total; sandbox Jobs schedule into the remaining 6.7 CPU, which holds
-the quota's twelve half-CPU stage Pods: eight running and four terminating. The
-test worker requests 100m with a 2-CPU limit, so it uses idle node CPU without
-taking request budget from stage Pods.
+the 10 vCPU / 24 GiB node with the other platform pods, about 3.2 CPU of
+requests in total; sandbox Jobs schedule into the remaining 6.8 CPU, which holds
+the quota's twelve half-CPU stage Pods: eight running and four terminating.
The quota caps admission but does not reserve node capacity: a stage Pod that does
not fit stays Pending (`Unschedulable`), which lowers the load-aware slot budget,
and after 30 s the worker retries it as `SandboxBackpressureError`. The VM uses the
@@ -405,15 +393,14 @@ time.
### Disruption and Shutdown
-| Setting | Web | Judge / platform / test worker |
-| ------------------------ | -------------------------------------- | ----------------------------------------------- |
-| PodDisruptionBudget | `maxUnavailable: 1` when `pdb.enabled` | `maxUnavailable: 1` when `pdb.enabled` |
-| Spread | zone + node, `ScheduleAnyway` | zone + node, `ScheduleAnyway` (judge, platform) |
-| Release window | drained to the maintenance page | judge and platform drained; test keeps running |
-| `terminationGracePeriod` | 60 s | 120 s |
-| Shutdown after SIGTERM | 10 s `preStop`, then adapter-node 30 s | Temporal `shutdownGraceTime` 30 s, 40 s total |
-| Load-balancer drain | GKE `BackendConfig` 30 s | — |
-| Probe timeout | readiness 3 s, liveness 5 s | 5 s (above the 3 s in-process check budget) |
+| Setting | Web | Judge / platform worker |
+| ------------------------ | -------------------------------------- | --------------------------------------------- |
+| PodDisruptionBudget | `maxUnavailable: 1` when `pdb.enabled` | `maxUnavailable: 1` when `pdb.enabled` |
+| Spread | zone + node, `ScheduleAnyway` | zone + node, `ScheduleAnyway` |
+| `terminationGracePeriod` | 60 s | 120 s |
+| Shutdown after SIGTERM | 10 s `preStop`, then adapter-node 30 s | Temporal `shutdownGraceTime` 30 s, 40 s total |
+| Load-balancer drain | GKE `BackendConfig` 30 s | — |
+| Probe timeout | readiness 3 s, liveness 5 s | 5 s (above the 3 s in-process check budget) |
The registry gets the same budget and spread as the workers; its replicas share
the object-storage blob backend and `REGISTRY_HTTP_SECRET`, so a push can
@@ -505,36 +492,6 @@ Svelte UI libraries) belongs in `devDependencies`. The WASM-OJ toolchain and
browser runtime assets are served from `build/client`, so the runtime stage deletes
their copies under `node_modules`; the server imports only the descriptor modules.
-#### Worker WASM-OJ runtime
-
-The worker image carries the test judge's runtime in two stages copied before any
-app layer (OPS-21):
-
-- `wasm-oj-runtime` builds `wasm-oj-compiler` and `wasm-oj-runner` from
- `wasm-oj/forge` `crates/runtime-core` at `WASM_OJ_FORGE_TAG` (`v0.2.3`), checks
- that the tag resolves to `WASM_OJ_FORGE_COMMIT`, and installs them stripped in
- `/opt/wasm-oj/bin`.
-- `wasm-oj-toolchains` runs `npm ci` on `infra/docker/wasm-oj-toolchains/`
- (`@wasm-oj/toolchain-clang` and `@wasm-oj/toolchain-python`) into
- `/opt/wasm-oj/toolchains`.
-
-Both reset file timestamps, so the layers are byte-identical across releases and
-only a forge upgrade re-pulls them (about 82 MB compressed). Renovate does not
-update `@wasm-oj/*` or the `rust` builder image. To upgrade WASM-OJ:
-
-1. Bump `WASM_OJ_FORGE_TAG` and `WASM_OJ_FORGE_COMMIT` in `infra/docker/worker.Dockerfile`.
-2. Bump the toolchain versions in `infra/docker/wasm-oj-toolchains/package.json`
- and regenerate its `package-lock.json` (`npm install --package-lock-only`).
-3. Bump the `@wasm-oj/*` pins in `apps/web/package.json` and `apps/worker/package.json`
- and their `minimumReleaseAgeExclude` entries in `pnpm-workspace.yaml`, drop or
- rebase `patches/@wasm-oj__server@*.patch` and its `patchedDependencies` entry, then
- `pnpm install`.
-4. Update `WASM_OJ_SERVER_VERSIONS` in `packages/core/src/judge/test-judge-program.ts`;
- the new server identity rebuilds every cached judge program on first use.
-5. Run `pnpm exec vitest run --project unit tests/unit/infra/wasm-oj-pins.test.ts`,
- build the worker image, and run the gated runtime test
- ([Testing Strategy](../runbooks/testing.md#service-prerequisites)).
-
#### Standard judge toolchain
`packages/core/src/judge-environment.json` is the source of truth for the
diff --git a/docs/operations/QUALITY_SCORE.md b/docs/operations/QUALITY_SCORE.md
index 85f7b8974..73dcbe834 100644
--- a/docs/operations/QUALITY_SCORE.md
+++ b/docs/operations/QUALITY_SCORE.md
@@ -57,7 +57,6 @@ Work that is known, not done, and not covered by an in-flight plan. Remove an it
### Production evidence
- Measure GKE judge concurrency against the sandbox quota ceiling (single-machine was measured on 2026-09-26; see the baseline). See OPS-11 and [Judge Queue](../runbooks/judge-queue.md).
-- Load-test checker and interactive Test before an exam uses them: about 65 students pressing Test with the virtual-student harness, watching `nojv-worker-test` CPU, `test_judge_busy` responses, web latency and the judge worker's slot budget, which Test CPU on the shared node can hold back (JDG-13). Python checker cost per case (about 1.1 s on a development machine) is unmeasured on production hardware (JDG-26).
### High availability
@@ -68,11 +67,12 @@ Work that is known, not done, and not covered by an in-flight plan. Remove an it
- SonarQube reports 83 functions over the cognitive complexity limit (rule S3776, threshold 15). The worst are the better-auth `hooks.before` middleware in `apps/web/src/lib/auth.server.ts` (77), `scripts/judge-benchmark.ts` (72), `scripts/check-supply-chain-policy.mjs` (63) and `durableJudgeWorkflow` (61; any split must keep replay determinism).
- The admin audit log's actor, action and target columns have no filters; adding them needs server-side filtering in `adminAuditLogRepo.listPaged` (UI-12).
- Browser Test (WASM-OJ) deferred scope: official Submit from the browser, Advanced problems and limit calibration stay server-only until decided otherwise (JDG-15).
-- Python interactors cannot run in Test until a `wasm-oj/forge` release accepts runtime-bundle interactors; then bump the pins (OPS-21) and drop the Python-interactor check in `packages/core/src/judge/test-capability.ts` and `apps/worker/src/activities/test-judge.ts`. JavaScript and TypeScript contestants on interactive problems wait for streaming QuickJS stdin upstream.
-- Upstream `interact` does not end a CPU-bound contestant at its logical-time or instruction budget as `run` does; the test judge reports a full wall stop as TLE (JDG-26). Raise it upstream, then drop the wall-stop mapping.
-- `@wasm-oj/server` 0.2.3 drops its child-stdin error listener when it cancels or times out a run, so a cancel during the request write would crash the worker with an uncaught `EPIPE`; `patches/@wasm-oj__server@0.2.3.patch` keeps a no-op listener. Fix `src/server/server-runner.ts` upstream, then drop the patch at the next forge bump.
-- Test-judge request objects orphaned by a web crash between write and delete (`test-judge-requests/`) and the judge-program build cache (`test-judge-programs/`) are never swept; add a bucket lifecycle expiry or a prefix sweep.
+- Browser checker and interactive Test need one `@wasm-oj` release that ships wasm-oj/forge#93 (Python interactors in `interact`), #95 (in-module metering of interactive programs), #96 (runtime-file export stall) and #99 (interactive sides in nested Workers); bump the pins to it before release (JDG-15). JavaScript and TypeScript contestants on interactive problems wait for streaming QuickJS stdin upstream.
+- The browser engine closes a stopped side's pipes, while official judging keeps reading the interactor's output after the contestant ends and only gives it EOF. An interactor that writes after the contestant stopped at a limit therefore fails in Test (a CPython interactor exits 120) and the case is SE where Submit gives TLE or MLE; a deadlock that reaches the shared wall stop is SE too. A Python interactor that starts after a fast-failing contestant has stopped hits this on its first write. Fix upstream by discarding a finished contestant's input instead of failing the write.
+- After such a broken pipe the browser transcript can repeat the line the interactor failed to write (one CPython line showed five times). Raise upstream with the fix above.
+- In WebKit an empty `for(;;);` loop is not stopped by the instruction meter and runs to the wall stop, in `run` and `interact` alike; a loop with a side effect stops at its budget. Raise upstream.
- Every official checker or interactive stage recompiles its judge program. Precompiling native judge programs once per source, language and sandbox image is open; measure the C++ per-stage compile cost first (on 2026-10-04 every production judge program was Python, which needs no compile).
+- Evaluate a Wasm fast path for official judging of standard problems: run standard-problem submissions under a server-side WASM-OJ runtime instead of a gVisor stage Pod. It could remove most per-stage Pod overhead, but it brings back a server WASM-OJ runtime and toolchains in the worker image (OPS-21, withdrawn), adds a second execution path whose logical-time limits need calibrating against native CPU time and whose toolchains must track `judge-environment.json`, and puts a WebAssembly runtime instead of gVisor between student code and the worker's credentials. Measure the per-stage overhead it would save first.
## Evidence rules
diff --git a/docs/operations/RELIABILITY.md b/docs/operations/RELIABILITY.md
index 90e7efa4d..65babed30 100644
--- a/docs/operations/RELIABILITY.md
+++ b/docs/operations/RELIABILITY.md
@@ -83,8 +83,6 @@ PostgreSQL is the only durable store for application data. Everything else is de
| Redis | Pub/sub, rate limits, short-lived security proofs and read-through caches only (DAT-10, DAT-11); see [Redis](../architecture/REDIS.md) |
| SSE events | Ephemeral nudges; clients reconnect and read current state |
-Test-judge request objects are transient and the judge-program build cache is rebuilt on a miss, so neither needs restoring.
-
Backups, retention and restore order are in [Backup & Restore](../runbooks/backup-restore.md); production refuses to render without off-host destinations (OPS-06).
## Service expectations
@@ -103,12 +101,12 @@ Backups, retention and restore order are in [Backup & Restore](../runbooks/backu
### Redis unavailable
-- **Impact**: no SSE events. `apiRateLimiter` falls back to a per-process memory limiter; every other limiter (write, draft, test judge, form, auth, sign-in, exam sign-in, registry token) fails closed with 503 (DAT-12), and the test-judge in-flight lock answers 503 `test_judge_unavailable`. Redis-held security proofs (step-up, admin MFA/mode, 2FA setup) are unavailable, so privileged actions require fresh verification. Caches fall through to PostgreSQL. Submission cooldown uses PostgreSQL. Dispatched judging continues.
+- **Impact**: no SSE events. `apiRateLimiter` falls back to a per-process memory limiter; every other limiter (write, draft, form, auth, sign-in, exam sign-in, registry token) fails closed with 503 (DAT-12). Redis-held security proofs (step-up, admin MFA/mode, 2FA setup) are unavailable, so privileged actions require fresh verification. Caches fall through to PostgreSQL. Submission cooldown uses PostgreSQL. Dispatched judging continues.
- **Recovery**: restore connectivity; clients reconnect and read current state. Nothing needs rebuilding.
### Temporal unavailable
-- **Impact**: no new workflows; in-flight workflows pause. Checker and interactive Test fail with 503; nothing is left to recover.
+- **Impact**: no new workflows; in-flight workflows pause.
- **Invariant**: acceptance commits source, immutable snapshot and dispatch intent before returning. Temporal start is a best-effort wakeup; failure leaves the outbox pending for the minute durable-work processor and never produces SE.
- **Topology**: self-hosted (OPS-09). Single-machine runs the official chart with one pod per role on the CNPG database, so node loss pauses workflows until pods reschedule. HA options: `infra/gcp/gke/temporal/HA-PRODUCTION.md`.
- **Recovery**: workflows resume from history; no data loss.
@@ -119,12 +117,6 @@ Backups, retention and restore order are in [Backup & Restore](../runbooks/backu
- **Topology**: GKE runs two platform workers. Their startup work is safe to run concurrently: `ensure*` starts singletons by fixed workflow ID and keeps a running one, the stale-submission sweep kills only through a conditional status update, and execution recovery enqueues dispatch with `skipDuplicates` and bumps the recovery epoch under row locks after re-checking the owner. The SQL-backed gauges are reported by each replica and alerts read them with `max()`. Single-machine runs one of each worker.
- **Recovery**: Deployments restart failed processes; accepted workflows resume and the outbox dispatches after Temporal is reachable. Node and container-runtime recovery is an operator action. With `pdb.enabled` (GKE), one voluntary eviction at a time.
-### Test worker unavailable
-
-- **Impact**: checker and interactive Test only. With no test worker polling, each request waits out its 30 s workflow timeout and answers `test_judge_busy`; standard Test, Submit and official judging are unaffected because Test never uses `judge` (JDG-26).
-- **Bounds**: every request ends within 30 s and leaves no state except its request object, which web and the worker both delete. A judge-program build dispatch lost after a judge-config save costs one on-demand build at the next Test.
-- **Recovery**: `kubectl -n nojv rollout restart deploy/nojv-worker-test`; see [Judge Queue](../runbooks/judge-queue.md#test-judge).
-
### Sandbox failure
- Program failures keep their normal verdict. Capacity waits (including quota rejections, even when the message says `forbidden`) stay `waiting_capacity` and retry every 30s without consuming the failure budget.
diff --git a/docs/operations/SECURITY.md b/docs/operations/SECURITY.md
index a33573e08..c3ad6447d 100644
--- a/docs/operations/SECURITY.md
+++ b/docs/operations/SECURITY.md
@@ -16,27 +16,27 @@ Security controls as implemented: what must hold and where it is enforced. Attac
- `apps/web/src/lib/utils/markdown.ts` — DOMPurify sanitizer and remote-image rewrite
- `apps/web/svelte.config.js` — CSP
- `apps/worker/src/sandbox/docker/args.ts`, `apps/worker/src/sandbox/kubernetes/pod-spec.ts` — sandbox hardening
-- `infra/charts/nojv/templates/{namespaces,sandbox-policy,worker-rbac,worker-test.deployment,web.ingress,cloudflared.deployment}.yaml` — cluster-level controls
-- `packages/application/src/test-judge/index.ts`, `apps/worker/src/activities/test-judge.ts` — test-judge authorisation, sample-only requests and response bounds
+- `infra/charts/nojv/templates/{namespaces,sandbox-policy,worker-rbac,web.ingress,cloudflared.deployment}.yaml` — cluster-level controls
+- `packages/application/src/problem/judge-program.ts` — judge-program source access for browser Test
## Sensitive Data
-| Data | Storage | Protection |
-| ------------------------ | ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------- |
-| Credential passwords | `Account.password` | bcrypt (cost 10) via `emailAndPassword.password.hash` |
-| Temporary exam passwords | `ExamCredential.passwordHash` / `passwordCiphertext` | scrypt hash + ciphertext keyed by `BETTER_AUTH_SECRET`; staff reveal; hard expiry |
-| OAuth provider tokens | `Account.accessToken` / `refreshToken` | Encrypted with `BETTER_AUTH_SECRET` (`account.encryptOAuthTokens`); never read by NOJV |
-| Session tokens | `Session.token` | httpOnly cookie; checked every request (no cookie cache) |
-| API tokens | `ApiToken` prefix + sha256 hash | Shown once, mandatory expiry (SEC-07) |
-| TOTP enrollment material | Redis, pending until confirmed | Encrypted; committed atomically with backup codes on confirmation |
-| Submission source | Object storage `submissions//sources/` | Read only via domain helpers and the worker |
-| Graded testcases | `TestcaseSet` / `Testcase` + object storage | Never reach non-staff (SEC-12); only `Problem.samples` is rendered |
-| Checkers and interactors | Object storage; WASM builds in `test-judge-programs/` | Read by problem editors and workers only; Test runs them on the server and returns verdicts and `teammessage` (JDG-05, SEC-15) |
-| Hidden workspace files | `ProblemWorkspaceFile` (`visibility = hidden`) | Presentation only for packaged helpers/drivers or opaque APIs; compile/run can read them. Never store secrets or answers (PRB-01) |
-| Advanced grade images | Registry `t//…` | Hold answers; namespace-scoped registry tokens ([Sandbox](#sandbox-isolation)) |
-| Code drafts | `CodeDraft` rows; unsynced edits in `localStorage` | Owner-only; local edits sealed ([Integrity](#exam-and-contest-integrity)) |
-| Problem / user images | Object storage, served same-origin via `/api/storage/*` | Public read; never store secret material there (PRB-05) |
-| Runtime secrets | `.env` locally; chart Secret `nojv-runtime-secrets` | `.env` untracked; `.env.example` shape-only |
+| Data | Storage | Protection |
+| ------------------------ | ------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| Credential passwords | `Account.password` | bcrypt (cost 10) via `emailAndPassword.password.hash` |
+| Temporary exam passwords | `ExamCredential.passwordHash` / `passwordCiphertext` | scrypt hash + ciphertext keyed by `BETTER_AUTH_SECRET`; staff reveal; hard expiry |
+| OAuth provider tokens | `Account.accessToken` / `refreshToken` | Encrypted with `BETTER_AUTH_SECRET` (`account.encryptOAuthTokens`); never read by NOJV |
+| Session tokens | `Session.token` | httpOnly cookie; checked every request (no cookie cache) |
+| API tokens | `ApiToken` prefix + sha256 hash | Shown once, mandatory expiry (SEC-07) |
+| TOTP enrollment material | Redis, pending until confirmed | Encrypted; committed atomically with backup codes on confirmation |
+| Submission source | Object storage `submissions//sources/` | Read only via domain helpers and the worker |
+| Graded testcases | `TestcaseSet` / `Testcase` + object storage | Never reach non-staff (SEC-12); only `Problem.samples` is rendered |
+| Checkers and interactors | Object storage, behind verified script pointers | Readable by anyone who may view the problem in a context, through `/api/problems/[id]/judge-program`; browser Test runs them (JDG-15). Authors own their robustness |
+| Workspace files | `ProblemWorkspaceFile` (`editable` / `readonly`) | Shown to students and run by browser Test; compile/run can read them. Never store secrets or answers (PRB-01) |
+| Advanced grade images | Registry `t//…` | Hold answers; namespace-scoped registry tokens ([Sandbox](#sandbox-isolation)) |
+| Code drafts | `CodeDraft` rows; unsynced edits in `localStorage` | Owner-only; local edits sealed ([Integrity](#exam-and-contest-integrity)) |
+| Problem / user images | Object storage, served same-origin via `/api/storage/*` | Public read; never store secret material there (PRB-05) |
+| Runtime secrets | `.env` locally; chart Secret `nojv-runtime-secrets` | `.env` untracked; `.env.example` shape-only |
## Request Boundary
@@ -69,7 +69,6 @@ SvelteKit `csrf.checkOrigin` is disabled so `/api/registry/token` can accept the
- Global adapter-node `BODY_SIZE_LIMIT` is 64 MiB (`infra/docker/web.Dockerfile`), sized for the largest upload (60 MB bundle). Do not lower it without re-checking that.
- `POST /api/submissions`: 2 MiB (`MAX_SUBMISSION_BODY_BYTES`), `Content-Length` pre-check then streamed count.
-- `POST /api/problems/[id]/test-judge`: 12 MiB (`TEST_JUDGE_REQUEST_BODY_BYTES`, room for an 8 MiB contestant artifact in base64), same pre-check and streamed count; the application refuses a larger decoded artifact (413) before storing the request.
- Other JSON mutation routes: 1 MiB via `assertJsonBodyWithinLimit` + `readJsonBody` (`JSON_BODY_LIMIT_BYTES`); `readJsonBody` counts streamed bytes and returns 413 regardless of `Content-Length` (SEC-10).
- Image, avatar and workspace-file multipart bodies count streamed bytes before FormData parsing, bounded by the file limit plus 64 KiB for form fields and multipart framing; the file limit is checked again after parsing. Checker/interactor and bundle uploads enforce their own size limits.
- `/api/auth/*` and registry credential forms stop at 64 KiB before JSON or FormData parsing.
@@ -83,7 +82,6 @@ SvelteKit `csrf.checkOrigin` is disabled so `/api/registry/token` can accept the
| `apiRateLimiter` | 300 / min | `apiHandler` routes; per-replica memory fallback when Redis is down |
| `writeApiRateLimiter` | 10 / min | `writeApiHandler` routes (submissions, uploads, plagiarism runs, …) |
| `draftApiRateLimiter` | 60 / min | `/api/drafts` |
-| `testJudgeApiRateLimiter` | 30 / min | `/api/problems/[id]/test-judge`, plus one request in flight per user ² |
| form actions | 20 / min | `withRateLimit` form actions |
| `apiTokenAuthRateLimiter` | 300 / min | Whitelisted bearer requests, by IP before token database lookup |
| `authRateLimiter` | 60 / min | Every `/api/auth/*` request, including OAuth and exam sign-in |
@@ -95,7 +93,6 @@ SvelteKit `csrf.checkOrigin` is disabled so `/api/registry/token` can accept the
| `remoteAssetFetchRateLimiter` | 120 / min | Authenticated remote-image relay, keyed by user |
¹ Students sharing a classroom IP do not consume each other's quota; invalid usernames share one bucket per IP.
-² A second concurrent request gets 429 `test_judge_busy`; the lock is `nojv:test-judge:in-flight:{userId}` ([Redis](../architecture/REDIS.md#rate-limiting)).
- Quotas count every attempt, including successful sign-ins.
- In production, every limiter except `apiRateLimiter` fails closed (429 limited, 503 unavailable) on operational Redis errors; programming errors propagate (DAT-12).
@@ -150,7 +147,7 @@ Behavior specs: [Exams](../features/exams.md), [Proctoring](../features/proctori
- Exam IP whitelist, IP binding and page lock are server-side (ASM-19, ASM-20). During an active exam session `hooks.server.ts` runs the proctoring gate on exam paths and, with page lock enabled, on every page and `/api` request, and exam submissions and draft reads/writes recheck every denial reason independently of page lock; a failed active-exam lookup fails closed with 503. Exam entry begins at `startsAt` with no early-entry grace and runs the gate before creating a session, the first IP pin is a conditional write so concurrent first requests cannot both bind, violations recorded by a denied entry or submission are committed before the denial, and every binding reset writes an `ip_reset` session event. Page lock also denies `/api/contests/*`, `/api/posts/*`, `/api/comments/*` and `/api/problems/[id]/posts`. Contests have no IP or page gating.
- Context-bound submissions and drafts must target a problem in that context (ASM-21).
-- Server-judged Test (`/api/problems/[id]/test-judge`) passes the same context check as drafts (`assertProblemContextAllowed`), including the exam gate and, for contests, participation inside the window, and it judges only problem samples by index with server-side data (SEC-15). Page lock does not block it.
+- `/api/problems/[id]/judge-program` is `exam-scoped`: it grants the problem page's view access for the requested context, and a page-locked exam session reaches only that exam's problems. Students keep reading a judge program after an exam or contest ends while they can still view the problem.
- The submission cooldown, `max(activity submitCooldownSec, SUBMIT_COOLDOWN_MIN_SEC)` in every context (PRB-22), is checked in PostgreSQL under a `pg_advisory_xact_lock` keyed by context, user and problem (`packages/application/src/shared/submit-cooldown.ts`); sample runs and reference solutions are exempt.
- Code drafts: `CodeDraft` rows are owner-only via `/api/drafts`. Exam drafts can be written only during an active session on a running exam for a problem in it; during a session only that exam's drafts are reachable. Unacknowledged local edits are stored under `nojv:draft:v2::…`, AES-GCM sealed with `HMAC-SHA256(BETTER_AUTH_SECRET, "code-draft:")` delivered only to that user's `(app)` layout, with the storage key as additional authenticated data. Legacy plaintext `nojv:draft:v1:` drafts are re-sealed for the first opener, except exam drafts, which are deleted on the first draft load in that browser and never adopted.
@@ -180,11 +177,11 @@ Kubernetes requirements (JDG-20):
- The worker refuses to start unless the `gvisor` RuntimeClass, a hardened smoke Pod and a NetworkPolicy enforcement probe succeed. A CNI that enforces NetworkPolicy is a hard dependency.
- Split identities: the judge worker's `sandbox-job-manager` role has only create/get/list/watch/delete on sandbox resources, plus `patch` on ConfigMaps for the testcase cache's annotations (JDG-23); the platform worker has only the registry-GC role (token unmounted when the registry is disabled). Never add update, Secret or cross-namespace access, or `patch` on any other resource.
-Test judge (JDG-15, JDG-26, SEC-15):
+Browser Test (JDG-15):
-- Interactive Test runs the student's browser-compiled Wasm, and both Test kinds run checkers and interactors, inside `nojv-worker-test` under the pinned WASM-OJ runtime (`wasm-oj-runner`). The boundary is WASM-OJ's WebAssembly sandbox: upstream static admission, logical-time and instruction budgets, memory limits and a per-case wall stop ([limits](../architecture/JUDGE_PIPELINE.md#test-judge)).
-- The pod runs as uid 1001 with a read-only root filesystem, `drop: ALL`, no privilege escalation and `RuntimeDefault` seccomp, mounts no service-account token and gets no database credentials. It holds Redis, object-storage and Temporal access ([Threat Model](THREAT_MODEL.md#open-gaps)). It never polls `judge` or creates sandbox Jobs; with `networkPolicy.enabled` the `worker-egress` policy applies to it as to the other workers.
-- Judge programs, `judgemessage`, interactor stderr and build diagnostics never leave the server for students; build diagnostics reach only the problem's editors.
+- The server never compiles or executes anything for Test. The student's program and the problem's checker or interactor run in the student's browser under WASM-OJ.
+- Test receives only the problem's samples, which the statement already shows, and the judge program's source. Hidden testcases never leave the server (SEC-12).
+- Students can read a judge program in DevTools; that is an accepted risk, and authors own judge programs that hold up when read ([Threat Model](THREAT_MODEL.md#authorization-and-data-exposure)).
Advanced Mode (JDG-16, JDG-17, SEC-14, PRB-12):
diff --git a/docs/operations/THREAT_MODEL.md b/docs/operations/THREAT_MODEL.md
index c2ebf52e7..575e226a2 100644
--- a/docs/operations/THREAT_MODEL.md
+++ b/docs/operations/THREAT_MODEL.md
@@ -23,7 +23,6 @@ Enforcement points are listed in [Security — Key code](SECURITY.md#key-code).
| Admin / super-admin access | Platform-wide data and role control |
| `DATABASE_URL`, object-storage keys | Read/write of all data, source and testcases |
| Graded testcases, Advanced grade images | Grading integrity across every context |
-| Checker and interactor programs | Reference algorithms and checker holes exposed; verdicts gameable |
| Submission source and drafts | Privacy, academic integrity |
| Exam configuration and IP records | Exam integrity |
| OAuth client secrets, provider tokens | Application impersonation; provider API access for linked accounts |
@@ -41,7 +40,6 @@ Untrusted | Trusted
Browser ---> Cloudflare -----|---> web (SvelteKit) ---> PostgreSQL, Redis, object storage, Temporal
| |
Student code ----------------|---> sandbox Pod/container <--- worker (Temporal)
-Student Wasm (Test) ---------|---> worker-test (WASM-OJ runner) ---> Redis, object storage, Temporal
Teacher Advanced images -----|---> run / grade / service containers
Docker client (registry) ----|---> /api/registry/token ---> in-cluster registry
```
@@ -51,7 +49,6 @@ Docker client (registry) ----|---> /api/registry/token ---> in-cluster registry
| Internet ↔ origin | High | Only via Cloudflare; client IP trust depends on it |
| Browser ↔ web | High | All input untrusted; cookie and bearer-token auth |
| Student code ↔ sandbox | Critical | Arbitrary code, highest-risk input |
-| Student Wasm ↔ test worker | High | Compiled by the student's browser; runs in a pod with service credentials |
| Teacher image ↔ sandbox | High | Semi-trusted authors; grade images hold answers |
| Worker ↔ Docker daemon / K8s API | High | Worker creates workloads; its credentials are privileged |
| Web/worker ↔ Postgres, Redis, S3 | Medium | Internal network only; Redis unauthenticated in compose and in-cluster chart |
@@ -113,9 +110,8 @@ Each item names the mitigating control; _Residual_ is what remains.
- **IDOR on submissions or source (`/api/submissions/[id]`, `/source`)** — `getSubmissionForActor` returns 404 for non-owners ([Authorization](SECURITY.md#authorization)).
- **Graded testcase leak through results (e.g. echo-stdin submission)** — `sanitizeStudentResult` on every student result path (SEC-12).
-- **Reading a checker or interactor through Test** — Judge programs run only in the test worker; a Test response carries verdicts, `teammessage`, contestant stderr and the transcript, never the program, `judgemessage`, interactor stderr or diagnostics (JDG-15).
-- **Using Test as a checker or interactor oracle with crafted cases** — Test judges only problem samples, by index, from server-side data; custom cases never reach the judge program (SEC-15). _Residual:_ a student can vary their own output on public samples, which shows only how the checker treats those inputs.
-- **Hidden workspace file leak** — Filtered from student editor/API reads; the worker merges them into compilation and execution. _Residual:_ Student programs can read these files, so editor hiding provides no runtime confidentiality.
+- **Reading a checker or interactor** — Accepted risk: anyone who may view the problem in a context reads its judge program through `/api/problems/[id]/judge-program`, and browser Test runs it (JDG-15). _Residual:_ bugs such as a missing validity check or an off-by-one query limit are easier to find and then exploit in official judging; authors own them. A checker should only verify and read the optimum from `judge_answer`, and an interactor should read its secret from `judge_input`, as the seed programs do.
+- **Hidden testcases through Test** — Test receives only `Problem.samples` and the judge program's source; the server runs nothing for Test, so no Test path touches graded testcases (SEC-12).
- **Co-editor overreach (publish, export, other courses' submissions)** — Resource-based problem permissions with in-transaction recheck ([Authorization](SECURITY.md#authorization)).
- **Revoked staff finishing an upload started while authorized** — `lockProblemForEdit` recheck before commit.
- **Revoked staff dispatches rejudge after snapshot preparation** — Commit holds scope-authority and requester locks and rechecks current authority before execution/outbox writes.
@@ -140,8 +136,6 @@ Controls: [Sandbox Isolation](SECURITY.md#sandbox-isolation).
- **Cross-teacher image theft or replacement** — Namespace-scoped registry tokens. _Residual:_ Isolation is per teacher, not per course.
- **Compromised worker creates privileged workloads** — Pod Security `restricted`, split service accounts, minimal `sandbox-job-manager` role. _Residual:_ The judge identity can still create sandbox Jobs.
- **Checker/interactor exploits run output** — Validators run in the same hardened sandbox; output treated as untrusted data.
-- **Student Wasm escapes the test judge runtime** — WASM-OJ static admission, logical-time and instruction budgets, memory limits and wall stops; non-root, read-only, capability-free pod with no service-account token and no database credentials ([Sandbox Isolation](SECURITY.md#sandbox-isolation)). _Residual:_ the pod holds Redis, object-storage and Temporal access, so an escape from the WebAssembly sandbox reaches them (Open Gaps).
-- **Test floods starve official judging** — Test runs only on `test-judge` in its own Deployment with fixed slots and a 2-CPU limit; per-user 30 requests a minute and one in flight; each request ends within 30 s. _Residual:_ the test worker shares node CPU with stage Pods, and a full class can keep every slot busy so further Tests answer "busy".
### Uploads, Markdown and object storage
@@ -153,7 +147,7 @@ Controls: [Content and Uploads](SECURITY.md#content-and-uploads).
- **Reader tracking via remote Markdown images** — Authenticated same-origin relay with short private caching; browsers never contact upstream hosts.
- **SSRF via the image proxy (private IPs, rebinding, redirects)** — Public-only DNS pinning, redirect revalidation, HTTPS/443 only.
- **Storage or bandwidth exhaustion** — Shared 50 MiB problem budget including images, 50 MiB user content-image budget, atomic capacity reservations including pending writes, bounded avatar replacement, and remote-fetch limiting. Reader GET requests create no permanent storage; author URL imports consume the owner's quota. _Residual:_ Existing over-quota content is retained; remote references remain dependent on upstream availability unless imported.
-- **Hidden workspace file read by student code** — Hidden controls editor/API presentation for helpers, drivers and opaque assumed APIs. Official compilation/execution receives the file; authors must not put secrets or answers there (PRB-01).
+- **Secrets in workspace files** — Workspace files are `editable` or `readonly`; students see both, browser Test runs them and official compilation and execution receive them. Authors must not put secrets or answers there (PRB-01).
### Exam and contest integrity
@@ -178,23 +172,22 @@ Controls: [Infrastructure](SECURITY.md#infrastructure).
## Criticality
-| Level | Threats |
-| -------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Critical | Sandbox escape; `BETTER_AUTH_SECRET` or database credential leak; authorization bypass to admin; graded testcase or answer exposure |
-| High | Cross-user source access; exam integrity bypass; stored XSS; object-storage or registry credential leak; worker orchestrator abuse; OAuth secret leak; test-worker runtime escape; judge-program disclosure |
-| Medium | Submission or SSE flooding; storage exhaustion; plagiarism-run load; Redis tampering in dev or in-cluster; Zod schema disclosure |
-| Low | Undetected tab/device switching; public probe disclosure; development-only defaults |
+| Level | Threats |
+| -------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
+| Critical | Sandbox escape; `BETTER_AUTH_SECRET` or database credential leak; authorization bypass to admin; graded testcase or answer exposure |
+| High | Cross-user source access; exam integrity bypass; stored XSS; object-storage or registry credential leak; worker orchestrator abuse; OAuth secret leak |
+| Medium | Submission or SSE flooding; storage exhaustion; plagiarism-run load; Redis tampering in dev or in-cluster; Zod schema disclosure |
+| Low | Undetected tab/device switching; public probe disclosure; development-only defaults |
## Open Gaps
-| Gap | Current state | Recommendation | Priority |
-| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------- |
-| SSE concurrency caps are per replica | `acquireSseSlot` caps 5 streams per user per stream type and 2000 per replica, in memory; SSE routes are not rate-limited | Move counters to Redis if a global cap is needed | Low |
-| No per-account sign-in lockout | Password sign-in is limited per IP (and per username for exam passwords) | Add a per-account limiter if distributed brute force appears | Low |
-| Redis unauthenticated | Compose and the in-cluster chart Redis have no password; GKE uses external Redis | Add Redis auth for single-machine deployments | Low |
-| Browser tab / device switching not detected | Page lock covers NOJV server routes only | Only if remote proctoring becomes a requirement | Low |
-| No plagiarism concurrency cap | Dolos runs in-process per activity, bounded by one target's submissions and the activity timeout | Add an activity concurrency limit if parser contention appears | Low |
-| Test worker holds service credentials | Student Wasm runs in `nojv-worker-test`, which has the runtime S3 keys (whole bucket), Redis and Temporal access | Scope its object-storage credential to the test-judge prefixes and validator reads, or run the runner without them | Medium |
+| Gap | Current state | Recommendation | Priority |
+| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------- | -------- |
+| SSE concurrency caps are per replica | `acquireSseSlot` caps 5 streams per user per stream type and 2000 per replica, in memory; SSE routes are not rate-limited | Move counters to Redis if a global cap is needed | Low |
+| No per-account sign-in lockout | Password sign-in is limited per IP (and per username for exam passwords) | Add a per-account limiter if distributed brute force appears | Low |
+| Redis unauthenticated | Compose and the in-cluster chart Redis have no password; GKE uses external Redis | Add Redis auth for single-machine deployments | Low |
+| Browser tab / device switching not detected | Page lock covers NOJV server routes only | Only if remote proctoring becomes a requirement | Low |
+| No plagiarism concurrency cap | Dolos runs in-process per activity, bounded by one target's submissions and the activity timeout | Add an activity concurrency limit if parser contention appears | Low |
## Related Docs
diff --git a/docs/product/PRODUCT_SENSE.md b/docs/product/PRODUCT_SENSE.md
index d476c4ce0..b18ed8eac 100644
--- a/docs/product/PRODUCT_SENSE.md
+++ b/docs/product/PRODUCT_SENSE.md
@@ -21,7 +21,7 @@ NOJV is a single-institution online judge for university programming courses: pr
### Problems
-- Three types: `full_source`, `multi_file` (workspace files with editable/readonly/hidden visibility) and `special_env` (Advanced Mode, teacher-built images; creation needs an admin-granted permission) (PRB-01, PRB-12, JDG-16).
+- Three types: `full_source`, `multi_file` (workspace files with editable/readonly visibility) and `special_env` (Advanced Mode, teacher-built images; creation needs an admin-granted permission) (PRB-01, PRB-12, JDG-16).
- Judge modes: standard token compare, checker, interactor; subtasks score all-or-nothing (JDG-01 to JDG-04). Samples are presentation data, not testcases (PRB-03).
- One Markdown + KaTeX statement per problem; images uploaded by drag-and-drop or paste to object storage (PRB-05).
- Library with URL filters (difficulty, tags, solved / attempted / untried / bookmarked) and full-text search (PRB-14).
@@ -30,7 +30,7 @@ NOJV is a single-institution online judge for university programming courses: pr
### Submissions
-- Monaco workspace (desktop only), optional in-browser test runs whose checker and interactive samples are judged on the server without exposing the judge program ([spec](../features/problem-test.md), JDG-15), official judging in Docker or Kubernetes sandboxes ([Judge Pipeline](../architecture/JUDGE_PIPELINE.md)).
+- Monaco workspace (desktop only), optional in-browser test runs that also run the problem's checker or interactor in the browser, so students can read judge programs ([spec](../features/problem-test.md), JDG-15), official judging in Docker or Kubernetes sandboxes ([Judge Pipeline](../architecture/JUDGE_PIPELINE.md)).
- Live status over SSE with polling fallback (PRB-21); history and ownership-gated source view.
- `system_error` never costs an attempt (PRB-16); rejudges are audited (PRB-18).
- Resubmitting the same problem waits a cooldown: the activity setting or the platform minimum, whichever is longer, in every context (PRB-22).
diff --git a/docs/runbooks/getting-started.md b/docs/runbooks/getting-started.md
index b98296327..db0d03e7f 100644
--- a/docs/runbooks/getting-started.md
+++ b/docs/runbooks/getting-started.md
@@ -81,7 +81,7 @@ Rebuild after changing `apps/sandbox-runner` or `infra/docker/sandbox-runner.Doc
pnpm dev
```
-Starts web at and the worker with `WORKER_MODE=all` (judge, judge-state, judge-cleanup and platform queues, plus test-judge once the [local test judge](#run-the-test-judge-locally) is set up).
+Starts web at and the worker with `WORKER_MODE=all` (judge, judge-state, judge-cleanup and platform queues).
## 7. Verify
@@ -107,29 +107,6 @@ Outside production, `getClientIp` reads the `x-dev-ip` header first, then the so
curl -H "x-dev-ip: 10.1.2.3" http://localhost:5173/exams/
```
-### Run the test judge locally
-
-Checker and interactive Test need the WASM-OJ runtime that the worker image bundles. Standard Test and Submit work without it. Build the native runtime from the forge tag the worker image pins (`WASM_OJ_FORGE_TAG` in `infra/docker/worker.Dockerfile`; needs a Rust toolchain) and install the server toolchains from the image's lockfile:
-
-```bash
-git clone --depth 1 --branch v0.2.3 https://github.com/wasm-oj/forge ~/src/wasm-oj-forge
-cargo build --locked --release --manifest-path ~/src/wasm-oj-forge/crates/runtime-core/Cargo.toml \
- --bin wasm-oj-runner --bin wasm-oj-compiler
-mkdir -p ~/.cache/nojv-wasm-oj-toolchains
-cp infra/docker/wasm-oj-toolchains/package*.json ~/.cache/nojv-wasm-oj-toolchains/
-npm ci --ignore-scripts --prefix ~/.cache/nojv-wasm-oj-toolchains
-```
-
-Then add to `.env` (absolute paths) and restart `pnpm dev`:
-
-```bash
-WASM_OJ_RUNTIME_DIR=/home/you/src/wasm-oj-forge/crates/runtime-core/target/release
-WASM_OJ_TOOLCHAIN_DIR=/home/you/.cache/nojv-wasm-oj-toolchains
-TEST_JUDGE_ENABLED=true
-```
-
-The `WORKER_MODE=all` worker then also serves `test-judge`. Seeded checker problems such as `problem_any-two-sum` show "Judged on the server" on their samples. The same two directories enable the gated runtime test ([Testing Strategy](testing.md#service-prerequisites)).
-
## Troubleshooting
| Problem | Fix |
diff --git a/docs/runbooks/judge-queue.md b/docs/runbooks/judge-queue.md
index 39f1f2dfc..bcf60cc89 100644
--- a/docs/runbooks/judge-queue.md
+++ b/docs/runbooks/judge-queue.md
@@ -35,38 +35,11 @@ in the Temporal Helm values (`infra/gcp/gke/temporal/`), then let the config rel
- `matching.enableFairness: true` — without it dispatch inside one priority is FIFO;
the per-student dispatch gate still limits a student to one dispatched execution
per queue class.
-- One read and one write partition for `judge`, `judge-state`, `judge-cleanup`,
- `platform` and `test-judge`. With
+- One read and one write partition for `judge`, `judge-state`, `judge-cleanup` and
+ `platform`. With
the default four, few pollers leave tasks in unpolled partitions for up to a long
poll.
-## Test judge
-
-Checker and interactive Test run on `test-judge`, served only by `nojv-worker-test`
-(JDG-26); it never shares slots with official judging.
-
-```bash
-temporal task-queue describe -t test-judge
-kubectl -n nojv get deploy nojv-worker-test
-kubectl -n nojv logs deploy/nojv-worker-test --since=15m
-```
-
-- Students see "Test is busy" when every engine was taken for the whole 30 s
- window, when a request's 24 s ran out before every sample was judged, or when no
- test worker polls the queue (the task-queue description lists
- no pollers). Restart a stuck worker with
- `kubectl -n nojv rollout restart deploy/nojv-worker-test`.
-- A test worker restarting with "A test-judge engine failed its warm-up 3 times"
- could not run its startup judge programs; "Retrying a test-judge engine warm-up"
- alone is a recovered stall.
-- Slots are `worker.test.slots` (`TEST_JUDGE_SLOTS`, one WASM-OJ engine each, at
- most 8) under a 2-CPU and 3 GiB limit; the memory holds two slots each running a
- contestant of up to 1 GiB beside a 256 MiB interactor, plus Node. Raise slots
- and both limits together, and only after a Test load test.
-- "This problem's checker or interactor can't run in Test" means its cached build
- failed; the problem's editors see the diagnostics in the edit page's judge
- section. Saving a fixed checker or interactor rebuilds it.
-
## Capacity
- Slots = `worker.judge.concurrency` × judge replicas. Each slot runs one stage Job
@@ -80,7 +53,7 @@ kubectl -n nojv logs deploy/nojv-worker-test --since=15m
container past the sandbox memory ceiling. Node allocatable CPU minus platform pod
requests also bounds how many Jobs schedule; the chart cannot check it. Read
`kubectl describe node` (Allocated resources) before raising slots: single-machine
- platform pods request about 3.3 CPU, so 10 vCPU leaves room for thirteen half-CPU
+ platform pods request about 3.2 CPU, so 10 vCPU leaves room for thirteen half-CPU
stage Pods. A stage Pod that does not fit reports `Unschedulable`, which caps the
load-aware budget; one still unscheduled after 30 s fails with
`SandboxBackpressureError … Insufficient cpu`.
diff --git a/docs/runbooks/observability-setup.md b/docs/runbooks/observability-setup.md
index 08a87465a..420fe6007 100644
--- a/docs/runbooks/observability-setup.md
+++ b/docs/runbooks/observability-setup.md
@@ -21,7 +21,7 @@ stay in the container log pipeline.
- Both apps start an OpenTelemetry NodeSDK when `OTEL_EXPORTER_OTLP_ENDPOINT` is a valid URL, exporting to `/v1/metrics` every 30s. Unset, empty or invalid means no-op (production logs a warning).
- `OTEL_EXPORTER_OTLP_HEADERS` is optional comma-separated `key=value`; Grafana Cloud needs `Authorization=Basic `.
- Auto-instrumentation covers `http`, `pg`, `ioredis` and `undici`; `fs` and `dns` are disabled.
-- Service names: `OTEL_SERVICE_NAME_WEB` (default `nojv-web`), `OTEL_SERVICE_NAME_WORKER` (default `nojv-worker-judge`, `nojv-worker-platform` or `nojv-worker-test` from `WORKER_MODE`, `nojv-worker` for `all`). The collector's Prometheus exporter turns `service.name` into `job`, and Prometheus stores it as `exported_job`; filter or group app series by `exported_job`, never sum them across services.
+- Service names: `OTEL_SERVICE_NAME_WEB` (default `nojv-web`), `OTEL_SERVICE_NAME_WORKER` (default `nojv-worker-judge` or `nojv-worker-platform` from `WORKER_MODE`, `nojv-worker` for `all`). The collector's Prometheus exporter turns `service.name` into `job`, and Prometheus stores it as `exported_job`; filter or group app series by `exported_job`, never sum them across services.
- The worker awaits `shutdownOtel()` during graceful shutdown; web has no explicit flush and can lose the last interval.
## Metrics
diff --git a/docs/runbooks/testing.md b/docs/runbooks/testing.md
index d553b0c29..bd4183048 100644
--- a/docs/runbooks/testing.md
+++ b/docs/runbooks/testing.md
@@ -60,7 +60,6 @@ CI jobs restore the pnpm store by lockfile hash, falling back to the newest stor
## Service prerequisites
- **Integration**: PostgreSQL, Redis, MinIO and Temporal from Compose, plus `nojv_test`. Temporal test servers download on first use and are cached in `$TMPDIR` for a day; CI pre-downloads them with `scripts/download-temporal-test-servers.sh`.
-- **Test-judge runtime** (`tests/integration/judge/test-judge-runtime.test.ts`, in `sandbox-integration`): skips unless `WASM_OJ_RUNTIME_DIR` points at the forge `wasm-oj-compiler` / `wasm-oj-runner` binaries and `WASM_OJ_TOOLCHAIN_DIR` at a directory whose `node_modules` holds `@wasm-oj/toolchain-clang` and `@wasm-oj/toolchain-python` at the versions in `WASM_OJ_SERVER_IDENTITY`.
- **E2E**: run `pnpm build` first (`tests/tsconfig.e2e.json` resolves built `dist/` packages), the same services, and `nojv_e2e_test`. Playwright starts its own strict-port dev server at `http://localhost:5174`; do not start one. It runs one worker because tests share one destructive database; do not pass `--workers`.
## Destructive test databases
From d1a039634407e6c78d00491e01f2dcae4ef8c0cb Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:16:08 +0800
Subject: [PATCH 15/22] docs: delete the browser Test plan and spec
The work they planned has shipped on this branch, and the decision log
and living docs now hold the result.
Co-Authored-By: Claude Opus 5.5
---
.../plans/2026-10-07-browser-test-judge.md | 225 ------------------
...0-07-browser-test-judge-programs-design.md | 175 --------------
2 files changed, 400 deletions(-)
delete mode 100644 docs/superpowers/plans/2026-10-07-browser-test-judge.md
delete mode 100644 docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
diff --git a/docs/superpowers/plans/2026-10-07-browser-test-judge.md b/docs/superpowers/plans/2026-10-07-browser-test-judge.md
deleted file mode 100644
index b50141356..000000000
--- a/docs/superpowers/plans/2026-10-07-browser-test-judge.md
+++ /dev/null
@@ -1,225 +0,0 @@
-# Browser checker and interactive Test: implementation plan
-
-> **For Claude:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development to implement this plan task by task.
-
-**Goal:** Checker and interactive problems get Test verdicts computed entirely in the browser. The browser compiles and runs the problem's own checker or interactor. The server keeps only a source read endpoint. The server-side half of #641 and the `hidden` workspace visibility go away, with no feature flag, no fallback and no dead code left behind.
-
-**Spec:** `docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md`.
-
-**Owner decisions (2026-10-07):**
-
-- No authoring check and no notice beyond a single "students can read this program" line; authors test their own problems.
-- Interactive ships together with checker. No "not available yet" state.
-- Judge programs compile in the student's browser. No test worker.
-- No `TEST_JUDGE_ENABLED` and no fallback.
-- Endpoint authorization is problem view access.
-- Mobile follows the existing toolchain preload.
-- The "Judged on the server" badge and the busy banner go.
-
-**Upstream gate:** merge only after a `@wasm-oj` release with browser `interact` passing `startupEntropyBytes` (new forge PR), #93 (Python interactors) and #95 (interactive metering). Until then, develop against locally built forge packages. Never patch forge inside NOJV.
-
-**Branch:** `feat/browser-test-judge`, cut from origin/main.
-
----
-
-## Task 0: Forge PR (parallel, outside NOJV)
-
-- Branch from forge `main` in the session scratchpad's forge clone.
-- In `src/runtime/runner.worker.ts` `interactiveCoreProgram`, pass `startupEntropyBytes: prepared.startupEntropyBytes`.
-- Add a browser test that runs `interact` end to end. A C interactor is enough. Use the repo's existing browser test harness, e.g. the strict-CSP suite or the Vitest browser setup.
-- Open the PR from TakalaWang/forge; CODEOWNERS requests JacobLinCool.
-- Comment on #93 and #95 that NOJV now needs them for browser interactive Test.
-
-## Task 1: Remove the server side
-
-Reapply commit `6bdacae9` from `feat/test-execution-only` with `git cherry-pick -n`, then review every hunk. It removes:
-
-- the test worker, its workflows, queue partitions, chart and image layers, plus `@wasm-oj/server` and its patch;
-- the Test API, the Redis lock, the limiter and the application test-judge module;
-- `TEST_JUDGE_ENABLED`;
-- the core request, record and cache-key code, and `build-artifact-wire`;
-- the storage and Redis keys;
-- the worker env vars;
-- the edit page's build status and `checkSamples` action;
-- their tests.
-
-Fix up after the cherry-pick:
-
-- **Restore in core** what the browser needs:
- - `judge/test-judge-verdict.ts` (`checkerCaseVerdict`, `interactiveCaseVerdict`, `truncateUtf8`);
- - `judge/wasm-oj-verdict.ts`;
- - `judge/python-judge-wrappers.ts`;
- - `judgeProgramCompileInput` and `JudgeProgramSource`, without `WASM_OJ_SERVER_*` or the cache key;
- - `interactiveContestantSupported`.
-
- Restore their tests too: `test-judge-verdict`, `wasm-oj-verdict`, `judge-program-sources`, and the compile-input part of `test-judge-program`.
-
-- **Capability:** keep a simplified `staticTestCapability({ isSpecialEnv })`, or inline it if it is only `special_env`. Make sure nothing still takes `testJudgeEnabled` or `judgeLanguage`.
-- **WEB-05 coverage:** move the contest participation and contest-window cases from the deleted `test-judge-domain.test.ts` (lines 527–715 on main) onto `listCodeDrafts`.
-- **`assertProblemContextAllowed`:** keep it unexported unless something outside `code-draft.ts` uses it.
-- **Lockfile:** `pnpm install --frozen-lockfile` must pass with a minimal lock diff.
-- **Renovate:** restore what `.github/renovate.json` said about `@wasm-oj/*` before #641, and drop the rust-image rule.
-- **Messages:** delete the keys only the removed UI used: `admin_checkSamples*`, `admin_checkingSamples`, `admin_judgeProgram*`, `admin_testJudgeDisabled`, `admin_testPythonInteractorUnsupported`, `editor_testJudgeBusy`, `editor_testTooLarge`, `editor_judgingOnServer`, `editor_judgedOnServer`. Grep each key before deleting it.
-
-## Task 2: Judge-program source endpoint
-
-- **Application:** add `getJudgeProgramSource(userId, problemId, context)` next to the problem queries, not in a new test-judge module.
- - Authorization is the same view access the problem page uses for that context.
- - The source is read through the verified script pointer.
- - It returns `{ role, language, source, sha256 }`, or 404 for standard and `special_env` problems.
-- **Web:** add `apps/web/src/routes/api/problems/[id]/judge-program/+server.ts`, using `apiHandler` with the standard limiter. Parse the context the way `/api/drafts` does.
-- Register the route in `tests/unit/security/exam-confinement-api-allowlist.test.ts` as `exam-scoped`.
-- **Tests:** application cases for practice, assignment membership, a running exam, an ended exam or contest where the problem is still viewable, a forbidden context, 404s, and a corrupt pointer. Add an HTTP integration test for the route.
-
-## Task 3: Browser judge-program preparation
-
-- **New `apps/web/src/lib/services/judge-program.ts`:**
- - `prepareJudgeProgram(problemId, context)` fetches the source and builds `judgeProgramCompileInput(source, WASM_OJ_LIBCXX_PCH_HEADER)` with the shared browser engine;
- - the result is memoized per `problemId` + `sha256` for the page session;
- - it exposes progress and returns `{ ok: true, artifact } | { ok: false, diagnostics }`.
- - It preloads the judge program's toolchain(s) alongside the student's: clang for C++, the Python runtime for Python.
- - On each editor mount it refetches the source and rebuilds only if `sha256` changed.
-- **`Editor.svelte` and `EditorActionBar.svelte`:**
- - start preparation in the existing toolchain-preload effect;
- - the Test button shows a preparing label (reuse the toolchain progress style) and waits;
- - a build failure disables Test with "This problem's checker/interactor failed to build", and the panel shows the diagnostics;
- - a fetch failure disables Test with "Couldn't load this problem's checker/interactor" after the existing retry pattern.
-- **Tests:**
- - unit tests for the service with a fake engine and fetch: Python, C++, a failing build, memoization and a `sha256` change;
- - component tests for the button states.
-
-## Task 4: Browser checker Test
-
-- **`browser-local-run.ts`:** add `runBrowserChecker(artifact, { input, answer, output, timeLimitMs })`:
- - args `/judge/input /judge/answer /judge/feedback`;
- - files `/judge/input`, `/judge/answer` and `/judge/feedback/.keep`;
- - stdin = the student's output;
- - output path `/judge/feedback/teammessage.txt`;
- - `validatorTimeoutMs`, 512 MiB;
- - result mapped through `checkerCaseVerdict`.
-- **`use-editor-run.svelte.ts`:**
- - Delete `requestTestJudge` use, the server error codes, `TEST_DISABLING_CODES`, `UNJUDGED_SAMPLE_CODES`, `serialiseBuildArtifact` and `MAX_CASE_STDOUT_BYTES` truncation.
- - Checker path: compile, run the samples, then run the checker on every sample that exited normally. Map run cases to samples by input, as today. Custom cases stay execution-only.
-- **`submission-service.ts`:** delete `requestTestJudge` and `testJudgeErrorCode`.
-- **`EditorBottomPanel.svelte`:** remove the server badge and the `serverNotice` banner. Gate the SE explanation on a browser "judged" flag. Keep the `teammessage` and Executed blocks.
-- **`apps/web/src/lib/types/index.ts`:** drop `serverJudged` and `serverNotice`. The transcript type becomes local if the core schema is gone.
-- **Tests:**
- - `editor-client-test`: replace the server-request cases with browser AC, WA and SE, plus custom execution-only cases;
- - `editor-output-comparison`: drop the server badge and notice cases;
- - add a unit test for `runBrowserChecker` wiring.
-
-## Task 5: Browser interactive Test
-
-- **`browser-local-run.ts`:** add `runBrowserInteraction(contestant, interactor, { interactorInput, limits })` with `engine.interact`. Use the contestant limits and the official interactor limits, and map through `interactiveCaseVerdict`. A wall-limit stop is TLE. Cap the transcript (move `TEST_JUDGE_TRANSCRIPT_BYTES` into web if core no longer needs it).
-- **`use-editor-run.svelte.ts`:** the interactive path compiles the contestant, then runs every selected sample (`interactorInput`) and every custom case through the interactor.
-- **Custom cases on interactive problems:** the case editor's input field is the interactor input. Use the existing interactor-input label and the `customCasesAllowed` switch. Reword `editor_interactiveTestNote` so it no longer says custom cases are impossible.
-- JS/TS contestants stay disabled with the existing reason (`interactiveContestantSupported`).
-- **Tests:**
- - unit tests for the interaction wiring with a fake engine;
- - component tests for the transcript and for interactive custom cases;
- - the real browser check in Task 8.
-
-## Task 6: Judge-tab note
-
-Add one line in the checker and interactor sections of `JudgeTab.svelte`, editors only: "Students can read this program when they press Test." Reword `admin_checkerHelpBody` and `admin_interactorHelpBody` so they no longer imply the program only ever runs in an isolated container. Add a small component test (none exists for `JudgeTab`).
-
-## Task 7: Remove `hidden` workspace visibility
-
-- **Migration:** follow `20260708000000_drop_userstatus_disabled`:
- 1. `UPDATE … SET visibility='readonly' WHERE visibility='hidden'`;
- 2. create the new enum;
- 3. `ALTER COLUMN … USING`;
- 4. rename;
- 5. drop the old enum.
-
- Add the `-- expand-contract-ok:` line for `scripts/check-migrations.mjs`.
-
-- **Schema and docs:** update `problem.prisma` and regenerate `DATABASE.generated.md`.
-- **Core:** `workspaceFileVisibilitySchema` becomes `["editable", "readonly"]`. The judge-snapshot parser maps a legacy `hidden` to `readonly`, and drops its duplicate pointer schema if one still exists.
-- **Application:** drop the content blanking in `details.ts` and the type in `workspace.ts`.
-- **UI:**
- - `StudentProblemView.svelte`: lock prefix, third badge option and hidden placeholder;
- - `editor-bindings.ts`: the filter becomes language-only; rename `publicFiles`;
- - `ReferenceSolutionSection.svelte`: the filter;
- - `WorkspaceFileEditor.svelte`: the option and type;
- - `WorkspaceFileList.svelte`: the fallback icon;
- - `WorkspaceFilesSection.svelte`: the warning;
- - `types/index.ts`.
-- **Messages:** delete `admin_fileHidden`, `admin_workspaceHiddenTestNote`, `workspace_fileHidden*` and `workspace_visibilityHidden`. Reword `admin_workspaceFilesHintMultiFile` and `problemDetail_multiFileHelp`.
-- **Tests:**
- - fixtures in `merge-sandbox-sources`, `db-read-model` (the blanking test goes), `problem-page-data`, `problem-queries`, `editor-workspace-parity`, `workspace-section`, `editor-client-test` and `tests/e2e/course-problem-library.test.ts`;
- - a migration test;
- - a legacy snapshot test.
-
-## Task 8: Docs, decisions, verification
-
-**Decisions:**
-
-- JDG-15 is rewritten per the spec.
-- JDG-26, SEC-15 and OPS-21 are withdrawn. Keep each heading and index line; `27fd8226` has the format.
-- JDG-05 is scoped to official judging.
-- Edit JDG-12, PRB-03, PRB-22, WEB-05 and OPS-18.
-- PRB-01 and PRB-09 drop `hidden`.
-- Every ID token in the README must still match a heading (`tests/unit/docs/doc-links.test.ts`).
-
-**Living docs:**
-
-- `ARCHITECTURE.md`, `JUDGE_PIPELINE.md` (Test sections and the visibility table), `FRONTEND.md`, `REDIS.md`, `DATABASE.md`;
-- `DEPLOYMENT.md`, `RELIABILITY.md`, `SECURITY.md`, `THREAT_MODEL.md`;
-- `QUALITY_SCORE.md`: drop the server-Test items, and add the Wasm official-judging fast path as an item to evaluate;
-- `PRODUCT_SENSE.md`, `docs/features/problem-test.md` (rewrite), `docs/features/contests.md`;
-- runbooks `judge-queue.md` and `getting-started.md`;
-- the READMEs of the worker, chart, storage and temporal, and `.env.example`.
-
-Delete this plan and the spec.
-
-**Verification:**
-
-1. `pnpm ci:verify` (long; run it in the background with a long timeout), `pnpm lint:helm` and `pnpm install --frozen-lockfile`.
-2. Run the touched integration tests against an isolated Postgres and Redis.
-3. **Browser check** on a seeded local stack: the worktree dev server on port 5174 with an isolated database, and forge built locally until the release, then the release. Take screenshots.
- - `any-two-sum` and `shortest-route-plan`: AC and WA, with `teammessage`.
- - A C++ checker fixture, and a broken one: Test disabled, diagnostics shown.
- - `guess-the-number`, `multi-interactive-bisect`, `noisy-oracle-hunt` and `interactive-peak` with C++ and Python contestants: AC, WA, an infinite loop that TLEs at the instruction budget, and a custom interactor input.
- - A JS contestant on an interactive problem: disabled.
- - A multi-file problem: no `hidden` option.
- - A standard problem: unchanged.
- - A student on an ended exam's problem that is still viewable: checker Test works.
-4. A worker image build without the WASM-OJ layers.
-5. Run `git grep` over the dead-code checklist below; expect zero hits.
-
-## Cleanup of earlier attempts
-
-- **Code:** the only test worker that ever landed is #641's `worker-test` Deployment. Task 1 removes it together with its `K8S_RUNTIME_CLASS_NAME` env, service account and RBAC, PDB entry, values and image layers. The gVisor executor attempt of 2026-10-07 was stopped before any commit, and its branch is deleted.
-- **Production:** nothing to undo. On 2026-10-07 no `worker-test`, executor, Service or NetworkPolicy existed in production (read-only `kubectl` check), because #641 was never released.
-- **Local:** after this PR merges, delete the merged or abandoned worktrees and branches:
- - `feat/checker-interactive-test` (#641);
- - `fix/source-map-js-audit` (#642);
- - `feat/test-execution-only` (stopped; its server-removal commit is reapplied in Task 1);
- - `docs/checker-interactive-test-spec`, after switching the main checkout back to `main`.
-
-## After merge
-
-1. Bump to the forge release, then release NOJV.
-2. The seed rows need the manual admin edits #641 listed:
- - checker re-uploads for `any-two-sum`, `course-order` and `shortest-route-plan`;
- - interaction notes, interactor inputs and transcripts for the four interactive problems.
-
- Restate them in the PR body.
-
-3. No backfill is needed: nothing is precompiled.
-
----
-
-## Dead-code checklist (`git grep` must find nothing outside history)
-
-| Area | Must be gone |
-| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Worker | `WORKER_MODE.*test`, `nojv-worker-test`, `TEST_JUDGE_SLOTS`, `WASM_OJ_RUNTIME_DIR`, `WASM_OJ_TOOLCHAIN_DIR`, `WASM_OJ_CACHE_DIR`, `@wasm-oj/server`, `@wasm-oj__server` |
-| Temporal | `test-judge` queue, `testJudgeWorkflow`, `testJudgeProgramBuildWorkflow`, `runTestJudgeWorkflow`, `dispatchTestJudgeProgramBuild`, `TEST_JUDGE_TASK_QUEUE` |
-| Application | `runTestJudge`, `buildTestJudgeProgram`, `withUserTestJudgeLock`, `checkSamplesWithChecker`, `getJudgeProgramStatus`, `isTestJudgeEnabled`, `TEST_JUDGE_ENABLED` |
-| Storage, Redis, limiter | `test-judge-requests/`, `test-judge-programs/`, `testJudgeRequestKey`, `testJudgeProgramKey`, `testJudgeInFlight`, `rl:test-judge`, `testJudgeApiHandler` |
-| Core | `testJudgeRequestSchema`, `testJudgeResponseSchema`, `testJudgeStoredRequestSchema`, `storedJudgeProgramSchema`, `testJudgeProgramCacheKey`, `WASM_OJ_SERVER_IDENTITY`, `serialiseBuildArtifact`, `boundedTestJudgeOutput` |
-| Web | `requestTestJudge`, `serverJudged`, `serverNotice`, `JudgeProgramTestStatus`, `checkSamples`, `client_test_judge_program`, `editor_testUnavailableForProblem` used for checker or interactive problems |
-| Message keys | `admin_checkSamples*`, `admin_judgeProgram*`, `admin_testJudgeDisabled`, `editor_testJudgeBusy`, `editor_testTooLarge`, `editor_judgingOnServer`, `editor_judgedOnServer`, `admin_fileHidden`, `admin_workspaceHiddenTestNote`, `workspace_fileHidden*`, `workspace_visibilityHidden` |
-| Infra | `worker-test`, `test-executor`, `TEST_JUDGE_EXECUTOR`, `worker-test.deployment.yaml`, `infra/docker/wasm-oj-toolchains`, the WASM-OJ stages in `worker.Dockerfile`, `"hidden"` as a workspace visibility |
diff --git a/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md b/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
deleted file mode 100644
index b38493c60..000000000
--- a/docs/superpowers/specs/2026-10-07-browser-test-judge-programs-design.md
+++ /dev/null
@@ -1,175 +0,0 @@
-# Browser Test runs checkers and interactors
-
-**Status:** Direction decided by the owner on 2026-10-07, after the final review round that day · **Touches:** JDG-03, JDG-05, JDG-12, JDG-15, JDG-26, OPS-18, OPS-21, PRB-01, PRB-03, PRB-09, PRB-22, SEC-12, SEC-15, WEB-05
-
-## Problem
-
-#641 (merged 2026-10-06, not released) added Test for checker and interactive problems by running the problem's checker or interactor on a server test worker. That worker runs TA-authored judge programs, plus the student's Wasm on interactive problems. The only thing between that code and the container's object-storage keys (every problem's hidden testcases) and Redis URL is the Wasm runtime.
-
-The owner's direction:
-
-- Test is the student's own run, so all of it happens in the student's browser: the judge program is compiled and run there too.
-- Judge programs become readable by students. Keeping them robust is the authors' responsibility, and authors test their own problems.
-- For Test, the server keeps nothing but a read endpoint for the judge program's source.
-- The `hidden` workspace visibility goes away; it never gave confidentiality.
-- Checker and interactive problems ship together, with no feature flag and no fallback.
-
-## Goals
-
-- Test on checker and interactive problems shows verdicts (AC/WA/TLE/RE/SE), the checker's `teammessage` and the interaction transcript, all computed in the browser.
-- The server executes and compiles nothing for Test. The test worker, its queue, build cache, Test API and the worker image's WASM-OJ layers are removed.
-- The judge program is prepared in the background when the editor opens, so pressing Test does not wait on it.
-- Hidden testcase data never leaves the server. Test uses only sample data, which the statement already makes public.
-- Multi-file problems keep only `editable` and `readonly` files.
-
-## Non-goals
-
-- Official judging is unchanged: Submit, verdicts and the sandbox, and official judging still compiles the judge program from source in its per-stage gVisor Pod.
-- No authoring-time check or build status for judge programs; authors press Test on their own problem.
-- `special_env` (Advanced) still has no Test.
-- No per-problem toggle for what students can read.
-- No multi-file precompile. The student's editable files change on every Test, and forge's in-browser object cache already reuses unchanged files.
-- A Wasm fast path for official judging is a separate future item (Quality Ledger).
-
-## What becomes public, and the accepted risk
-
-| Data | Test sends it to the browser |
-| -------------------------------------- | ------------------------------ |
-| Checker / interactor source | yes |
-| Sample input, output, interactor input | yes (already in the statement) |
-| `readonly` workspace files | yes (already today) |
-| Hidden testcases (input or answer) | **never** (SEC-12) |
-
-The accepted risk: bugs in a judge program, such as a missing validity check or an off-by-one query limit, become easier to find. Those bugs exist whether or not the source is public, and authors own them. The judge tab states once that students can read the program.
-
-The docs record safe authoring practice:
-
-- A checker only verifies, and reads the optimum from `judge_answer` (the testcase's expected-output field), as the seed checkers do.
-- An interactor reads its secret from `judge_input`.
-
-## Design
-
-### Judge program delivery
-
-`GET /api/problems/[id]/judge-program?context=…` returns `{ role, language, source, sha256 }` for checker and interactive problems, and 404 otherwise.
-
-- The source is read through the existing verified script pointer.
-- **Authorization:** anyone who may view the problem in that context. Checker Test therefore keeps working after an exam or contest ends, like standard Test.
-- The route is classified `exam-scoped` in the exam-confinement allowlist and uses the standard API rate limiter.
-
-### Preparing the judge program in the browser
-
-When the editor opens on a checker or interactive problem, it:
-
-1. fetches the source;
-2. builds it with `@wasm-oj/browser`, using core's `judgeProgramCompileInput` (the DOMjudge Python wrapper, or the C++ `bits/stdc++.h` shim with the PCH-only-when-included rule):
- - **Python** is packaged into a runtime bundle at once, with no compile;
- - **C++** compiles in the background;
-3. preloads whatever toolchains the judge program needs, next to the student's own: clang for a C++ judge program, the Python runtime for a Python one.
-
-The built program stays in memory for the page session, keyed by `sha256`. It is rebuilt when the editor reopens with different source.
-
-Test waits for it, as it already waits for the student's toolchain, and the button shows that it is preparing. If the judge program fails to build, Test is disabled with a reason and the panel shows the compiler diagnostics. The source is public, so its diagnostics are too.
-
-### Checker problems
-
-1. Compile the student's program and run each selected sample with `stdin = sample.input`, as today.
-2. For each sample whose run exited normally, run the checker with:
- - args `/judge/input /judge/answer /judge/feedback`;
- - files: input = `sample.input`, answer = `sample.output`;
- - stdin = the student's stdout;
- - output path `/judge/feedback/teammessage.txt`;
- - official judging's `validatorTimeoutMs` and 512 MiB.
-3. Map the result with `checkerCaseVerdict`: exit 42 is AC, 43 is WA, anything else is SE. Show `teammessage`.
-4. Custom cases stay execution-only, because they have no `judge_answer` in the author's format.
-
-### Interactive problems
-
-- `engine.interact(contestant, interactor, …)` from `@wasm-oj/browser` runs both programs. The interactor gets `sample.interactorInput` as `/judge/input`.
-- Map the result with `interactiveCaseVerdict`, and show the transcript, contestant stderr and time. A contestant stopped by its wall limit is TLE.
-- Custom cases take an interactor input, which the student types as the secret, and the real interactor judges them.
-- JS/TS contestants stay unavailable until upstream streams QuickJS stdin.
-
-### Upstream dependency
-
-This PR merges only after a `@wasm-oj` release that contains:
-
-| Change | Status | Why |
-| ----------------------------------------------- | ---------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| Browser `interact` passes `startupEntropyBytes` | new PR | `runner.worker.ts` `interactiveCoreProgram` omits the field, but runtime-core requires it (`startup_entropy_bytes: u64`, no serde default), so every browser `interact` fails while decoding its request |
-| Python (runtime-bundle) interactors | wasm-oj/forge#93 | all four interactive problems use Python interactors |
-| In-module interactive metering | wasm-oj/forge#95 | a CPU-bound contestant must stop at its instruction budget instead of keeping the student's tab busy until the wall limit |
-
-Until that release, NOJV develops against locally built packages. It then pins `@wasm-oj/browser` and the toolchains to the release. Forge must not be patched inside NOJV (JDG-15).
-
-### Removing `hidden` workspace visibility
-
-- On 2026-10-07 production had 9 `editable` files, 9 `readonly` files and **0 `hidden`** files (read-only query).
-- A migration sets any `hidden` rows to `readonly`, then recreates the `WorkspaceFileVisibility` enum without `hidden`.
-- Stored judge snapshots map a legacy `hidden` to `readonly`, so old submissions still rejudge.
-- The UI, application, seeds and tests drop it.
-- PRB-01 records the rejection: `hidden` gave no confidentiality, nobody used it, and it misled authors.
-
-### Removed from #641
-
-- The whole server side:
- - the `WORKER_MODE=test` worker and its Deployment;
- - the `test-judge` queue and its partitions;
- - the Test and build workflows, and the build-after-save dispatch;
- - the `test-judge-programs/` cache and the Test API;
- - the Redis lock and the `rl:test-judge` limiter;
- - `TEST_JUDGE_ENABLED`, `WASM_OJ_*` and `TEST_JUDGE_SLOTS`;
- - `@wasm-oj/server` and its EPIPE patch;
- - the worker image's WASM-OJ layers, `infra/docker/wasm-oj-toolchains/` and the related Renovate pins.
-- In core: the request, response and record schemas, the cache key and server identity, and `build-artifact-wire`.
-- On the edit page: the judge-program build status and the "check samples with the checker" action.
-
-### Kept from #641
-
-- `interactorInput` and `interactionFormat`.
-- The seed fixes.
-- "Executed" for custom cases without expected output.
-- The contest code-draft authorization fix (WEB-05).
-- The PCH-only-with-`` rule.
-- In core: the verdict helpers, Python wrappers, C++ shim and `judgeProgramCompileInput`, now used by the browser.
-
-## Decision log changes
-
-- **JDG-15:** rewritten as "Test runs entirely in the browser, judge programs included".
- - Rejected:
- - running judge programs on the server (#641, withdrawn before release);
- - compiling them on the server (that keeps a server WASM-OJ runtime and worker for a preview feature);
- - execution-only Test;
- - exposure toggles.
- - Rules:
- - what is public, and that authors own their judge programs;
- - the server neither compiles nor executes anything for Test;
- - Test never receives non-sample testcase data.
-- **JDG-26, SEC-15 and OPS-21:** withdrawn, each with its reason and a pointer to JDG-15. IDs are never reused.
-- **JDG-05:** the "validators never seen" rule is scoped to official judging.
-- **JDG-12, PRB-03, PRB-22 and WEB-05:** the server-Test wording goes.
-- **PRB-01 and PRB-09:** `hidden` goes; PRB-01 gets a Rejected line.
-- **OPS-18:** the wording #641 added goes.
-
-## Testing
-
-- **Unit:**
- - the browser checker and interactor runs with a fake engine (args, files, verdict mapping);
- - judge-program preparation (Python packaging, C++ compile, failure);
- - Test capability per problem type;
- - endpoint authorization (exam, contest window, assignment membership, practice);
- - the visibility migration;
- - legacy-`hidden` snapshot parsing.
-- **Component:**
- - Test button states (preparing, build failed, JS/TS on interactive problems);
- - the checker result panel with `teammessage`;
- - the transcript panel;
- - interactive custom cases;
- - the judge-tab note.
-- **Browser check on seeds:**
- - `any-two-sum` and `shortest-route-plan`: AC and WA;
- - a C++ checker fixture, and a broken one that disables Test;
- - `guess-the-number` and the other three interactive problems: AC, WA, and TLE from an infinite loop;
- - a multi-file problem showing no `hidden` option;
- - standard problems unchanged.
-- `pnpm ci:verify`, `pnpm lint:helm`, and a worker image build without the WASM-OJ layers.
From 0650a5ba100fd9673cad2821845f26dfae51afb7 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:21:54 +0800
Subject: [PATCH 16/22] fix(web): let a contestant limit stop decide
interactive Test
Official judging keeps draining the interactor's output after the
contestant ends, so when the contestant hits a time or memory limit the
interactor reads EOF, exits 43, and the case is TLE or MLE. The browser
engine closes a stopped side's pipes instead, so a Python interactor that
writes after that exits 120 and core's interactiveCaseVerdict gave SE
where Submit gives TLE or MLE; a deadlock reaching the shared wall stop
did the same.
interactiveCaseVerdict now returns TLE or MLE first when the contestant
was stopped by logical time, the instruction budget, the wall stop or the
memory limit. Every other case keeps official judging's order: an
interactor failure is SE, then a contestant RE, then the interactor's AC
or WA.
JDG-15 records the rule and why, the Judge Pipeline and Problem Test spec
describe it, and the Quality Ledger item now covers only what is left: a
contestant that exits before the interactor's next write.
Co-Authored-By: Claude Opus 5.5
---
docs/architecture/JUDGE_PIPELINE.md | 16 ++++++------
docs/decisions/judge.md | 2 ++
docs/features/problem-test.md | 2 +-
docs/operations/QUALITY_SCORE.md | 2 +-
packages/core/src/judge/test-judge-verdict.ts | 5 ++--
tests/unit/core/test-judge-verdict.test.ts | 25 ++++++++++++++++---
.../unit/web/browser-local-execution.test.ts | 12 ++++++++-
7 files changed, 47 insertions(+), 17 deletions(-)
diff --git a/docs/architecture/JUDGE_PIPELINE.md b/docs/architecture/JUDGE_PIPELINE.md
index 9b11a0d89..1d4e77900 100644
--- a/docs/architecture/JUDGE_PIPELINE.md
+++ b/docs/architecture/JUDGE_PIPELINE.md
@@ -576,17 +576,17 @@ While it prepares, Test stays clickable and shows the toolchain download or
`interact(contestant, interactor)` once per case. The interactor gets the checker's
args, the case's input as `/judge/input`, an empty `/judge/answer` and an empty
`/judge/feedback/`.
-3. `interactiveCaseVerdict` merges as official judging does: an interactor that exits
- with anything but 42 or 43, or is stopped by a limit, is SE; otherwise a contestant
- TLE, MLE or RE wins; otherwise the interactor's AC or WA stands.
+3. `interactiveCaseVerdict` merges each case: a contestant stopped by a time limit
+ (logical time, instruction budget or the wall stop) or by the memory limit is TLE or
+ MLE, whatever the interactor did; otherwise an interactor that exits with anything
+ but 42 or 43, or is stopped by a limit, is SE; otherwise a contestant RE wins;
+ otherwise the interactor's AC or WA stands.
4. Each case shows the verdict, contestant stderr (100,000 bytes), both transcript
directions (64 KiB each, cut at a UTF-8 boundary) and the contestant's logical
time. The interactor's `teammessage` and stderr are not shown.
-5. The engine closes the pipes of a side that stops, while official judging keeps
- reading the interactor's output after the contestant ends and only gives it EOF. A
- Python interactor that writes after the contestant stopped at a limit therefore
- exits 120, and the case is SE in Test where Submit gives TLE or MLE; so is a case
- where both sides reach the shared wall stop.
+5. When the contestant exits, normally or with an error, before the interactor's next
+ write, that write fails (a Python interactor exits 120) and the case is SE; Submit
+ gives that case RE or the interactor's verdict.
### Limits
diff --git a/docs/decisions/judge.md b/docs/decisions/judge.md
index 87df95c86..7dc9e7810 100644
--- a/docs/decisions/judge.md
+++ b/docs/decisions/judge.md
@@ -188,6 +188,8 @@ Test never creates a submission and runs entirely in the student's browser. The
- Rule: generic runtime fixes land in wasm-oj/forge and are consumed as pinned releases; NOJV never forks or patches forge. Keep document CSP at `wasm-unsafe-eval`. JavaScript and TypeScript contestants on interactive problems stay unavailable until upstream streams QuickJS stdin.
- Rule: browser results are previews on Forge logical time; do not claim resource or toolchain-version equivalence; fix sample data, not engine input.
- Rule: verdicts map as in official judging through core's `checkerCaseVerdict` and `interactiveCaseVerdict`, the one place for that rule; Test shows `teammessage`, the transcript and contestant stderr, never `judgemessage` or interactor stderr.
+- Rule: in interactive Test a contestant stopped by a time limit (logical time, instruction budget or the wall stop) or the memory limit is TLE or MLE even when the interactor then fails. The browser engine closes a stopped side's pipes, while official judging keeps draining the interactor's output and gives it EOF, so an interactor that writes after a limit stop fails only in Test and Submit gives TLE or MLE.
+- Rejected: letting an interactor failure win over a contestant limit stop in Test, as official judging's merge order does; Test showed SE where Submit gives TLE or MLE.
- Rule: source diagnostics are CE, toolchain/infrastructure faults SE. Custom cases without expected output are execution-only; on a checker problem only a case whose input is a sample's input is checked, with that sample's output as the answer, while an interactive custom case supplies the interactor's input and is judged.
- Rule: Test is exempt from the submit cooldown (PRB-22).
- Rule: the editor preloads the selected language's toolchain through `prefetchBrowserToolchain` (Worker URLs, so the HTTP cache serves the build) and, on checker and interactive problems, prepares the judge program, and Test waits for both; download failures are reported without building and never point students to Submit. A first build otherwise spends its 60 s boundary downloading on slow exam networks and fails the same way on every retry. A judge program that fails to build disables Test and shows its diagnostics.
diff --git a/docs/features/problem-test.md b/docs/features/problem-test.md
index bdf9368d9..fc9c58f90 100644
--- a/docs/features/problem-test.md
+++ b/docs/features/problem-test.md
@@ -56,7 +56,7 @@ The endpoint requires a session and the problem page's view access for the conte
- Given an interactive problem, the case panel lists each sample's interactor input and shows the problem's interaction notes; a sample without an interactor input is left out.
- Given the student adds a case, its input is an interactor input (for example the hidden number), and the real interactor judges it like a sample. The case panel says "…Cases you add are judged the same way."
- When the student presses Test, the browser compiles the program and runs it against the interactor once per case. Each case shows its verdict, the student's stderr, the time, and a transcript with "From the interactor" and "From your program", each cut at 64 KiB.
-- Given the interactor fails (an exit code other than 42 or 43, or a limit), the case is SE; otherwise the student's TLE, MLE or RE wins; otherwise the interactor's AC or WA stands, as in official judging.
+- Given the student's program is stopped by a time or memory limit, the case is TLE or MLE, whatever the interactor does; otherwise, given the interactor fails (an exit code other than 42 or 43, or a limit), the case is SE; otherwise the student's RE wins; otherwise the interactor's AC or WA stands.
- Given JavaScript or TypeScript is selected, Test is disabled with "Test can't run interactive problems in JavaScript or TypeScript yet…"; switching to another language enables it.
### Problem page and authoring
diff --git a/docs/operations/QUALITY_SCORE.md b/docs/operations/QUALITY_SCORE.md
index 73dcbe834..a470fdc2a 100644
--- a/docs/operations/QUALITY_SCORE.md
+++ b/docs/operations/QUALITY_SCORE.md
@@ -68,7 +68,7 @@ Work that is known, not done, and not covered by an in-flight plan. Remove an it
- The admin audit log's actor, action and target columns have no filters; adding them needs server-side filtering in `adminAuditLogRepo.listPaged` (UI-12).
- Browser Test (WASM-OJ) deferred scope: official Submit from the browser, Advanced problems and limit calibration stay server-only until decided otherwise (JDG-15).
- Browser checker and interactive Test need one `@wasm-oj` release that ships wasm-oj/forge#93 (Python interactors in `interact`), #95 (in-module metering of interactive programs), #96 (runtime-file export stall) and #99 (interactive sides in nested Workers); bump the pins to it before release (JDG-15). JavaScript and TypeScript contestants on interactive problems wait for streaming QuickJS stdin upstream.
-- The browser engine closes a stopped side's pipes, while official judging keeps reading the interactor's output after the contestant ends and only gives it EOF. An interactor that writes after the contestant stopped at a limit therefore fails in Test (a CPython interactor exits 120) and the case is SE where Submit gives TLE or MLE; a deadlock that reaches the shared wall stop is SE too. A Python interactor that starts after a fast-failing contestant has stopped hits this on its first write. Fix upstream by discarding a finished contestant's input instead of failing the write.
+- The browser engine closes a stopped side's pipes, while official judging keeps reading the interactor's output after the contestant ends and only gives it EOF. When the contestant exits, normally or with an error, before the interactor's next write, that write fails in Test (a CPython interactor exits 120) and the case is SE where Submit gives RE or the interactor's verdict. A Python interactor that starts after a fast-failing contestant has stopped hits this on its first write. A contestant stopped by a time or memory limit is already TLE or MLE in Test whatever the interactor does (JDG-15). Fix upstream by discarding a finished contestant's input instead of failing the write.
- After such a broken pipe the browser transcript can repeat the line the interactor failed to write (one CPython line showed five times). Raise upstream with the fix above.
- In WebKit an empty `for(;;);` loop is not stopped by the instruction meter and runs to the wall stop, in `run` and `interact` alike; a loop with a side effect stops at its budget. Raise upstream.
- Every official checker or interactive stage recompiles its judge program. Precompiling native judge programs once per source, language and sandbox image is open; measure the C++ per-stage compile cost first (on 2026-10-04 every production judge program was Python, which needs no compile).
diff --git a/packages/core/src/judge/test-judge-verdict.ts b/packages/core/src/judge/test-judge-verdict.ts
index a6f075bf7..fd7a746b2 100644
--- a/packages/core/src/judge/test-judge-verdict.ts
+++ b/packages/core/src/judge/test-judge-verdict.ts
@@ -40,11 +40,12 @@ export function interactiveCaseVerdict(
result: { contestant: ProcessTermination; interactor: ProcessTermination },
teamMessage?: string,
): { verdict: TestJudgeVerdict; teamMessage?: string } {
- const interactor = checkerCaseVerdict(result.interactor, teamMessage);
- if (interactor.verdict === "SE") return interactor;
const contestant = wasmOjTerminationVerdict(
result.contestant.termination,
result.contestant.code,
);
+ if (contestant === "TLE" || contestant === "MLE") return { verdict: contestant };
+ const interactor = checkerCaseVerdict(result.interactor, teamMessage);
+ if (interactor.verdict === "SE") return interactor;
return contestant === "AC" ? interactor : { verdict: contestant };
}
diff --git a/tests/unit/core/test-judge-verdict.test.ts b/tests/unit/core/test-judge-verdict.test.ts
index d19306cd3..7c7b7f24c 100644
--- a/tests/unit/core/test-judge-verdict.test.ts
+++ b/tests/unit/core/test-judge-verdict.test.ts
@@ -107,17 +107,34 @@ describe("interactiveCaseVerdict", () => {
).toEqual({ verdict: "TLE" });
});
- it.each(["memory-limit", "instruction-limit", "wall-time-limit"])(
- "lets an interactor failure win over a contestant %s, as official judging does",
- (termination) => {
+ it.each([
+ ["logical-time-limit", "TLE"],
+ ["instruction-limit", "TLE"],
+ ["wall-time-limit", "TLE"],
+ ["memory-limit", "MLE"],
+ ])(
+ "keeps a contestant %s when the interactor then dies on the closed pipe",
+ (termination, verdict) => {
expect(
interactiveCaseVerdict({
contestant: { termination, code: 0 },
interactor: exited(120),
}),
- ).toEqual({ verdict: "SE" });
+ ).toEqual({ verdict });
},
);
+
+ it("maps an interactor crash to SE when the contestant exited normally", () => {
+ expect(
+ interactiveCaseVerdict({ contestant: exited(0), interactor: exited(120) }, "partial"),
+ ).toEqual({ verdict: "SE" });
+ });
+
+ it("maps an interactor crash to SE when the contestant crashed", () => {
+ expect(interactiveCaseVerdict({ contestant: exited(1), interactor: exited(120) })).toEqual({
+ verdict: "SE",
+ });
+ });
});
describe("truncateUtf8", () => {
diff --git a/tests/unit/web/browser-local-execution.test.ts b/tests/unit/web/browser-local-execution.test.ts
index 2fc7ec8df..dab51697b 100644
--- a/tests/unit/web/browser-local-execution.test.ts
+++ b/tests/unit/web/browser-local-execution.test.ts
@@ -596,6 +596,16 @@ it.each([
contestant: side("instruction-limit", 137),
interactor: side("exited", 120),
}),
+ "TLE",
+ ],
+ [
+ "the contestant exceeds its memory and the interactor dies on the closed pipe",
+ interaction({ contestant: side("memory-limit", 0), interactor: side("exited", 120) }),
+ "MLE",
+ ],
+ [
+ "the contestant exits normally and the interactor dies on the closed pipe",
+ interaction({ interactor: side("exited", 120) }),
"SE",
],
[
@@ -609,7 +619,7 @@ it.each([
contestant: side("wall-time-limit", 0),
interactor: side("wall-time-limit", 0),
}),
- "SE",
+ "TLE",
],
])("maps an interaction where %s", async (_label, run, verdict) => {
engine.interact.mockResolvedValueOnce(run);
From 8a5d84a5c8fe4483ef1a2ed1007cdd7f83895986 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:25:15 +0800
Subject: [PATCH 17/22] feat(web): tell authors students can read their checker
or interactor
Test runs the checker or interactor in the student's browser, so the
judge tab now says "Students can read this program when they press Test."
under the checker and the interactor language. The tab is only on the
edit page, so only editors see it.
The checker and interactor help no longer say the program only runs in
an isolated container: it runs in the judge sandbox on Submit and in the
student's browser on Test. The interactor help and the examples drop
set_score and score.txt. Neither Python wrapper defines set_score and no
judge path reads score.txt (JDG-03 removed partial credit), so the Python
interactor example died with a NameError on every correct guess.
A component test covers the note on checker and interactive problems and
its absence on standard ones.
Co-Authored-By: Claude Opus 5.5
---
apps/web/messages/en.json | 5 +-
apps/web/messages/zh-TW.json | 5 +-
.../features/problem/tabs/JudgeTab.svelte | 4 ++
.../problem/tabs/judge/script-examples.ts | 15 ++---
docs/features/problem-test.md | 2 +-
tests/component/web/judge-tab.test.ts | 55 +++++++++++++++++++
6 files changed, 72 insertions(+), 14 deletions(-)
create mode 100644 tests/component/web/judge-tab.test.ts
diff --git a/apps/web/messages/en.json b/apps/web/messages/en.json
index 47370cbc2..a3e9a3ed6 100644
--- a/apps/web/messages/en.json
+++ b/apps/web/messages/en.json
@@ -270,7 +270,7 @@
"admin_announcementsUnpin": "Unpin",
"admin_announcementsUnpublish": "Unpublish",
"admin_cancel": "Cancel",
- "admin_checkerHelpBody": "Your validator follows the DOMjudge output-validator interface and runs in its own isolated container — the student's program never sees the answer. It is invoked as `validator ` with the student output on stdin. It renders an accept/reject verdict only (no partial scoring). Python: `judge_input`, `judge_answer`, `team_output` are pre-bound as strings, and helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the right feedback files and exit. C++: read `argv[1]` / `argv[2]` and stdin yourself, write `teammessage.txt` into the `argv[3]` feedback dir, and exit 42 (accept) or 43 (wrong) — no judge header required. Feedback is shown to the student. Validator timeout is 30 seconds; a crash or hang becomes a system error and does not count against the student.",
+ "admin_checkerHelpBody": "Your validator follows the DOMjudge output-validator interface. On Submit it runs in the judge sandbox, apart from the student's program, which never sees the answer; on Test it runs in the student's browser and checks the samples. It is invoked as `validator ` with the student output on stdin. It renders an accept/reject verdict only (no partial scoring). Python: `judge_input`, `judge_answer`, `team_output` are pre-bound as strings, and helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the right feedback files and exit. C++: read `argv[1]` / `argv[2]` and stdin yourself, write `teammessage.txt` into the `argv[3]` feedback dir, and exit 42 (accept) or 43 (wrong) — no judge header required. Feedback is shown to the student. Validator timeout is 30 seconds; a crash or hang becomes a system error and does not count against the student.",
"admin_checkerHelpTitle": "How to write a checker",
"admin_compareCaseSensitive": "Case-sensitive comparison",
"admin_compareFloatTolerance": "Float tolerance (ε)",
@@ -311,11 +311,12 @@
"admin_inputFormatTooltip": "Describe the input format. Supports Markdown.",
"admin_interactionFormat": "Interaction notes",
"admin_interactionFormatTooltip": "Describe what the interactor reads from its input file and how it talks to the student's program. Shown to students. Supports Markdown.",
- "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface and runs in its own isolated container, wired to the student program by a byte proxy: student stdout flows to interactor stdin, interactor stdout flows to student stdin. It is invoked as `interactor `. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). Helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `set_score(x)`, `judge_log(msg)` write the feedback files and exit 42/43. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `score.txt` / `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
+ "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface. On Submit it runs in the judge sandbox, and on Test in the student's browser; either way student stdout flows to interactor stdin, and interactor stdout flows to student stdin. It is invoked as `interactor `. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). Helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the feedback files and exit 42/43. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
"admin_interactorHelpTitle": "How to write an interactor",
"admin_interactorLanguage": "Interactor language",
"admin_judgeChecker": "Checker script",
"admin_judgeInteractive": "Interactive",
+ "admin_judgeProgramReadableNote": "Students can read this program when they press Test.",
"admin_judgeStandard": "Standard (stdin/stdout diff)",
"admin_judgeType": "Judge type",
"admin_judgeTypeHint": "How each testcase is evaluated.",
diff --git a/apps/web/messages/zh-TW.json b/apps/web/messages/zh-TW.json
index df287dc75..8077c6054 100644
--- a/apps/web/messages/zh-TW.json
+++ b/apps/web/messages/zh-TW.json
@@ -270,7 +270,7 @@
"admin_announcementsUnpin": "取消置頂",
"admin_announcementsUnpublish": "取消發佈",
"admin_cancel": "取消",
- "admin_checkerHelpBody": "你的 validator 採用 DOMjudge 輸出驗證器介面,在獨立隔離容器中執行——學生程式永遠看不到答案。呼叫方式為 `validator `,學生輸出由 stdin 傳入。只輸出接受/錯誤判決(不做部分計分)。Python:`judge_input`、`judge_answer`、`team_output` 已預先注入為字串,輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入對應回饋檔並結束。C++:自行讀取 `argv[1]` / `argv[2]` 與 stdin,把 `teammessage.txt` 寫進 `argv[3]` 的 feedback 目錄,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。回饋訊息會顯示給學生。Validator 超時上限 30 秒;掛掉或逾時記為系統錯誤,不影響學生判決。",
+ "admin_checkerHelpBody": "你的 validator 採用 DOMjudge 輸出驗證器介面。提交時在判題沙盒中執行,與學生程式分開,學生程式看不到答案;按「測試」時則在學生的瀏覽器中執行,檢查範例。呼叫方式為 `validator `,學生輸出由 stdin 傳入。只輸出接受/錯誤判決(不做部分計分)。Python:`judge_input`、`judge_answer`、`team_output` 已預先注入為字串,輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入對應回饋檔並結束。C++:自行讀取 `argv[1]` / `argv[2]` 與 stdin,把 `teammessage.txt` 寫進 `argv[3]` 的 feedback 目錄,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。回饋訊息會顯示給學生。Validator 超時上限 30 秒;掛掉或逾時記為系統錯誤,不影響學生判決。",
"admin_checkerHelpTitle": "如何撰寫 Checker",
"admin_compareCaseSensitive": "區分大小寫比對",
"admin_compareFloatTolerance": "浮點容差(ε)",
@@ -311,11 +311,12 @@
"admin_inputFormatTooltip": "描述輸入資料的格式,支援 Markdown。",
"admin_interactionFormat": "互動說明",
"admin_interactionFormatTooltip": "說明 interactor 會從輸入檔讀到什麼,以及它如何與學生程式互動。會顯示給學生,支援 Markdown。",
- "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面,在獨立隔離容器中執行,並由 worker 的位元組代理與學生程式相接:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`set_score(x)`、`judge_log(msg)` 會寫入回饋檔並以 exit 42/43 結束。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `score.txt` / `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
+ "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面。提交時在判題沙盒中執行,按「測試」時則在學生的瀏覽器中執行;兩種情況都一樣:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入回饋檔並以 exit 42/43 結束。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
"admin_interactorHelpTitle": "如何撰寫 Interactor",
"admin_interactorLanguage": "Interactor 語言",
"admin_judgeChecker": "Checker 腳本",
"admin_judgeInteractive": "互動題",
+ "admin_judgeProgramReadableNote": "學生按下「測試」時可以讀到這支程式。",
"admin_judgeStandard": "標準(stdin/stdout 比對)",
"admin_judgeType": "評測類型",
"admin_judgeTypeHint": "每筆測資的評測方式。",
diff --git a/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte b/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
index 65858f352..d66f584c9 100644
--- a/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
+++ b/apps/web/src/lib/components/features/problem/tabs/JudgeTab.svelte
@@ -231,6 +231,8 @@
+
+
{m.admin_interactorHelpTitle()}
diff --git a/apps/web/src/lib/components/features/problem/tabs/judge/script-examples.ts b/apps/web/src/lib/components/features/problem/tabs/judge/script-examples.ts
index 43be2123d..de18a457f 100644
--- a/apps/web/src/lib/components/features/problem/tabs/judge/script-examples.ts
+++ b/apps/web/src/lib/components/features/problem/tabs/judge/script-examples.ts
@@ -1,5 +1,5 @@
export const PYTHON_CHECKER_EXAMPLE = `# Bound: judge_input, judge_answer, team_output (strings)
-# Helpers: accept(team_msg=""), wrong(team_msg=""), set_score(x), judge_log(msg)
+# Helpers: accept(team_msg=""), wrong(team_msg=""), judge_log(msg)
u = team_output.split()
e = judge_answer.split()
@@ -12,13 +12,12 @@ accept()
`;
export const PYTHON_INTERACTOR_EXAMPLE = `# Bound: judge_input, judge_answer (strings); read(), write(msg)
-# Helpers: accept(team_msg=""), wrong(team_msg=""), set_score(x), judge_log(msg)
+# Helpers: accept(team_msg=""), wrong(team_msg=""), judge_log(msg)
secret = int(judge_input.strip())
for i in range(1, 21):
g = int(read())
if g == secret:
- set_score(max(0, 100 - (i - 1) * 5))
accept(f"correct in {i} guesses")
write("higher" if secret > g else "lower")
wrong("exceeded 20 guesses")
@@ -62,8 +61,7 @@ int main(int argc, char* argv[]) {
std::ifstream in(argv[1]);
std::string feedback_dir = argv[3];
- auto finish = [&](int code, int score, const std::string& msg) {
- std::ofstream(feedback_dir + "/score.txt") << score;
+ auto finish = [&](int code, const std::string& msg) {
std::ofstream(feedback_dir + "/teammessage.txt") << msg;
std::exit(code);
};
@@ -72,12 +70,11 @@ int main(int argc, char* argv[]) {
in >> secret;
for (int guess_count = 1; guess_count <= 20; guess_count++) {
int g;
- if (!(std::cin >> g)) finish(43, 0, "solution closed its output early");
+ if (!(std::cin >> g)) finish(43, "solution closed its output early");
if (g == secret)
- finish(42, std::max(0, 100 - (guess_count - 1) * 5),
- "correct in " + std::to_string(guess_count) + " guesses");
+ finish(42, "correct in " + std::to_string(guess_count) + " guesses");
std::cout << (secret > g ? "higher" : "lower") << std::endl;
}
- finish(43, 0, "exceeded 20 guesses");
+ finish(43, "exceeded 20 guesses");
}
`;
diff --git a/docs/features/problem-test.md b/docs/features/problem-test.md
index fc9c58f90..1c7aad60d 100644
--- a/docs/features/problem-test.md
+++ b/docs/features/problem-test.md
@@ -64,4 +64,4 @@ The endpoint requires a session and the problem page's view access for the conte
- On an interactive problem the statement shows the "Interaction" notes, and each sample shows its interactor input with a copy button, above the transcript-style input and output.
- The basic info section shows "Interaction notes" (Markdown, up to 8,000 characters) only for interactive problems.
- On an interactive problem each sample has an "Interactor input" field, and its input and output are labelled as the two sides of the transcript. Saving samples on an interactive problem with any sample lacking a non-blank interactor input fails with "Every interactive sample needs an interactor input."
-- Students can read the checker or interactor source; `judgemessage` and the interactor's stderr are not shown in Test.
+- Students can read the checker or interactor source, and the judge tab of the edit page says so under the checker or interactor language ("Students can read this program when they press Test."); `judgemessage` and the interactor's stderr are not shown in Test.
diff --git a/tests/component/web/judge-tab.test.ts b/tests/component/web/judge-tab.test.ts
new file mode 100644
index 000000000..abf866d09
--- /dev/null
+++ b/tests/component/web/judge-tab.test.ts
@@ -0,0 +1,55 @@
+// @vitest-environment jsdom
+
+import { mount, tick, unmount } from "svelte";
+import { afterEach, expect, it, vi } from "vitest";
+import type { JudgeType } from "@nojv/core";
+import type { ProblemDetail } from "$lib/types";
+
+import { m } from "$lib/paraglide/messages.js";
+
+vi.mock("$lib/utils/actions", () => ({ submitFormAction: vi.fn() }));
+vi.mock("$lib/components/primitives/ui/MonacoScriptEditor.svelte", async () => ({
+ default: (await import("../../fixtures/web/empty-component.svelte")).default,
+}));
+
+const { default: JudgeTab } =
+ await import("$lib/components/features/problem/tabs/JudgeTab.svelte");
+
+let target: HTMLDivElement | undefined;
+let component: ReturnType | undefined;
+
+afterEach(async () => {
+ if (component) await unmount(component);
+ target?.remove();
+ component = undefined;
+ target = undefined;
+});
+
+async function render(type: JudgeType) {
+ target = document.createElement("div");
+ document.body.append(target);
+ component = mount(JudgeTab, {
+ target,
+ props: {
+ problem: { id: "problem_1", judgeConfig: { type } } as unknown as ProblemDetail,
+ validatorScripts: { checkerScript: "", interactorScript: "" },
+ },
+ });
+ await tick();
+ return target;
+}
+
+it.each(["checker", "interactive"] as const)(
+ "tells editors that students can read the %s program",
+ async (type) => {
+ const view = await render(type);
+
+ expect(view.textContent).toContain(m.admin_judgeProgramReadableNote());
+ },
+);
+
+it("shows no judge-program note on a standard problem", async () => {
+ const view = await render("standard");
+
+ expect(view.textContent).not.toContain(m.admin_judgeProgramReadableNote());
+});
From 59ed49094269812f1de890c5981eb3a4762b4e94 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:29:28 +0800
Subject: [PATCH 18/22] fix(web): describe judge_log and the validator timeout
accurately
Co-Authored-By: Claude Opus 5.5
---
apps/web/messages/en.json | 4 ++--
apps/web/messages/zh-TW.json | 4 ++--
2 files changed, 4 insertions(+), 4 deletions(-)
diff --git a/apps/web/messages/en.json b/apps/web/messages/en.json
index a3e9a3ed6..2b36fab47 100644
--- a/apps/web/messages/en.json
+++ b/apps/web/messages/en.json
@@ -270,7 +270,7 @@
"admin_announcementsUnpin": "Unpin",
"admin_announcementsUnpublish": "Unpublish",
"admin_cancel": "Cancel",
- "admin_checkerHelpBody": "Your validator follows the DOMjudge output-validator interface. On Submit it runs in the judge sandbox, apart from the student's program, which never sees the answer; on Test it runs in the student's browser and checks the samples. It is invoked as `validator ` with the student output on stdin. It renders an accept/reject verdict only (no partial scoring). Python: `judge_input`, `judge_answer`, `team_output` are pre-bound as strings, and helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the right feedback files and exit. C++: read `argv[1]` / `argv[2]` and stdin yourself, write `teammessage.txt` into the `argv[3]` feedback dir, and exit 42 (accept) or 43 (wrong) — no judge header required. Feedback is shown to the student. Validator timeout is 30 seconds; a crash or hang becomes a system error and does not count against the student.",
+ "admin_checkerHelpBody": "Your validator follows the DOMjudge output-validator interface. On Submit it runs in the judge sandbox, apart from the student's program, which never sees the answer; on Test it runs in the student's browser and checks the samples. It is invoked as `validator ` with the student output on stdin. It renders an accept/reject verdict only (no partial scoring). Python: `judge_input`, `judge_answer`, `team_output` are pre-bound as strings, and `accept(team_msg=\"\")` and `wrong(team_msg=\"\")` write the feedback and exit, and `judge_log(msg)` writes the staff-only judge message. C++: read `argv[1]` / `argv[2]` and stdin yourself, write `teammessage.txt` into the `argv[3]` feedback dir, and exit 42 (accept) or 43 (wrong) — no judge header required. Feedback is shown to the student. The validator gets the larger of 30 seconds and the time limit; a crash or hang becomes a system error and does not count against the student.",
"admin_checkerHelpTitle": "How to write a checker",
"admin_compareCaseSensitive": "Case-sensitive comparison",
"admin_compareFloatTolerance": "Float tolerance (ε)",
@@ -311,7 +311,7 @@
"admin_inputFormatTooltip": "Describe the input format. Supports Markdown.",
"admin_interactionFormat": "Interaction notes",
"admin_interactionFormatTooltip": "Describe what the interactor reads from its input file and how it talks to the student's program. Shown to students. Supports Markdown.",
- "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface. On Submit it runs in the judge sandbox, and on Test in the student's browser; either way student stdout flows to interactor stdin, and interactor stdout flows to student stdin. It is invoked as `interactor `. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). Helpers `accept(team_msg=\"\")`, `wrong(team_msg=\"\")`, `judge_log(msg)` write the feedback files and exit 42/43. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
+ "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface. On Submit it runs in the judge sandbox, and on Test in the student's browser; either way student stdout flows to interactor stdin, and interactor stdout flows to student stdin. It is invoked as `interactor `. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). `accept(team_msg=\"\")` and `wrong(team_msg=\"\")` write the feedback and exit 42/43, and `judge_log(msg)` writes the staff-only judge message. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
"admin_interactorHelpTitle": "How to write an interactor",
"admin_interactorLanguage": "Interactor language",
"admin_judgeChecker": "Checker script",
diff --git a/apps/web/messages/zh-TW.json b/apps/web/messages/zh-TW.json
index 8077c6054..1639160a1 100644
--- a/apps/web/messages/zh-TW.json
+++ b/apps/web/messages/zh-TW.json
@@ -270,7 +270,7 @@
"admin_announcementsUnpin": "取消置頂",
"admin_announcementsUnpublish": "取消發佈",
"admin_cancel": "取消",
- "admin_checkerHelpBody": "你的 validator 採用 DOMjudge 輸出驗證器介面。提交時在判題沙盒中執行,與學生程式分開,學生程式看不到答案;按「測試」時則在學生的瀏覽器中執行,檢查範例。呼叫方式為 `validator `,學生輸出由 stdin 傳入。只輸出接受/錯誤判決(不做部分計分)。Python:`judge_input`、`judge_answer`、`team_output` 已預先注入為字串,輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入對應回饋檔並結束。C++:自行讀取 `argv[1]` / `argv[2]` 與 stdin,把 `teammessage.txt` 寫進 `argv[3]` 的 feedback 目錄,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。回饋訊息會顯示給學生。Validator 超時上限 30 秒;掛掉或逾時記為系統錯誤,不影響學生判決。",
+ "admin_checkerHelpBody": "你的 validator 採用 DOMjudge 輸出驗證器介面。提交時在判題沙盒中執行,與學生程式分開,學生程式看不到答案;按「測試」時則在學生的瀏覽器中執行,檢查範例。呼叫方式為 `validator `,學生輸出由 stdin 傳入。只輸出接受/錯誤判決(不做部分計分)。Python:`judge_input`、`judge_answer`、`team_output` 已預先注入為字串,`accept(team_msg=\"\")`、`wrong(team_msg=\"\")` 會寫入回饋並結束,`judge_log(msg)` 則寫入只給助教看的判題訊息。C++:自行讀取 `argv[1]` / `argv[2]` 與 stdin,把 `teammessage.txt` 寫進 `argv[3]` 的 feedback 目錄,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。回饋訊息會顯示給學生。Validator 的時限為 30 秒與題目時限取較大者;掛掉或逾時記為系統錯誤,不影響學生判決。",
"admin_checkerHelpTitle": "如何撰寫 Checker",
"admin_compareCaseSensitive": "區分大小寫比對",
"admin_compareFloatTolerance": "浮點容差(ε)",
@@ -311,7 +311,7 @@
"admin_inputFormatTooltip": "描述輸入資料的格式,支援 Markdown。",
"admin_interactionFormat": "互動說明",
"admin_interactionFormatTooltip": "說明 interactor 會從輸入檔讀到什麼,以及它如何與學生程式互動。會顯示給學生,支援 Markdown。",
- "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面。提交時在判題沙盒中執行,按「測試」時則在學生的瀏覽器中執行;兩種情況都一樣:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。輔助函式 `accept(team_msg=\"\")`、`wrong(team_msg=\"\")`、`judge_log(msg)` 會寫入回饋檔並以 exit 42/43 結束。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
+ "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面。提交時在判題沙盒中執行,按「測試」時則在學生的瀏覽器中執行;兩種情況都一樣:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。`accept(team_msg=\"\")`、`wrong(team_msg=\"\")` 會寫入回饋並以 exit 42/43 結束,`judge_log(msg)` 則寫入只給助教看的判題訊息。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
"admin_interactorHelpTitle": "如何撰寫 Interactor",
"admin_interactorLanguage": "Interactor 語言",
"admin_judgeChecker": "Checker 腳本",
From 16790de6e91f534c95ad0cd64f75d27caffbaf2e Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:59:19 +0800
Subject: [PATCH 19/22] fix(web): drop a queued judge-program build when its
editor closes
Each editor mount queued an uncancellable checker or interactor build, so a
student who clicked through several C++-checker problems waited behind every
one of them before their own Test ran.
The editor now passes an abort signal that fires on unmount.
compileBrowserJudgeProgram uses it only while waiting in the engine queue:
a build that has not started leaves the queue and the page-session memo,
so a later visit rebuilds it, and a build that has started keeps running
so the memo still gets its result. The memo key now includes the role,
because the Python wrapper differs between checker and interactor.
Co-Authored-By: Claude Opus 5.5
---
.../features/problem/editors/Editor.svelte | 8 +-
.../web/src/lib/services/browser-local-run.ts | 51 +++----
apps/web/src/lib/services/judge-program.ts | 15 ++-
docs/architecture/JUDGE_PIPELINE.md | 11 +-
docs/features/problem-test.md | 1 +
.../web/editor-judge-program.test.ts | 15 ++-
.../web/editor-test-button-state.test.ts | 6 +-
tests/unit/web/browser-engine-queue.test.ts | 18 ++-
tests/unit/web/judge-program.test.ts | 127 ++++++++++++++----
9 files changed, 189 insertions(+), 63 deletions(-)
diff --git a/apps/web/src/lib/components/features/problem/editors/Editor.svelte b/apps/web/src/lib/components/features/problem/editors/Editor.svelte
index 523421e27..481e5e5e0 100644
--- a/apps/web/src/lib/components/features/problem/editors/Editor.svelte
+++ b/apps/web/src/lib/components/features/problem/editors/Editor.svelte
@@ -91,6 +91,8 @@
let judgeProgramProgress = $state(null);
let judgeProgramFailure = $state | null>(null);
let judgeProgramPreparation: Promise | null = null;
+ const judgeProgramAbort = new AbortController();
+ onDestroy(() => judgeProgramAbort.abort());
const judgeProgramMessages =
initialProblem.judgeType === "interactive"
? {
@@ -108,7 +110,11 @@
function prepareProblemJudgeProgram(): Promise {
judgeProgramPreparation ??= prepareJudgeProgram(
- { problemId: initialProblem.id, context: untrack(() => context) },
+ {
+ problemId: initialProblem.id,
+ context: untrack(() => context),
+ signal: judgeProgramAbort.signal,
+ },
(progress) => (judgeProgramProgress = progress),
).then((prepared) => {
judgeProgramProgress = null;
diff --git a/apps/web/src/lib/services/browser-local-run.ts b/apps/web/src/lib/services/browser-local-run.ts
index e2e44a99e..7f5ed27ed 100644
--- a/apps/web/src/lib/services/browser-local-run.ts
+++ b/apps/web/src/lib/services/browser-local-run.ts
@@ -95,11 +95,7 @@ export async function prewarmBrowserLocalEngine(): Promise {
await getBrowserEngine();
}
-async function waitForTurn(
- previous: Promise,
- signal: AbortSignal | undefined,
-): Promise {
- if (!signal) return previous;
+async function waitForTurn(previous: Promise, signal: AbortSignal): Promise {
signal.throwIfAborted();
let leave!: () => void;
const left = new Promise((resolve) => (leave = resolve));
@@ -113,8 +109,9 @@ async function waitForTurn(
}
async function withBrowserEngine(
- signal: AbortSignal | undefined,
+ signal: AbortSignal,
operation: (engine: Engine) => Promise,
+ { cancelOnAbort = true } = {},
): Promise {
const previous = engineQueueTail;
let finish!: () => void;
@@ -127,13 +124,14 @@ async function withBrowserEngine(
}
try {
const engine = await getBrowserEngine();
- signal?.throwIfAborted();
+ signal.throwIfAborted();
+ if (!cancelOnAbort) return await operation(engine);
const cancel = () => engine.cancel();
- signal?.addEventListener("abort", cancel, { once: true });
+ signal.addEventListener("abort", cancel, { once: true });
try {
return await operation(engine);
} finally {
- signal?.removeEventListener("abort", cancel);
+ signal.removeEventListener("abort", cancel);
}
} finally {
finish();
@@ -419,22 +417,27 @@ export async function compileBrowserLocally(
export async function compileBrowserJudgeProgram(
problemId: string,
program: JudgeProgramSource,
+ signal: AbortSignal,
): Promise<{ ok: true; artifact: BuildArtifact } | { ok: false; diagnostics: string }> {
- return withBrowserEngine(undefined, async (browserEngine) => {
- const build = await browserEngine.compile(
- {
- ...judgeProgramCompileInput(program, WASM_OJ_LIBCXX_PCH_HEADER),
- target: "wasip1",
- optimization: "release",
- name: `NOJV ${program.role} ${problemId}`,
- projectId: `nojv-judge-program-v1-${problemId}-${program.role}-${program.language}`,
- },
- { cache: true },
- );
- if (!build.success || !build.artifact)
- return { ok: false, diagnostics: compileFeedback(build) };
- return { ok: true, artifact: build.artifact };
- });
+ return withBrowserEngine(
+ signal,
+ async (browserEngine) => {
+ const build = await browserEngine.compile(
+ {
+ ...judgeProgramCompileInput(program, WASM_OJ_LIBCXX_PCH_HEADER),
+ target: "wasip1",
+ optimization: "release",
+ name: `NOJV ${program.role} ${problemId}`,
+ projectId: `nojv-judge-program-v1-${problemId}-${program.role}-${program.language}`,
+ },
+ { cache: true },
+ );
+ if (!build.success || !build.artifact)
+ return { ok: false, diagnostics: compileFeedback(build) };
+ return { ok: true, artifact: build.artifact };
+ },
+ { cancelOnAbort: false },
+ );
}
export async function runBrowserCases(
diff --git a/apps/web/src/lib/services/judge-program.ts b/apps/web/src/lib/services/judge-program.ts
index f8faac444..210c7609a 100644
--- a/apps/web/src/lib/services/judge-program.ts
+++ b/apps/web/src/lib/services/judge-program.ts
@@ -58,11 +58,12 @@ async function fetchJudgeProgram(
function buildJudgeProgram(
problemId: string,
program: JudgeProgramSourceView,
+ signal: AbortSignal,
): Promise {
- const key = `${problemId}:${program.language}:${program.sha256}`;
+ const key = `${problemId}:${program.role}:${program.language}:${program.sha256}`;
const cached = builds.get(key);
if (cached) return cached;
- const build = compileBrowserJudgeProgram(problemId, program).then(
+ const build = compileBrowserJudgeProgram(problemId, program, signal).then(
(outcome): PreparedJudgeProgram =>
outcome.ok
? {
@@ -73,7 +74,7 @@ function buildJudgeProgram(
}
: { ok: false, reason: "build_failed", diagnostics: outcome.diagnostics },
(error: unknown): PreparedJudgeProgram => {
- console.warn("Couldn't build the judge program.", error);
+ if (!signal.aborted) console.warn("Couldn't build the judge program.", error);
builds.delete(key);
return LOAD_FAILED;
},
@@ -83,7 +84,11 @@ function buildJudgeProgram(
}
export async function prepareJudgeProgram(
- { problemId, context }: { problemId: string; context: SubmissionContext },
+ {
+ problemId,
+ context,
+ signal,
+ }: { problemId: string; context: SubmissionContext; signal: AbortSignal },
onProgress: (progress: JudgeProgramProgress) => void = () => undefined,
): Promise {
onProgress({ phase: "fetch" });
@@ -98,5 +103,5 @@ export async function prepareJudgeProgram(
return LOAD_FAILED;
}
onProgress({ phase: "build" });
- return buildJudgeProgram(problemId, program);
+ return buildJudgeProgram(problemId, program, signal);
}
diff --git a/docs/architecture/JUDGE_PIPELINE.md b/docs/architecture/JUDGE_PIPELINE.md
index 1d4e77900..425f90b7b 100644
--- a/docs/architecture/JUDGE_PIPELINE.md
+++ b/docs/architecture/JUDGE_PIPELINE.md
@@ -499,8 +499,10 @@ judge program's source. **Submit** always uses the server pipeline above.
checker runs and interactions) goes through one module-level FIFO queue on one
shared engine, which runs one foreground compile at a time. A caller that aborts
while waiting leaves the queue, and aborting cancels the engine only while that
- caller's own operation runs. A judge-program build nobody waits for any more keeps
- running, because the page-session memo wants its result.
+ caller's own operation runs. A judge-program build waits on its editor's signal: when
+ the editor closes before the build starts, it leaves the queue and the page-session
+ memo, so a later visit rebuilds it; a build that has started keeps running, because
+ the memo wants its result.
### Availability
@@ -540,8 +542,9 @@ for the selected language, `prepareJudgeProgram`
(`python-judge-wrappers.ts`, byte-identical to the sandbox runner's) and is
packaged without a compile; C++ gets the platform `bits/stdc++.h` shim and the
libc++ PCH header under the rule above;
-4. keeps the built program for the page session per problem, language and `sha256`.
- Each editor mount refetches the source, so it rebuilds only when the source changed.
+4. keeps the built program for the page session per problem, role, language and
+ `sha256`. Each editor mount refetches the source, so it rebuilds only when the
+ source changed.
While it prepares, Test stays clickable and shows the toolchain download or
"Preparing checker..." / "Preparing interactor..."; pressing Test waits for it.
diff --git a/docs/features/problem-test.md b/docs/features/problem-test.md
index 1c7aad60d..9ffbc5918 100644
--- a/docs/features/problem-test.md
+++ b/docs/features/problem-test.md
@@ -41,6 +41,7 @@ The endpoint requires a session and the problem page's view access for the conte
- Given a checker or interactive problem, when the editor opens, the browser fetches the problem's checker or interactor and builds it while the student works; the Test button shows the toolchain download or "Preparing checker..." / "Preparing interactor...".
- Given the judge program is still preparing, when the student presses Test, Test waits for it and then runs.
- Given the same problem is reopened in the page session with an unchanged judge program, it is not rebuilt; given its source changed, it is rebuilt.
+- Given the student leaves a problem before its judge program starts building, that build is dropped and does not delay Test on the next problem; reopening the problem builds it then.
- Given the judge program fails to build, Test is disabled with "This problem's checker failed to build." (or "interactor"), and the Test Result panel opens on the compiler output.
- Given the source request is refused (403 or 404), Test is disabled with "Couldn't load this problem's checker." (or "interactor"); given it fails for any other reason (a server or network failure is retried twice first), the message adds "Reload the page to try again."
- Given a student who may view the problem in the context (including after an exam or contest ends, while the problem is still viewable), the source loads; given an active page-locked exam session, only that exam's problems load.
diff --git a/tests/component/web/editor-judge-program.test.ts b/tests/component/web/editor-judge-program.test.ts
index f8d68d756..420d2472e 100644
--- a/tests/component/web/editor-judge-program.test.ts
+++ b/tests/component/web/editor-judge-program.test.ts
@@ -8,7 +8,7 @@ import { m } from "$lib/paraglide/messages.js";
type Prepared = import("$lib/services/judge-program").PreparedJudgeProgram;
const mocks = vi.hoisted(() => ({
- prepare: vi.fn<() => Promise>(),
+ prepare: vi.fn<(request: { signal: AbortSignal }) => Promise>(),
compile: vi.fn(),
runCases: vi.fn(),
check: vi.fn(),
@@ -127,3 +127,16 @@ it("prepares the checker once across opening the editor and pressing Test", asyn
expect(mocks.check).toHaveBeenCalledOnce();
expect(mocks.prepare).toHaveBeenCalledOnce();
});
+
+it("lets a queued checker build go when the editor closes", async () => {
+ mocks.prepare.mockReturnValue(new Promise(() => undefined));
+ mountCheckerEditor();
+ await vi.waitFor(() => expect(mocks.prepare).toHaveBeenCalledOnce());
+ const [{ signal }] = mocks.prepare.mock.calls[0]!;
+ expect(signal.aborted).toBe(false);
+
+ await unmount(component!);
+ component = undefined;
+
+ expect(signal.aborted).toBe(true);
+});
diff --git a/tests/component/web/editor-test-button-state.test.ts b/tests/component/web/editor-test-button-state.test.ts
index cf4fe94b1..c70a874c7 100644
--- a/tests/component/web/editor-test-button-state.test.ts
+++ b/tests/component/web/editor-test-button-state.test.ts
@@ -142,7 +142,11 @@ describe("Test button state", () => {
const button = await renderTestButton({ judgeType: "checker" });
await vi.waitFor(() => expect(button.textContent).toContain(m.editor_checkerPreparing()));
expect(mocks.prepare).toHaveBeenCalledWith(
- { problemId: "problem_1", context: { type: "practice" } },
+ {
+ problemId: "problem_1",
+ context: { type: "practice" },
+ signal: expect.any(AbortSignal),
+ },
expect.any(Function),
);
expect(button.disabled).toBe(false);
diff --git a/tests/unit/web/browser-engine-queue.test.ts b/tests/unit/web/browser-engine-queue.test.ts
index 5d0b7684f..1bf92b03f 100644
--- a/tests/unit/web/browser-engine-queue.test.ts
+++ b/tests/unit/web/browser-engine-queue.test.ts
@@ -76,7 +76,11 @@ async function settle() {
}
it("queues a student compile behind a judge-program build that is still running", async () => {
- const checkerBuild = compileBrowserJudgeProgram("checker-problem", checker);
+ const checkerBuild = compileBrowserJudgeProgram(
+ "checker-problem",
+ checker,
+ new AbortController().signal,
+ );
await vi.waitFor(() => expect(engine.compile).toHaveBeenCalledOnce());
const studentBuild = compileBrowserLocally(request, "next", new AbortController().signal);
@@ -93,7 +97,11 @@ it("queues a student compile behind a judge-program build that is still running"
});
it("lets an aborted waiter leave the queue without cancelling the build ahead of it", async () => {
- const checkerBuild = compileBrowserJudgeProgram("checker-problem", checker);
+ const checkerBuild = compileBrowserJudgeProgram(
+ "checker-problem",
+ checker,
+ new AbortController().signal,
+ );
await vi.waitFor(() => expect(engine.compile).toHaveBeenCalledOnce());
const leaving = new AbortController();
const abandoned = compileBrowserLocally(request, "next", leaving.signal);
@@ -120,7 +128,11 @@ it("cancels the engine only for the operation that is running", async () => {
const leaving = new AbortController();
const studentBuild = compileBrowserLocally(request, "next", leaving.signal);
await vi.waitFor(() => expect(engine.compile).toHaveBeenCalledOnce());
- const checkerBuild = compileBrowserJudgeProgram("checker-problem", checker);
+ const checkerBuild = compileBrowserJudgeProgram(
+ "checker-problem",
+ checker,
+ new AbortController().signal,
+ );
leaving.abort();
expect(engine.cancel).toHaveBeenCalledOnce();
diff --git a/tests/unit/web/judge-program.test.ts b/tests/unit/web/judge-program.test.ts
index 18dfc29f2..139f8fd2e 100644
--- a/tests/unit/web/judge-program.test.ts
+++ b/tests/unit/web/judge-program.test.ts
@@ -45,19 +45,22 @@ const cppChecker = {
sha256: "b".repeat(64),
};
const context = { type: "practice" } as const;
+const signal = new AbortController().signal;
+const built = {
+ success: true,
+ artifact: { id: "checker-artifact" },
+ diagnostics: [],
+ stdout: "",
+ stderr: "",
+};
function respondWith(body: unknown, status = 200) {
return new Response(JSON.stringify(body), { status });
}
beforeEach(() => {
- fakes.engine.compile.mockReset().mockResolvedValue({
- success: true,
- artifact: { id: "checker-artifact" },
- diagnostics: [],
- stdout: "",
- stderr: "",
- });
+ fakes.engine.compile.mockReset().mockResolvedValue(built);
+ fakes.engine.cancel.mockReset();
fakes.preload.mockReset().mockResolvedValue(undefined);
fakes.fetch.mockReset().mockImplementation(async () => respondWith(pythonChecker));
fakes.warn.mockReset();
@@ -74,8 +77,9 @@ afterEach(() => {
it("fetches a Python checker for its context and packages it with the DOMjudge wrapper", async () => {
const progress: unknown[] = [];
- const prepared = await prepareJudgeProgram({ problemId: "python-1", context }, (update) =>
- progress.push(update),
+ const prepared = await prepareJudgeProgram(
+ { problemId: "python-1", context, signal },
+ (update) => progress.push(update),
);
expect(prepared).toEqual({
@@ -109,8 +113,9 @@ it("compiles a C++ checker with the libc++ PCH shim and reports its toolchain do
});
const progress: unknown[] = [];
- const prepared = await prepareJudgeProgram({ problemId: "cpp-1", context }, (update) =>
- progress.push(update),
+ const prepared = await prepareJudgeProgram(
+ { problemId: "cpp-1", context, signal },
+ (update) => progress.push(update),
);
expect(prepared).toMatchObject({ ok: true, language: "cpp" });
@@ -141,7 +146,9 @@ it("returns the compiler output when the checker fails to build", async () => {
stderr: "main.cpp:2:1: error: expected ';'",
});
- await expect(prepareJudgeProgram({ problemId: "broken-1", context })).resolves.toEqual({
+ await expect(
+ prepareJudgeProgram({ problemId: "broken-1", context, signal }),
+ ).resolves.toEqual({
ok: false,
reason: "build_failed",
diagnostics: "main.cpp:2:1: error: expected ';'",
@@ -149,19 +156,19 @@ it("returns the compiler output when the checker fails to build", async () => {
});
it("refetches the source on every preparation but builds each digest once", async () => {
- await prepareJudgeProgram({ problemId: "memo-1", context });
- await prepareJudgeProgram({ problemId: "memo-1", context });
+ await prepareJudgeProgram({ problemId: "memo-1", context, signal });
+ await prepareJudgeProgram({ problemId: "memo-1", context, signal });
expect(fakes.fetch).toHaveBeenCalledTimes(2);
expect(fakes.engine.compile).toHaveBeenCalledOnce();
});
it("rebuilds when the source's digest changes", async () => {
- await prepareJudgeProgram({ problemId: "digest-1", context });
+ await prepareJudgeProgram({ problemId: "digest-1", context, signal });
fakes.fetch.mockImplementation(async () =>
respondWith({ ...pythonChecker, source: "reject()\n", sha256: "c".repeat(64) }),
);
- await prepareJudgeProgram({ problemId: "digest-1", context });
+ await prepareJudgeProgram({ problemId: "digest-1", context, signal });
expect(fakes.engine.compile).toHaveBeenCalledTimes(2);
expect(fakes.engine.compile.mock.calls[1]![0].files["main.py"]!.endsWith("reject()\n")).toBe(
@@ -170,23 +177,87 @@ it("rebuilds when the source's digest changes", async () => {
});
it("rebuilds when only the language changes", async () => {
- await prepareJudgeProgram({ problemId: "language-1", context });
+ await prepareJudgeProgram({ problemId: "language-1", context, signal });
fakes.fetch.mockImplementation(async () =>
respondWith({ ...cppChecker, sha256: pythonChecker.sha256 }),
);
await expect(
- prepareJudgeProgram({ problemId: "language-1", context }),
+ prepareJudgeProgram({ problemId: "language-1", context, signal }),
).resolves.toMatchObject({ ok: true, language: "cpp" });
expect(fakes.engine.compile).toHaveBeenCalledTimes(2);
expect(fakes.engine.compile.mock.calls[1]![0].language).toBe("cpp");
});
+it("rebuilds when only the role changes", async () => {
+ await prepareJudgeProgram({ problemId: "role-1", context, signal });
+ fakes.fetch.mockImplementation(async () =>
+ respondWith({ ...pythonChecker, role: "interactor" }),
+ );
+
+ await expect(
+ prepareJudgeProgram({ problemId: "role-1", context, signal }),
+ ).resolves.toMatchObject({ ok: true, role: "interactor" });
+ expect(fakes.engine.compile).toHaveBeenCalledTimes(2);
+ expect(fakes.engine.compile.mock.calls[1]![0].projectId).toBe(
+ "nojv-judge-program-v1-role-1-interactor-python",
+ );
+});
+
+it("drops a queued build when its editor closes, so the next visit builds it", async () => {
+ let finishBusy!: (value: unknown) => void;
+ fakes.engine.compile.mockImplementationOnce(
+ () => new Promise((resolve) => (finishBusy = resolve)),
+ );
+ const busy = prepareJudgeProgram({ problemId: "busy-1", context, signal });
+ await vi.waitFor(() => expect(fakes.engine.compile).toHaveBeenCalledOnce());
+ const leaving = new AbortController();
+ const progress: unknown[] = [];
+ const left = prepareJudgeProgram(
+ { problemId: "left-1", context, signal: leaving.signal },
+ (update) => progress.push(update),
+ );
+ await vi.waitFor(() => expect(progress).toContainEqual({ phase: "build" }));
+
+ leaving.abort();
+ finishBusy(built);
+ await expect(busy).resolves.toMatchObject({ ok: true });
+ await expect(left).resolves.toEqual({ ok: false, reason: "load_failed" });
+ expect(fakes.engine.compile).toHaveBeenCalledOnce();
+ expect(fakes.warn).not.toHaveBeenCalled();
+
+ await expect(
+ prepareJudgeProgram({ problemId: "left-1", context, signal }),
+ ).resolves.toMatchObject({ ok: true });
+ expect(fakes.engine.compile).toHaveBeenCalledTimes(2);
+ expect(fakes.engine.compile.mock.calls[1]![0].projectId).toBe(
+ "nojv-judge-program-v1-left-1-checker-python",
+ );
+});
+
+it("finishes and keeps a build that started before its editor closed", async () => {
+ let finishBuild!: (value: unknown) => void;
+ fakes.engine.compile.mockImplementationOnce(
+ () => new Promise((resolve) => (finishBuild = resolve)),
+ );
+ const leaving = new AbortController();
+ const left = prepareJudgeProgram({ problemId: "started-1", context, signal: leaving.signal });
+ await vi.waitFor(() => expect(fakes.engine.compile).toHaveBeenCalledOnce());
+
+ leaving.abort();
+ finishBuild(built);
+ await expect(left).resolves.toMatchObject({ ok: true });
+ expect(fakes.engine.cancel).not.toHaveBeenCalled();
+
+ await prepareJudgeProgram({ problemId: "started-1", context, signal });
+ expect(fakes.engine.compile).toHaveBeenCalledOnce();
+});
+
it("retries a failing source request before reporting a load failure", async () => {
vi.useFakeTimers();
fakes.fetch.mockImplementation(async () => respondWith({}, 503));
- const pending = prepareJudgeProgram({ problemId: "flaky-1", context });
+ const pending = prepareJudgeProgram({ problemId: "flaky-1", context, signal });
await vi.advanceTimersByTimeAsync(7_000);
await expect(pending).resolves.toEqual({ ok: false, reason: "load_failed" });
@@ -203,7 +274,9 @@ it.each([403, 404])(
async (status) => {
fakes.fetch.mockImplementation(async () => respondWith({ message: "No" }, status));
- await expect(prepareJudgeProgram({ problemId: "refused-1", context })).resolves.toEqual({
+ await expect(
+ prepareJudgeProgram({ problemId: "refused-1", context, signal }),
+ ).resolves.toEqual({
ok: false,
reason: "unavailable",
});
@@ -218,7 +291,7 @@ it("reports a malformed source response as a load failure", async () => {
vi.useFakeTimers();
fakes.fetch.mockImplementation(async () => respondWith({ ...pythonChecker, sha256: "x" }));
- const pending = prepareJudgeProgram({ problemId: "malformed-1", context });
+ const pending = prepareJudgeProgram({ problemId: "malformed-1", context, signal });
await vi.advanceTimersByTimeAsync(7_000);
await expect(pending).resolves.toEqual({ ok: false, reason: "load_failed" });
@@ -233,7 +306,9 @@ it("reports a toolchain that can't be loaded as a load failure", async () => {
const failure = new TypeError("Failed to fetch");
fakes.preload.mockRejectedValue(failure);
- await expect(prepareJudgeProgram({ problemId: "toolchain-1", context })).resolves.toEqual({
+ await expect(
+ prepareJudgeProgram({ problemId: "toolchain-1", context, signal }),
+ ).resolves.toEqual({
ok: false,
reason: "load_failed",
});
@@ -248,12 +323,16 @@ it("reports an engine failure as a load failure and builds again next time", asy
const failure = new Error("Failed to fetch");
fakes.engine.compile.mockRejectedValueOnce(failure);
- await expect(prepareJudgeProgram({ problemId: "engine-1", context })).resolves.toEqual({
+ await expect(
+ prepareJudgeProgram({ problemId: "engine-1", context, signal }),
+ ).resolves.toEqual({
ok: false,
reason: "load_failed",
});
expect(fakes.warn).toHaveBeenCalledWith("Couldn't build the judge program.", failure);
- await expect(prepareJudgeProgram({ problemId: "engine-1", context })).resolves.toMatchObject({
+ await expect(
+ prepareJudgeProgram({ problemId: "engine-1", context, signal }),
+ ).resolves.toMatchObject({
ok: true,
});
expect(fakes.engine.compile).toHaveBeenCalledTimes(2);
From 485a3bb185f03695da877665d8fbb82671872fdc Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 20:59:29 +0800
Subject: [PATCH 20/22] fix: align Test docs, interactor help and verdict
helper with the code
- interactiveCaseVerdict drops its teamMessage parameter: no caller passed
it, and interactive Test never shows an interactor's teammessage.
- The interactor help says Test passes an empty judge_answer, so the
interactor should read its secret from judge_input.
- tests/tsconfig.json includes unit/worker/mailer-startup.test.ts again,
so typecheck:tests covers it.
- The judge pipeline doc says contest organisers read the judge program
only while the contest is published, and that Test on checker and
interactive problems waits while the judge program prepares. It and the
problem-test spec say the result shows the run's longest logical time,
not a time per case.
- JDG-15 says exit codes map as in official judging but the interactive
merge order differs, and that teammessage is shown for checkers only.
- The Quality Ledger drops the pin-bump item, which must land before this
merges, keeping the JavaScript and TypeScript clause, and drops the
unrelated Wasm fast path item.
Co-Authored-By: Claude Opus 5.5
---
apps/web/messages/en.json | 2 +-
apps/web/messages/zh-TW.json | 2 +-
docs/architecture/JUDGE_PIPELINE.md | 34 ++++++++--------
docs/decisions/judge.md | 2 +-
docs/features/problem-test.md | 2 +-
docs/operations/QUALITY_SCORE.md | 3 +-
packages/core/src/judge/test-judge-verdict.ts | 10 ++---
tests/tsconfig.json | 1 +
tests/unit/core/test-judge-verdict.test.ts | 39 +++++++++----------
9 files changed, 47 insertions(+), 48 deletions(-)
diff --git a/apps/web/messages/en.json b/apps/web/messages/en.json
index 2b36fab47..cf531edf8 100644
--- a/apps/web/messages/en.json
+++ b/apps/web/messages/en.json
@@ -311,7 +311,7 @@
"admin_inputFormatTooltip": "Describe the input format. Supports Markdown.",
"admin_interactionFormat": "Interaction notes",
"admin_interactionFormatTooltip": "Describe what the interactor reads from its input file and how it talks to the student's program. Shown to students. Supports Markdown.",
- "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface. On Submit it runs in the judge sandbox, and on Test in the student's browser; either way student stdout flows to interactor stdin, and interactor stdout flows to student stdin. It is invoked as `interactor `. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). `accept(team_msg=\"\")` and `wrong(team_msg=\"\")` write the feedback and exit 42/43, and `judge_log(msg)` writes the staff-only judge message. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
+ "admin_interactorHelpBody": "Your interactor follows the DOMjudge interactive interface. On Submit it runs in the judge sandbox, and on Test in the student's browser; either way student stdout flows to interactor stdin, and interactor stdout flows to student stdin. It is invoked as `interactor `; Test passes an empty `judge_answer`, so read the secret from `judge_input`. Python: `judge_input` and `judge_answer` are pre-bound strings; `read()` reads one line from the student (auto-rejecting if the stream closes) and `write(msg)` sends one line back (already flushed). `accept(team_msg=\"\")` and `wrong(team_msg=\"\")` write the feedback and exit 42/43, and `judge_log(msg)` writes the staff-only judge message. C++: read `argv[1]`/`argv[2]`, talk to the solution over `cin`/`cout` (flush before reading), write `teammessage.txt` into `argv[3]`, and exit 42 (accept) or 43 (wrong) — no judge header required. The student and interactor share the problem's time limit.",
"admin_interactorHelpTitle": "How to write an interactor",
"admin_interactorLanguage": "Interactor language",
"admin_judgeChecker": "Checker script",
diff --git a/apps/web/messages/zh-TW.json b/apps/web/messages/zh-TW.json
index 1639160a1..4e04191cb 100644
--- a/apps/web/messages/zh-TW.json
+++ b/apps/web/messages/zh-TW.json
@@ -311,7 +311,7 @@
"admin_inputFormatTooltip": "描述輸入資料的格式,支援 Markdown。",
"admin_interactionFormat": "互動說明",
"admin_interactionFormatTooltip": "說明 interactor 會從輸入檔讀到什麼,以及它如何與學生程式互動。會顯示給學生,支援 Markdown。",
- "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面。提交時在判題沙盒中執行,按「測試」時則在學生的瀏覽器中執行;兩種情況都一樣:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。`accept(team_msg=\"\")`、`wrong(team_msg=\"\")` 會寫入回饋並以 exit 42/43 結束,`judge_log(msg)` 則寫入只給助教看的判題訊息。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
+ "admin_interactorHelpBody": "你的 interactor 採用 DOMjudge 互動介面。提交時在判題沙盒中執行,按「測試」時則在學生的瀏覽器中執行;兩種情況都一樣:學生 stdout → interactor stdin,interactor stdout → 學生 stdin。呼叫方式為 `interactor `;按「測試」時 `judge_answer` 是空的,請改從 `judge_input` 讀取祕密資料。Python:`judge_input`、`judge_answer` 已預先注入為字串;`read()` 讀學生送來的一行(自動去掉換行符號,若學生提早關閉串流會自動 wrong);`write(msg)` 寫一行給學生(已自動 flush)。`accept(team_msg=\"\")`、`wrong(team_msg=\"\")` 會寫入回饋並以 exit 42/43 結束,`judge_log(msg)` 則寫入只給助教看的判題訊息。C++:讀取 `argv[1]` / `argv[2]`,透過 `cin` / `cout` 與學生互動(讀取前先 flush),把 `teammessage.txt` 寫進 `argv[3]`,並以 exit 42(接受)或 43(錯誤)結束——不需任何評測標頭檔。Interactor 與學生共享題目的時間限制。",
"admin_interactorHelpTitle": "如何撰寫 Interactor",
"admin_interactorLanguage": "Interactor 語言",
"admin_judgeChecker": "Checker 腳本",
diff --git a/docs/architecture/JUDGE_PIPELINE.md b/docs/architecture/JUDGE_PIPELINE.md
index 425f90b7b..8b0383cf2 100644
--- a/docs/architecture/JUDGE_PIPELINE.md
+++ b/docs/architecture/JUDGE_PIPELINE.md
@@ -506,13 +506,13 @@ judge program's source. **Submit** always uses the server pipeline above.
### Availability
-| Case | Test |
-| -------------------------------------------------------- | -------------------------------------------- |
-| `standard` | Available |
-| `checker` or `interactive` | Available once the judge program is prepared |
-| `special_env` | Disabled ("doesn't support Test") |
-| Interactive with a JavaScript or TypeScript contestant | Disabled for that language |
-| The judge program failed to build or could not be loaded | Disabled for the rest of the editor session |
+| Case | Test |
+| -------------------------------------------------------- | ------------------------------------------------- |
+| `standard` | Available |
+| `checker` or `interactive` | Available; waits while the judge program prepares |
+| `special_env` | Disabled ("doesn't support Test") |
+| Interactive with a JavaScript or TypeScript contestant | Disabled for that language |
+| The judge program failed to build or could not be loaded | Disabled for the rest of the editor session |
### Judge programs
@@ -522,12 +522,13 @@ allowlist) returns `{ role, language, source, sha256 }` for a checker or interac
problem, read through the verified script pointer.
- Access is the problem page's view access for that context. A context grants its
- problem while its page would render: course staff and contest organisers always;
- students in an open assignment, a running published contest they joined, their own
- running virtual contest, or a running exam session that passes the proctoring gate.
- Otherwise the practice view rules apply, so the program stays readable after an exam
- or contest ends to students who can still view the problem. A page-locked exam
- session keeps the request inside that exam's problems.
+ problem while its page would render: course staff always; contest organisers while
+ the contest is published; students in an open assignment, a running published
+ contest they joined, their own running virtual contest, or a running exam session
+ that passes the proctoring gate. Otherwise the practice view rules apply, so the
+ program stays readable after an exam or contest ends to students who can still view
+ the problem. A page-locked exam session keeps the request inside that exam's
+ problems.
- Hidden testcases never leave the server; Test uses only the problem's samples and
the student's own cases.
@@ -584,9 +585,10 @@ While it prepares, Test stays clickable and shows the toolchain download or
MLE, whatever the interactor did; otherwise an interactor that exits with anything
but 42 or 43, or is stopped by a limit, is SE; otherwise a contestant RE wins;
otherwise the interactor's AC or WA stands.
-4. Each case shows the verdict, contestant stderr (100,000 bytes), both transcript
- directions (64 KiB each, cut at a UTF-8 boundary) and the contestant's logical
- time. The interactor's `teammessage` and stderr are not shown.
+4. Each case shows the verdict, contestant stderr (100,000 bytes) and both transcript
+ directions (64 KiB each, cut at a UTF-8 boundary); the run shows the largest
+ contestant logical time across its cases. The interactor's `teammessage` and stderr
+ are not shown.
5. When the contestant exits, normally or with an error, before the interactor's next
write, that write fails (a Python interactor exits 120) and the case is SE; Submit
gives that case RE or the interactor's verdict.
diff --git a/docs/decisions/judge.md b/docs/decisions/judge.md
index 7dc9e7810..6a5b2459d 100644
--- a/docs/decisions/judge.md
+++ b/docs/decisions/judge.md
@@ -187,7 +187,7 @@ Test never creates a submission and runs entirely in the student's browser. The
- Rule: Test never receives data from non-sample testcases (SEC-12).
- Rule: generic runtime fixes land in wasm-oj/forge and are consumed as pinned releases; NOJV never forks or patches forge. Keep document CSP at `wasm-unsafe-eval`. JavaScript and TypeScript contestants on interactive problems stay unavailable until upstream streams QuickJS stdin.
- Rule: browser results are previews on Forge logical time; do not claim resource or toolchain-version equivalence; fix sample data, not engine input.
-- Rule: verdicts map as in official judging through core's `checkerCaseVerdict` and `interactiveCaseVerdict`, the one place for that rule; Test shows `teammessage`, the transcript and contestant stderr, never `judgemessage` or interactor stderr.
+- Rule: core's `checkerCaseVerdict` and `interactiveCaseVerdict` are the one place for Test's verdict mapping. Judge-program exit codes mean what they mean in official judging (42 AC, 43 WA, anything else SE), but the interactive merge order differs, as the next rule says. Test shows a checker's `teammessage`, the transcript and contestant stderr; it never shows `judgemessage`, an interactor's `teammessage` or interactor stderr.
- Rule: in interactive Test a contestant stopped by a time limit (logical time, instruction budget or the wall stop) or the memory limit is TLE or MLE even when the interactor then fails. The browser engine closes a stopped side's pipes, while official judging keeps draining the interactor's output and gives it EOF, so an interactor that writes after a limit stop fails only in Test and Submit gives TLE or MLE.
- Rejected: letting an interactor failure win over a contestant limit stop in Test, as official judging's merge order does; Test showed SE where Submit gives TLE or MLE.
- Rule: source diagnostics are CE, toolchain/infrastructure faults SE. Custom cases without expected output are execution-only; on a checker problem only a case whose input is a sample's input is checked, with that sample's output as the answer, while an interactive custom case supplies the interactor's input and is judged.
diff --git a/docs/features/problem-test.md b/docs/features/problem-test.md
index 9ffbc5918..fbc201ac3 100644
--- a/docs/features/problem-test.md
+++ b/docs/features/problem-test.md
@@ -56,7 +56,7 @@ The endpoint requires a session and the problem page's view access for the conte
- Given an interactive problem, the case panel lists each sample's interactor input and shows the problem's interaction notes; a sample without an interactor input is left out.
- Given the student adds a case, its input is an interactor input (for example the hidden number), and the real interactor judges it like a sample. The case panel says "…Cases you add are judged the same way."
-- When the student presses Test, the browser compiles the program and runs it against the interactor once per case. Each case shows its verdict, the student's stderr, the time, and a transcript with "From the interactor" and "From your program", each cut at 64 KiB.
+- When the student presses Test, the browser compiles the program and runs it against the interactor once per case. Each case shows its verdict, the student's stderr, and a transcript with "From the interactor" and "From your program", each cut at 64 KiB; the result shows the longest logical time across the cases.
- Given the student's program is stopped by a time or memory limit, the case is TLE or MLE, whatever the interactor does; otherwise, given the interactor fails (an exit code other than 42 or 43, or a limit), the case is SE; otherwise the student's RE wins; otherwise the interactor's AC or WA stands.
- Given JavaScript or TypeScript is selected, Test is disabled with "Test can't run interactive problems in JavaScript or TypeScript yet…"; switching to another language enables it.
diff --git a/docs/operations/QUALITY_SCORE.md b/docs/operations/QUALITY_SCORE.md
index a470fdc2a..a604b0a37 100644
--- a/docs/operations/QUALITY_SCORE.md
+++ b/docs/operations/QUALITY_SCORE.md
@@ -67,12 +67,11 @@ Work that is known, not done, and not covered by an in-flight plan. Remove an it
- SonarQube reports 83 functions over the cognitive complexity limit (rule S3776, threshold 15). The worst are the better-auth `hooks.before` middleware in `apps/web/src/lib/auth.server.ts` (77), `scripts/judge-benchmark.ts` (72), `scripts/check-supply-chain-policy.mjs` (63) and `durableJudgeWorkflow` (61; any split must keep replay determinism).
- The admin audit log's actor, action and target columns have no filters; adding them needs server-side filtering in `adminAuditLogRepo.listPaged` (UI-12).
- Browser Test (WASM-OJ) deferred scope: official Submit from the browser, Advanced problems and limit calibration stay server-only until decided otherwise (JDG-15).
-- Browser checker and interactive Test need one `@wasm-oj` release that ships wasm-oj/forge#93 (Python interactors in `interact`), #95 (in-module metering of interactive programs), #96 (runtime-file export stall) and #99 (interactive sides in nested Workers); bump the pins to it before release (JDG-15). JavaScript and TypeScript contestants on interactive problems wait for streaming QuickJS stdin upstream.
+- JavaScript and TypeScript contestants on interactive problems wait for streaming QuickJS stdin upstream (JDG-15).
- The browser engine closes a stopped side's pipes, while official judging keeps reading the interactor's output after the contestant ends and only gives it EOF. When the contestant exits, normally or with an error, before the interactor's next write, that write fails in Test (a CPython interactor exits 120) and the case is SE where Submit gives RE or the interactor's verdict. A Python interactor that starts after a fast-failing contestant has stopped hits this on its first write. A contestant stopped by a time or memory limit is already TLE or MLE in Test whatever the interactor does (JDG-15). Fix upstream by discarding a finished contestant's input instead of failing the write.
- After such a broken pipe the browser transcript can repeat the line the interactor failed to write (one CPython line showed five times). Raise upstream with the fix above.
- In WebKit an empty `for(;;);` loop is not stopped by the instruction meter and runs to the wall stop, in `run` and `interact` alike; a loop with a side effect stops at its budget. Raise upstream.
- Every official checker or interactive stage recompiles its judge program. Precompiling native judge programs once per source, language and sandbox image is open; measure the C++ per-stage compile cost first (on 2026-10-04 every production judge program was Python, which needs no compile).
-- Evaluate a Wasm fast path for official judging of standard problems: run standard-problem submissions under a server-side WASM-OJ runtime instead of a gVisor stage Pod. It could remove most per-stage Pod overhead, but it brings back a server WASM-OJ runtime and toolchains in the worker image (OPS-21, withdrawn), adds a second execution path whose logical-time limits need calibrating against native CPU time and whose toolchains must track `judge-environment.json`, and puts a WebAssembly runtime instead of gVisor between student code and the worker's credentials. Measure the per-stage overhead it would save first.
## Evidence rules
diff --git a/packages/core/src/judge/test-judge-verdict.ts b/packages/core/src/judge/test-judge-verdict.ts
index fd7a746b2..c0276c21d 100644
--- a/packages/core/src/judge/test-judge-verdict.ts
+++ b/packages/core/src/judge/test-judge-verdict.ts
@@ -36,16 +36,16 @@ export function checkerCaseVerdict(
return { verdict: outcome.verdict, teamMessage: capFeedback(outcome.teamMessage) };
}
-export function interactiveCaseVerdict(
- result: { contestant: ProcessTermination; interactor: ProcessTermination },
- teamMessage?: string,
-): { verdict: TestJudgeVerdict; teamMessage?: string } {
+export function interactiveCaseVerdict(result: {
+ contestant: ProcessTermination;
+ interactor: ProcessTermination;
+}): { verdict: TestJudgeVerdict } {
const contestant = wasmOjTerminationVerdict(
result.contestant.termination,
result.contestant.code,
);
if (contestant === "TLE" || contestant === "MLE") return { verdict: contestant };
- const interactor = checkerCaseVerdict(result.interactor, teamMessage);
+ const interactor = checkerCaseVerdict(result.interactor);
if (interactor.verdict === "SE") return interactor;
return contestant === "AC" ? interactor : { verdict: contestant };
}
diff --git a/tests/tsconfig.json b/tests/tsconfig.json
index 228022ca1..e22627028 100644
--- a/tests/tsconfig.json
+++ b/tests/tsconfig.json
@@ -92,6 +92,7 @@
"unit/worker/docker-resource-sweeper.test.ts",
"unit/worker/service-container-readiness.test.ts",
"unit/worker/worker-app.test.ts",
+ "unit/worker/mailer-startup.test.ts",
"unit/worker/judge-benchmark.test.ts",
"unit/worker/judge-phase-metrics.test.ts",
"unit/worker/k8s-interactive-orchestration.test.ts",
diff --git a/tests/unit/core/test-judge-verdict.test.ts b/tests/unit/core/test-judge-verdict.test.ts
index 7c7b7f24c..2237938d1 100644
--- a/tests/unit/core/test-judge-verdict.test.ts
+++ b/tests/unit/core/test-judge-verdict.test.ts
@@ -69,15 +69,15 @@ describe("interactiveCaseVerdict", () => {
});
it("maps a contestant crash to RE", () => {
- expect(
- interactiveCaseVerdict({ contestant: exited(1), interactor: exited(43) }, "eof"),
- ).toEqual({ verdict: "RE" });
+ expect(interactiveCaseVerdict({ contestant: exited(1), interactor: exited(43) })).toEqual({
+ verdict: "RE",
+ });
});
- it("maps interactor 43 to WA with the team message", () => {
- expect(
- interactiveCaseVerdict({ contestant: exited(0), interactor: exited(43) }, "wrong guess"),
- ).toEqual({ verdict: "WA", teamMessage: "wrong guess" });
+ it("maps interactor 43 to WA", () => {
+ expect(interactiveCaseVerdict({ contestant: exited(0), interactor: exited(43) })).toEqual({
+ verdict: "WA",
+ });
});
it("maps interactor 42 to AC", () => {
@@ -88,22 +88,19 @@ describe("interactiveCaseVerdict", () => {
it("maps an interactor that did not exit to SE", () => {
expect(
- interactiveCaseVerdict(
- { contestant: exited(0), interactor: { termination: "trap", code: 0 } },
- "partial",
- ),
+ interactiveCaseVerdict({
+ contestant: exited(0),
+ interactor: { termination: "trap", code: 0 },
+ }),
).toEqual({ verdict: "SE" });
});
it("keeps a contestant limit when the interactor then reads EOF and rejects", () => {
expect(
- interactiveCaseVerdict(
- {
- contestant: { termination: "logical-time-limit", code: 0 },
- interactor: exited(43),
- },
- "solution closed its output early",
- ),
+ interactiveCaseVerdict({
+ contestant: { termination: "logical-time-limit", code: 0 },
+ interactor: exited(43),
+ }),
).toEqual({ verdict: "TLE" });
});
@@ -125,9 +122,9 @@ describe("interactiveCaseVerdict", () => {
);
it("maps an interactor crash to SE when the contestant exited normally", () => {
- expect(
- interactiveCaseVerdict({ contestant: exited(0), interactor: exited(120) }, "partial"),
- ).toEqual({ verdict: "SE" });
+ expect(interactiveCaseVerdict({ contestant: exited(0), interactor: exited(120) })).toEqual({
+ verdict: "SE",
+ });
});
it("maps an interactor crash to SE when the contestant crashed", () => {
From 69ff517f7f3d73c2c362022f564bb2ed2fba5158 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Wed, 7 Oct 2026 21:00:56 +0800
Subject: [PATCH 21/22] docs: link decision sources to #645
Co-Authored-By: Claude Opus 5.5
---
docs/decisions/judge.md | 10 +++++-----
docs/decisions/platform.md | 4 ++--
docs/decisions/problems.md | 8 ++++----
docs/decisions/security.md | 4 ++--
docs/decisions/web.md | 2 +-
5 files changed, 14 insertions(+), 14 deletions(-)
diff --git a/docs/decisions/judge.md b/docs/decisions/judge.md
index 6a5b2459d..2cb731a11 100644
--- a/docs/decisions/judge.md
+++ b/docs/decisions/judge.md
@@ -27,7 +27,7 @@ The Standard Mode pipeline is implicit and fixed, not a list of configurable sta
### JDG-03 DOMjudge validator protocol for checkers and interactors, AC/WA only
-**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-13-judge-config-simplification-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-13-judge-config-simplification-design.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-06-13-domjudge-alignment](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-06-13-domjudge-alignment.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-13-judge-config-simplification-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-13-judge-config-simplification-design.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-06-13-domjudge-alignment](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-06-13-domjudge-alignment.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
Checkers and interactors are written in `python` or `cpp` only and follow the DOMjudge/Kattis protocol: `validator `, team output on stdin, exit 42 = AC, 43 = WA, anything else = SE; `teammessage.txt` reaches students, `judgemessage.txt` is staff-only. The product owner chose DOMjudge for standards compliance; undocumented argv/exit protocols forced boilerplate.
@@ -50,7 +50,7 @@ A TestcaseSet earns its full weight only if every case is AC, else 0, in practic
### JDG-05 Run/check separation: in official judging untrusted code never sees answers or validators
-**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-02-judge-pipeline-spec](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-02-judge-pipeline-spec.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-09-23-judge-single-sandbox-per-stage](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-judge-single-sandbox-per-stage.md), [PR #624](https://github.com/NOJV-TW/NOJV/pull/624), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-05, revised 2026-10 · **Source:** [2026-04-02-judge-pipeline-spec](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-02-judge-pipeline-spec.md), [2026-05-28-judge-isolation-domjudge-validator](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-28-judge-isolation-domjudge-validator.md), [2026-09-23-judge-single-sandbox-per-stage](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-judge-single-sandbox-per-stage.md), [PR #624](https://github.com/NOJV-TW/NOJV/pull/624), [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
In official judging, the container that runs student code never mounts expected answers, validator/interactor source or secret interactor input; checking happens in a separate container that starts only after the run container exits (interactive pairs a solution side with an interactor side that alone holds the secret data). A shared mount namespace once let programs read `expected.txt` and always get AC. Isolation must need no extra privileges and work on Docker and K8s.
@@ -170,7 +170,7 @@ With `WORKER_MIN_CONCURRENCY` set, the judge worker's activity slots come from a
### JDG-15 Test runs entirely in the browser, judge programs included
-**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-08-18-forge-judge-spike-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-08-18-forge-judge-spike-design.md), [2026-08-21-browser-local-run-npm-migration](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-21-browser-local-run-npm-migration.md), [2026-09-08-test-submit-parity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-08-test-submit-parity.md), [2026-09-09-test-reliability](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-09-test-reliability.md), [#639](https://github.com/NOJV-TW/NOJV/pull/639), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-08-18-forge-judge-spike-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-08-18-forge-judge-spike-design.md), [2026-08-21-browser-local-run-npm-migration](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-21-browser-local-run-npm-migration.md), [2026-09-08-test-submit-parity](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-08-test-submit-parity.md), [2026-09-09-test-reliability](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-09-09-test-reliability.md), [#639](https://github.com/NOJV-TW/NOJV/pull/639), [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
Test never creates a submission and runs entirely in the student's browser. The contestant program compiles and runs client-side in all eight languages (a firm requirement to keep cost off the server) using pinned `@wasm-oj/browser` and `@wasm-oj/toolchain-*`, same-origin assets verified by digest. Standard problems compare with the shared comparator. On checker and interactive problems the editor fetches the problem's checker or interactor source when it opens, builds it in the browser, and Test runs it: the checker on each sample case the program finished normally, the interactor against the program on every case through `interact`. Submit and Advanced stay on the server. Exact stdin bytes are preserved on both paths. #641 judged checker and interactive samples on a server test worker, which ran TA-authored judge programs, and on interactive problems the student's Wasm, with only the WASM-OJ runtime between them and the container's object-storage keys (every problem's hidden testcases) and Redis URL. Test is the student's own run, so the product owner moved all of it to the student's machine. Judge programs become readable by students: bugs such as a missing validity check or an off-by-one query limit get easier to find, but they exist whether or not the source is public, and authors own them.
@@ -319,6 +319,6 @@ A standard Kubernetes stage returns once its result is read and saved, freeing i
### JDG-26 Withdrawn: Test judging on its own queue and worker
-**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
-Withdrawn 2026-10 (#TBD): Test no longer runs anything on the server (JDG-15), so there is no Test queue, test worker or judge-program build cache. The separate worker kept Test off the CPU-bound judge queue, but it still ran TA judge programs and student Wasm next to Redis, object-storage and Temporal credentials with only the WASM-OJ runtime in between.
+Withdrawn 2026-10 (#645): Test no longer runs anything on the server (JDG-15), so there is no Test queue, test worker or judge-program build cache. The separate worker kept Test off the CPU-bound judge queue, but it still ran TA judge programs and student Wasm next to Redis, object-storage and Temporal credentials with only the WASM-OJ runtime in between.
diff --git a/docs/decisions/platform.md b/docs/decisions/platform.md
index 795da1a78..c590bddf4 100644
--- a/docs/decisions/platform.md
+++ b/docs/decisions/platform.md
@@ -236,6 +236,6 @@ Production pulls images over a ~0.5 MB/s uplink, and the web image was 1.1 GB co
### OPS-21 Withdrawn: the worker image carried the WASM-OJ runtime as stable layers
-**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
-Withdrawn 2026-10 (#TBD): with server-side Test gone (JDG-15) the worker image carries no WASM-OJ runtime or server toolchains, and Renovate no longer skips `@wasm-oj/*` or a `rust` builder image. The layers were pinned and timestamp-normalised because production pulls images over a ~0.5 MB/s uplink (OPS-20) and they cost about 82 MB compressed.
+Withdrawn 2026-10 (#645): with server-side Test gone (JDG-15) the worker image carries no WASM-OJ runtime or server toolchains, and Renovate no longer skips `@wasm-oj/*` or a `rust` builder image. The layers were pinned and timestamp-normalised because production pulls images over a ~0.5 MB/s uplink (OPS-20) and they cost about 82 MB compressed.
diff --git a/docs/decisions/problems.md b/docs/decisions/problems.md
index 071e4bafb..71328a083 100644
--- a/docs/decisions/problems.md
+++ b/docs/decisions/problems.md
@@ -4,7 +4,7 @@ Durable decisions for the problem model, authoring, publication, ownership, and
### PRB-01 Three problem types; workspace files instead of templates
-**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [2026-04-12-codebase-cleanup-audit](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-12-codebase-cleanup-audit.md), [2026-05-12-full-source-system-templates-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-12-full-source-system-templates-design.md), [2026-04-01-cp-problem-judge-mapping](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-01-cp-problem-judge-mapping.md), [#628](https://github.com/NOJV-TW/NOJV/pull/628), [#629](https://github.com/NOJV-TW/NOJV/pull/629), #TBD
+**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [2026-04-12-codebase-cleanup-audit](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-12-codebase-cleanup-audit.md), [2026-05-12-full-source-system-templates-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-12-full-source-system-templates-design.md), [2026-04-01-cp-problem-judge-mapping](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-01-cp-problem-judge-mapping.md), [#628](https://github.com/NOJV-TW/NOJV/pull/628), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
Problem types are `full_source`, `multi_file` and `special_env`. `multi_file` problems use `ProblemWorkspaceFile` (problem, language, path) with whole-file visibility `editable`/`readonly`; the server merges the student's editable files with the rest and judges the whole tree. `full_source` accepts every supported language with system `LANGUAGE_TEMPLATES` starters and no teacher starters. One model covers single-file, fill-in-function, library and multi-file problems without hidden wrapping code.
@@ -14,7 +14,7 @@ Problem types are `full_source`, `multi_file` and `special_env`. `multi_file` pr
- Rule: a `multi_file` language is allowed exactly when it ships an editable `main.`; there is no allowed-languages column, so the editor derives its ticks from entry files and drops unticked languages' files on save.
- Rule: published problems never change type through any path, and the workspace action refuses `special_env` problems, whose limits and config have their own guarded actions.
- Rule: workspace files give no confidentiality: students read every one in the editor or through Test, and student code can read them during judging. Do not put secrets or testcase answers in workspace files.
-- Rejected: a `hidden` visibility (removed 2026-10, #TBD). It gave no confidentiality, since official judging and student code read the file and browser Test would have to ship it; production had no hidden files on 2026-10-07; and authors took it for a secret. Earlier: isolating hidden workspace code from student execution; files that must run with student code stay inside that execution's trust boundary.
+- Rejected: a `hidden` visibility (removed 2026-10, [#645](https://github.com/NOJV-TW/NOJV/pull/645)). It gave no confidentiality, since official judging and student code read the file and browser Test would have to ship it; production had no hidden files on 2026-10-07; and authors took it for a secret. Earlier: isolating hidden workspace code from student execution; files that must run with student code stay inside that execution's trust boundary.
- Code: `packages/core/src/types.ts`, `packages/application/src/problem/details.ts`, `packages/core/src/language-templates.ts`
### PRB-02 Judge settings live in one validated `judgeConfig` JSON column
@@ -29,7 +29,7 @@ Eight scattered judge columns became one Zod-validated `Problem.judgeConfig` (ty
### PRB-03 Samples are presentation data, not testcases
-**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-04, revised 2026-10 · **Source:** [2026-04-09-problem-ui-redesign](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-04-09-problem-ui-redesign.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
Sample input/output pairs live in `Problem.samples` (JSON); every `TestcaseSet` is a judged subtask with weight ≥ 0. Samples are problem presentation, not grading data. A teacher may add a 0-point set (for example the sample cases) so every submission is judged on it without it adding points; a failing 0-point set still shows in the verdict. Publishing requires the subtask weights to total more than 0, and a problem that is published or used in an activity cannot have its set weights or sets changed so that the total drops to 0; an unused draft may pass through 0 while its subtasks are being built.
@@ -101,7 +101,7 @@ New problems start as `draft` (students may only create private drafts) so autho
### PRB-09 Publication requires a private, current reference solution
-**Decided:** 2026-08, revised 2026-10 · **Source:** [2026-08-08-reference-solution-validation](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-08-reference-solution-validation.md), [2026-08-15-reference-validation-editor-form](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-15-reference-validation-editor-form.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), #TBD
+**Decided:** 2026-08, revised 2026-10 · **Source:** [2026-08-08-reference-solution-validation](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-08-reference-solution-validation.md), [2026-08-15-reference-validation-editor-form](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/active/2026-08-15-reference-validation-editor-form.md), [#629](https://github.com/NOJV-TW/NOJV/pull/629), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
A standard problem publishes only with an accepted reference solution for the current judge configuration, authored in the editor's "Reference solution" section. It is an ordinary practice submission flagged `isReferenceSolution` against the full testcase set, pointed to by `Problem.referenceSolutionSubmissionId`, and tied to the problem's storage generation so any judge-affecting edit invalidates it. It validates the testcase and judge contract, not correctness.
diff --git a/docs/decisions/security.md b/docs/decisions/security.md
index b08913b37..4326b77c8 100644
--- a/docs/decisions/security.md
+++ b/docs/decisions/security.md
@@ -157,6 +157,6 @@ Run output is copied host-side by `safeCopyTree` (lstat first, drop symlinks and
### SEC-15 Withdrawn: server-judged Test ran only problem samples, from server-side data
-**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-10, withdrawn 2026-10 · **Source:** [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
-Withdrawn 2026-10 (#TBD): Test no longer judges anything on the server (JDG-15). The rule kept private judge programs from becoming an oracle for client-supplied cases; judge programs are now readable by students, so the browser judges a student's interactive cases with the real interactor. Test still never receives non-sample testcase data (SEC-12).
+Withdrawn 2026-10 (#645): Test no longer judges anything on the server (JDG-15). The rule kept private judge programs from becoming an oracle for client-supplied cases; judge programs are now readable by students, so the browser judges a student's interactive cases with the real interactor. Test still never receives non-sample testcase data (SEC-12).
diff --git a/docs/decisions/web.md b/docs/decisions/web.md
index e93250062..266197479 100644
--- a/docs/decisions/web.md
+++ b/docs/decisions/web.md
@@ -49,7 +49,7 @@ A `Notification` row is a persistent review-later event behind the navbar bell;
### WEB-05 The server holds the code draft of record, keyed by context, problem and language
-**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-05-11-code-draft-autosave-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-11-code-draft-autosave-design.md), [2026-09-23-server-code-drafts](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-server-code-drafts.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), #TBD
+**Decided:** 2026-09, revised 2026-10 · **Source:** [2026-05-11-code-draft-autosave-design](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-05-11-code-draft-autosave-design.md), [2026-09-23-server-code-drafts](https://github.com/NOJV-TW/NOJV/blob/f0347eb12ab7eb0b2269dcf774aff442f837bb85/docs/plans/completed/2026-09-23-server-code-drafts.md), [#641](https://github.com/NOJV-TW/NOJV/pull/641), [#645](https://github.com/NOJV-TW/NOJV/pull/645)
`CodeDraft` keyed by (user, contextKey, problem, language) holds unsubmitted code with autosave; the browser keeps only unacknowledged edits in a v2 local cache sealed with a per-user AES-GCM key and deletes them once acknowledged. Contexts (`practice`, `assignment:`, `exam:`, `contest:`, `virtual:`) never share drafts. Students lost code on reload, and plain localStorage lost exam code on shared lab PCs and could leak it to the next user.
From a87415694c14354a19142dbbdaf26f5fb016c214 Mon Sep 17 00:00:00 2001
From: TakalaWang
Date: Fri, 9 Oct 2026 16:06:12 +0800
Subject: [PATCH 22/22] build(web): pin @wasm-oj/browser 0.2.4
0.2.4 ships wasm-oj/forge#93, #94, #95, #96 and #99: runtime-bundle
interactors, in-module metering of interactive programs, the runtime-file
export stall fix and interactive sides in nested Workers. Its core and
contracts move to 0.2.4 with it. The toolchain packages stay at 0.2.0,
which forge versions independently.
Co-Authored-By: Claude Opus 5.5
---
apps/web/package.json | 2 +-
pnpm-lock.yaml | 28 ++++++++++++++--------------
pnpm-workspace.yaml | 6 +++---
3 files changed, 18 insertions(+), 18 deletions(-)
diff --git a/apps/web/package.json b/apps/web/package.json
index 83d9f6cdd..90c787a29 100644
--- a/apps/web/package.json
+++ b/apps/web/package.json
@@ -36,7 +36,7 @@
"@opentelemetry/exporter-metrics-otlp-http": "^0.222.0",
"@opentelemetry/resources": "^2.11.0",
"@opentelemetry/sdk-node": "^0.222.0",
- "@wasm-oj/browser": "0.2.3",
+ "@wasm-oj/browser": "0.2.4",
"@wasm-oj/toolchain-clang": "0.2.0",
"@wasm-oj/toolchain-go": "0.2.0",
"@wasm-oj/toolchain-java": "0.2.0",
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
index 5a707cf32..484bc06ab 100644
--- a/pnpm-lock.yaml
+++ b/pnpm-lock.yaml
@@ -266,8 +266,8 @@ importers:
specifier: ^0.222.0
version: 0.222.0(@opentelemetry/api@1.9.1)
'@wasm-oj/browser':
- specifier: 0.2.3
- version: 0.2.3
+ specifier: 0.2.4
+ version: 0.2.4
'@wasm-oj/toolchain-clang':
specifier: 0.2.0
version: 0.2.0
@@ -3524,20 +3524,20 @@ packages:
'@vitest/utils@4.1.11':
resolution: {integrity: sha512-zTCVGpyFsGWBhllOyKlTw/vnr6D9qxsfSDyfbyZmTyjHw5N/VuvzHpHoQjm2ZJzn4RJgx5w4r7V0er69CmLgPQ==}
- '@wasm-oj/browser@0.2.3':
- resolution: {integrity: sha512-M6k0Sjmr0bHeE8T9VNyL7B4uy2cF4d9pPUPa0g9kO2O91E6f22qZGMYWllnp6aQSEI8r8UvSdqFyO75IydBHZw==}
+ '@wasm-oj/browser@0.2.4':
+ resolution: {integrity: sha512-OT2dCZIKKO2mvfWxcXNvd/PKkv9zMARUVf2oTCnZfCIrQAhYSE0gSbdSg9LVXWRPo8gauQm8Frd5GeCGaHDuZQ==}
engines: {node: '>=24.18.0 <25'}
'@wasm-oj/contracts@0.2.2':
resolution: {integrity: sha512-Hvjx4xvkd45iRBIU3yAXOBkzLFtF33jSml9OMNtMHeEG6dCXPdlDz86QOveYLCj7xELH3db5vWx0hE6rFzaG7w==}
engines: {node: '>=24.18.0 <25'}
- '@wasm-oj/contracts@0.2.3':
- resolution: {integrity: sha512-a48ZH3NDVNAbCU+OzXW2aF/Sc0fgPQcNEVttajEJovi0t03il+lR8fbz1doevY4Zq+RL04o69FqYD0hz2j2a7A==}
+ '@wasm-oj/contracts@0.2.4':
+ resolution: {integrity: sha512-v0iYzF7cNecV1a2wlgsiUj1nWN4lhen2TPc+n1m7uNzjANX5LDGPDNXlkqUigWzlGlQ6URFI/yQWnbswlEXtww==}
engines: {node: '>=24.18.0 <25'}
- '@wasm-oj/core@0.2.3':
- resolution: {integrity: sha512-zM8jPrjLdRijIViqPRcDJyHKtPZlklkXvi3M8GfUmtk0OZb7ds+hTBkDIh06iYdDtisI0TKmS3u4ELl5m0k/pg==}
+ '@wasm-oj/core@0.2.4':
+ resolution: {integrity: sha512-lgYMHzwjNkbUpSD8hqJUMmx3VI9nARJiiu4bLeVZ8qwKAd5MpySQhyNT+TqRAexN9g/yjOPf9NtMVKc+CgSW/w==}
engines: {node: '>=24.18.0 <25'}
'@wasm-oj/toolchain-clang@0.2.0':
@@ -9847,21 +9847,21 @@ snapshots:
tinyrainbow: 3.1.1
optional: true
- '@wasm-oj/browser@0.2.3':
+ '@wasm-oj/browser@0.2.4':
dependencies:
- '@wasm-oj/contracts': 0.2.3
- '@wasm-oj/core': 0.2.3
+ '@wasm-oj/contracts': 0.2.4
+ '@wasm-oj/core': 0.2.4
'@wasmer/sdk': 0.10.0
es-module-lexer: 2.3.1
fflate: 0.8.3
'@wasm-oj/contracts@0.2.2': {}
- '@wasm-oj/contracts@0.2.3': {}
+ '@wasm-oj/contracts@0.2.4': {}
- '@wasm-oj/core@0.2.3':
+ '@wasm-oj/core@0.2.4':
dependencies:
- '@wasm-oj/contracts': 0.2.3
+ '@wasm-oj/contracts': 0.2.4
fflate: 0.8.3
'@wasm-oj/toolchain-clang@0.2.0':
diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml
index 50b0dc752..016e0fe8c 100644
--- a/pnpm-workspace.yaml
+++ b/pnpm-workspace.yaml
@@ -115,9 +115,9 @@ minimumReleaseAgeExclude:
- mysql2@3.23.1
- "@sveltejs/kit@2.70.2"
- "@hono/node-server@1.19.15"
- - "@wasm-oj/browser@0.2.3"
- - "@wasm-oj/contracts@0.2.2 || 0.2.3"
- - "@wasm-oj/core@0.2.3"
+ - "@wasm-oj/browser@0.2.4"
+ - "@wasm-oj/contracts@0.2.2 || 0.2.4"
+ - "@wasm-oj/core@0.2.4"
- "@wasm-oj/toolchain-clang@0.2.0"
- "@wasm-oj/toolchain-go@0.2.0"
- "@wasm-oj/toolchain-java@0.2.0"