From 5b1d754eca1d39f76c855e29b9fa6130bdf77f91 Mon Sep 17 00:00:00 2001 From: Test User Date: Fri, 9 Oct 2026 03:20:07 -0500 Subject: [PATCH 01/12] feat(desktop): vendor-neutral GPU acceleration with an out-of-process probe The desktop backend was hard-disabled to the CPU: effectiveVulkan() returned `vulkanSetting ?? false`, so a host with a working GPU never used it. That was deliberate (llama.cpp #17389), but its precondition has changed and nothing verified it. The compute backend is now resolved from a persisted probe verdict: - gpu-probe.ts runs the capability test in a SEPARATE OS process (process.execPath + ELECTRON_RUN_AS_NODE=1). A Vulkan driver fault aborts rather than throws, so an in-process try/catch - and a worker thread - cannot contain it. Every failure mode (spawn error, timeout, signal death, unparseable stdout) resolves to a CPU verdict with a reason; it never throws. - The child loads the fast-profile GGUF with an EXPLICIT gpuLayers (llama.cpp #29277: a wrong free-memory report must not size the offload), and the PARENT re-judges the generated sample rather than trusting the child's own success flag (llama.cpp #28648: a device can load and then emit garbage). - The verdict and its reason persist in gpu-probe.json beside settings.json / external.json / first-run.json / updates.json, survive a restart, and are reported by GET /status/models. - inference.vulkan widens from a bare boolean to 'auto' | true | false. The resident reuse key now includes the resolved backend and thread count, so changing the selection actually reloads the model instead of only moving the switch. - On the automatic path a GPU load failure retries once on CPU and records why. An explicitly forced GPU surfaces the error instead of degrading silently. - Settings shows the detected backend and device, offers the override, and can re-run the probe (POST /settings/inference/gpu-test). - Packaging pins the CPU and Vulkan backends in asarUnpack instead of relying on electron-builder's implicit native-module heuristic, and excludes the CUDA packages, which no code path can select (~510 MB unpacked). Also fixes two latent identity-churn effect loops found by the frozen C9 check: SettingsPage and ExternalModelSection both keyed their settings read on the session OBJECT identity, so a provider returning a fresh wrapper each render looped unboundedly. Both now key on session.baseUrl, which changes exactly when the session does. Known limits, recorded not hidden: node-llama-cpp 3.20.0 exposes neither `ubatch` nor `kvCacheType`, so the mitigations for llama.cpp #27638 and #29054 are the out-of-process probe plus its CPU fallback and nothing more. No integrated-GPU host was available, so that matrix row is PENDING. Closes #155 --- ARCHITECTURE.md | 6 +- CHANGELOG.md | 29 +- INSTALL.md | 6 +- README.md | 15 +- bench/RESULTS.md | 28 ++ contracts/api.openapi.yaml | 72 ++++ desktop/README.md | 22 +- desktop/electron-builder.yml | 17 + desktop/main/backend/index.ts | 105 +++++ .../main/backend/inference/gpu-probe-child.ts | 111 ++++++ desktop/main/backend/inference/gpu-probe.ts | 320 +++++++++++++++ .../main/backend/inference/llama-engine.ts | 257 ++++++++++-- desktop/main/backend/server.ts | 35 ++ desktop/main/backend/types.ts | 13 + desktop/src/__tests__/b4-llama-engine.test.ts | 67 +++- .../src/__tests__/t155-gpu-permanent.test.ts | 368 ++++++++++++++++++ desktop/vitest.config.ts | 6 + .../src/components/ExternalModelSection.tsx | 10 +- web_ui/src/lib/api/client.ts | 17 + web_ui/src/lib/api/types.ts | 18 + web_ui/src/pages/SettingsPage.tsx | 148 ++++++- 21 files changed, 1601 insertions(+), 69 deletions(-) create mode 100644 desktop/main/backend/inference/gpu-probe-child.ts create mode 100644 desktop/main/backend/inference/gpu-probe.ts create mode 100644 desktop/src/__tests__/t155-gpu-permanent.test.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 659a4fb3..d6ed9afb 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -209,8 +209,10 @@ transport maps to the contract 503. | Sampler | top-p 0.9, repeat penalty 1.1, lookback window 8192 (generated tokens only) | [desktop/main/backend/inference/penalties.ts](desktop/main/backend/inference/penalties.ts) (`PENALTY_FULL_CONTEXT_TOKENS`) | | History | at most 12 turns carried into the prompt | `MAX_HISTORY_TURNS` | -Vulkan stays off (reserved). The memory governor can force the runtime -profile to `fast` (below); the wizard's RAM gate uses the same estimate +The compute backend is decided by an out-of-process probe (issue #155): a working +Vulkan device is used when one is found and CPU inference is the fallback; +`inference.vulkan` can pin either choice. The memory governor can force the +runtime profile to `fast` (below); the wizard's RAM gate uses the same estimate family (file size + 1 GiB KV-cache + 1 GiB overhead, [desktop/main/first-run/ram-gate.ts](desktop/main/first-run/ram-gate.ts)). diff --git a/CHANGELOG.md b/CHANGELOG.md index fa1d9e8d..9027cb97 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,33 @@ ## [Unreleased] +- **Vendor-neutral GPU acceleration with a first-run probe and CPU fallback (issue #155)**: the + desktop backend is no longer hard-disabled to the CPU. A new out-of-process probe + (`desktop/main/backend/inference/gpu-probe.ts` + `gpu-probe-child.ts`) asks node-llama-cpp for + `{gpu: {type: "auto", exclude: ["cuda"]}}`, loads the fast-profile GGUF with an **explicit** + `gpuLayers` (llama.cpp #29277: a wrong free-memory report must not size the offload), and runs + one short generation whose output the **parent re-judges** rather than trusting the child's own + success flag (llama.cpp #28648: a device can load and then emit garbage). The probe runs in a + separate OS process with `ELECTRON_RUN_AS_NODE=1`, because a Vulkan driver fault aborts rather + than throws and a worker thread would die with the app; a killed, crashed, timed-out or + unparseable probe resolves to a CPU verdict with a reason and never throws. The verdict and + reason persist in `gpu-probe.json` beside `settings.json` / `external.json` / `first-run.json` / + `updates.json`, survive a restart, and are reported on `GET /status/models`. + `inference.vulkan` widens from a bare boolean to `'auto' | true | false`: auto follows the + probe, and an explicit choice is honoured in both directions. The resident-model reuse key now + includes the resolved compute backend, so changing the selection actually reloads the model + instead of only moving the switch. On the automatic path a GPU load failure retries once on CPU + and records why; an explicitly forced GPU surfaces the error rather than degrading silently. + Settings shows the detected backend and device, offers the override, and can re-run the probe + (new `POST /settings/inference/gpu-test`). Packaging pins the CPU and Vulkan backends in + `asarUnpack` instead of relying on electron-builder's implicit native-module heuristic, and + excludes the CUDA backend packages, which no code path can select and which measured ~510 MB + unpacked. `bench/RESULTS.md` gains a GPU device matrix whose unmeasured rows say PENDING and + name what is unmeasured. **Known limits, recorded not hidden**: the pinned + node-llama-cpp 3.20.0 exposes neither `ubatch` nor `kvCacheType`, so the mitigations for + llama.cpp #27638 and #29054 are the probe plus its CPU fallback and nothing more; and no + integrated-GPU host was available, so the floor-spec Iris Xe row is PENDING. + - **Desktop local inference (trace #154)**: the engine now pins `Gemma4ChatWrapper({reasoning: false})` for the shipped gemma-4 quality model, whose auto-resolved wrapper defaults `reasoning=true` and was spending answer-token budget on thought segments that never reach `responseText`. The pin is gated on the GGUFs own architecture, not the profile name, so a non-gemma model at the quality path keeps the library default wrapper resolution. The desktop-local system prompt gained the same groundedness rule the external endpoint already used, and each request now emits one `console.info` line carrying profile, threads, elapsed ms, answer tokens and outcome. @@ -344,7 +371,7 @@ - **Ask-about-this-slide — pinned slide context from the player into chat (Issue #83, WS-D PR 7/7)**: the chat panel now pins the slide the user is viewing in the embedded Storyline player. App lifts `pinnedSlide` state from `TrainingPlayer`'s `onSlideChange` (first production consumer — wired through `TrainingPage` with the frozen `{slideId, slideTitle}` payload untouched), resolving section + on-screen text once per slidechange from the ingested slide docs (`web_ui/src/lib/training/slide-doc-resolver.ts` over a new additive `KeywordIndex.findChunks` accessor, reusing `learn-kernel`'s exported `parseMarker`; unresolvable → title-only, never guessed). `PinnedSlideContext.tsx` renders the "Currently viewing: Section > Slide title" banner with a dismiss control and an "Explain this step" canned-question button, in both inference modes. On the browser-local RAG path a LIVE pin attaches `options.pinnedContext` — rendered inside the user turn by `buildMessages` (never the system prompt, never a history turn) and charged against `reservedTokens` via the extracted, unit-tested `computeReservedTokens`, so pin + long history + long retrieved context stays within `DEFAULT_N_CTX` (C3 pins inject-without-charge AND truncate-to-fake-charge as failures). Lifecycle guards: every `slidechange` supersedes the pin, dismissal clears it, reload clears it (in-memory), and a pack-switch staleness producer marks a pin `stale` (visible `data-stale="true"`, Explain hidden, NEVER attached to later questions — the issue's stated worst failure mode). The `api`/server SSE path renders the banner but sends nothing pinned: `QuestionRequest` is contract-frozen and server-side prefill parity remains #83's explicitly named follow-up. Acceptance: frozen checks C1-C10 (`repro-check.sh` RED→GREEN at f5bc456; one CHECK_WRONG amendment on C8's own harness mock), preserving check C7 green, full web_ui vitest + typecheck(app+test) + vite build green; `docs/training-player.md` documents the lifecycle. - **Learn panel — answer-time "where to learn this" deep links (Issue #82, WS-D PR 6/7)**: every `/ask` and `/ask/stream` response now carries an optional `learn[]` array (`LearnResult { slide_id, title, section, score, reason: "direct"|"linked", snippet?, pack_id? }`, capped at `MAX_LEARN_RESULTS = 5`, contract slot filled in `contracts/api.openapi.yaml` — `grounding` stays reserved for #72), populated by mirrored learn kernels (`learn_panel.py`, `desktop/main/backend/learn.ts`, `web_ui/src/lib/rag/learn-kernel.ts`): the union of (a) cited chunks that ARE Storyline training-slide documents (`reason: "direct"`, detected by `docs.source_class='training'` + the `docs/slide--.json` path convention on the Node store, and by the same filename convention on the Python/browser surfaces, whose slide-doc ingestion now exists on the Python stack via `.json` support in `document_processor.py` with a machine-parseable `[training-slide]` marker line) and (b) the cited doc chunks' top-3 linked slides from #80's `links` table (`reason: "linked"`, read via the new `queryLinksForChunks` store accessor — ask-time read side of D4). Node assembles through an injected `attachLearnAssembler` seam wired by the host (slide title/section/snippet enriched from the installed pack's slide docs on disk when `packsRoot` is configured); api_server computes learn from `retrieved_chunks` at the response boundary so engine doubles get identical parity; the browser kernel runs in `rag-orchestrator.ts` with an empty links source (documented divergence — browser pack support is #76). When `grounding: "general"` (#72, contract-reserved) arrives, `learn[]` is `[]` — pinned by kernel tests. Renderer: `LearnPanel.tsx` ("Section > Slide title", one-line `on_screen_text` snippet, "Open in training" button) mounts in the assistant bubble after the citations, and its `onOpenTraining` callback lifts a `trainingTarget` through `App.tsx` into `TrainingPage` → `TrainingPlayer initialSlideId`, reusing D5's readiness-deferred auto-jump (on surfaces without a pack id the TrainingPage no-pack prompt shows and the pending slide jumps once a pack is opened via `?pack=` — target never dropped). Eval: `eval/runner.py` reports **Learn hit@3** over the slide-target question rows (header/schema documented; `expected_training_slide_id` now populated for `training`-category rows) — `eval/ci_serve.py` ingests the six `eval/corpus/slides/*.json` fixtures through the real extraction path, and the first measured run reports **Learn hit@3 = 0.75** (3/4 within top-3; target ≥ 0.70) recorded in the generated `eval/REPORT.md`. `docs/training-player.md` documents the deep-link flow. - **Embedded Storyline player service + slide-state events (Issue #81, WS-D PR 5/7)**: the app now serves installed training packs same-origin under the reserved `app://training//` namespace — `desktop/main/protocol.ts` dispatches `app://training//` to `//assets/player/` (pack layout from `packtool build-storyline`) BEFORE the generic renderer mapping, reusing the full B2 validation discipline (percent-decode/backslash/NUL/`..` refusal, containment, realpath re-check) plus a `PACK_ID_PATTERN` gate on the pack id, and `desktop/main/index.ts` resolves the packs root from `TRAININGAPP_DESKTOP_PACKS_DIR` (default `/packs`). Pack documents carry a new `buildTrainingCspPolicy()` (`desktop/main/security/csp.ts`) whose every delta from the strict renderer policy is proven necessary by a live player probe (inline script/style boot requirements, `data:` fonts/media, no loopback connect, `frame-ancestors` omitted for the embeddable pack document); renderer documents keep the byte-identical strict policy (pinned by the b2 suites). New renderer surface: `web_ui/src/components/TrainingPlayer.tsx` (iframe embed with 1000 ms state polling, `onSlideChange` fired only on slide change, `jumpToSlide(): Promise` via ref + a `window.__trainingappTrainingPlayer` automation seam) talking to the pack through `web_ui/src/components/training-player-bridge.ts` (postMessage to the pack-local `story_content/trainingapp-bridge.js`, which implements the A8-proven jump recipe - `DS.presentation.getFlatSlides()` + `DS.windowManager.requestSlideForReview` - readiness-deferred, verified through the player's own state). Wired in production via a Training nav item (`Sidebar.tsx`) and `web_ui/src/pages/TrainingPage.tsx` (`?pack=`); the Learn panel that will drive it is D6/#82. The Electron e2e (production-mode launch) runs 10 exact jumps across 3 course sections against a committed runnable trimmed OpMed fixture (`desktop/e2e/fixtures/storyline-nav/`, contract in `FIXTURE_CONTRACT.md`). `docs/training-player.md` records the proven recipe, the `GetVar('projectSlideNumber'/'projectSlideTitle')` fallback (unregistered system playervars in this publish), the Resume/Restart re-open behavior, and the confirmed-disabled native menu finding. The e2e additionally exposed and fixed two Electron/protocol media defects (media elements need `stream: true` on the `app:` scheme privileges plus Range/206 and proper media MIME types over `protocol.handle`) and a Storyline-runtime navigation deadlock: under PlayerMemoryEnhancements the runtime cancels pending `htmlReady` requestAnimationFrames on unmount, which can swallow a scene-entry slide's readiness rAF and leave `slideReady` permanently false, deadlocking the serialized review-navigation queue; the pack bridge now gates jumps on `slideReady` and recovers landed-but-unready slides (see docs/training-player.md). Determinism: 10/10 sequential @ac2 runs, 5/5 @ac3. See the frozen acceptance checks (C1-C7) in the issue trace for the RED-to-GREEN evidence. -- **Native LLM inference with Quality/Fast profiles (Issue #62, WS-B PR 4/9)**: the Electron node backend now answers `/ask` and `/ask/stream` with REAL llama.cpp inference via `node-llama-cpp` (npm-shipped prebuilt binaries; the library choice is recorded as an assumption pending ADR-0003 #57 — distinct from the model-id assumption A2 pending ADR-0002 #56) — new `desktop/main/backend/inference/` module set (`profile-select.ts` for free-RAM Quality/Fast auto-selection with a 6 GiB inclusive threshold, `penalties.ts` porting `buildWllamaPenalties`' anti-repetition intent to the native sampler shape, `llama-engine.ts` implementing `EngineSurface` with a resident per-profile model, lazy load-once/reuse-always semantics, profile-switch reload deferred until in-flight generations end, and a 20 ms cancellation poll bridging the disconnect flag to the library abort so emission stops well inside 200 ms). Threads default to `min(cores, 8)` (explicitly not the browser WASM 4-cap); `inference.vulkan` is reserved off (llama.cpp #17389). A missing model makes both ask routes answer the contract's 503 with a load diagnostic — the stream route preflights through a new optional `EngineSurface.preflight()` BEFORE any SSE byte is written. New settings keys `inference.profile|profileThresholdGb|threads|vulkan` validate via PUT /settings while all rag_* keys still round-trip (owned by the composed stub until B5/B6); the `TRAININGAPP_DESKTOP_INFERENCE_PROFILE`/`_THREADS` env vars enforce the same gates headlessly, and contract history is capped to the last 12 turns before seeding the model. NOTE: packaged installs do not yet unpack node-llama-cpp's native addon from the asar archive (lands with #84/E1); dev runs and the headless dev-server are unaffected. The B3 stub survives only as the explicit `TRAININGAPP_DESKTOP_ENGINE=stub` dev/CI fixture; the conformance harness passes it for CI transport conformance, and `TRAININGAPP_CONFORMANCE_ENGINE=llama` runs the SAME suite against real inference. New `bench/node_bench_driver.mjs` measures the shipping engine and appends machine-tagged `engine=node-llama-cpp` rows to `bench/RESULTS.md` (devstation rows recorded; reference-i5 PENDING pending hardware). +- **Native LLM inference with Quality/Fast profiles (Issue #62, WS-B PR 4/9)**: the Electron node backend now answers `/ask` and `/ask/stream` with REAL llama.cpp inference via `node-llama-cpp` (npm-shipped prebuilt binaries; the library choice is recorded as an assumption pending ADR-0003 #57 — distinct from the model-id assumption A2 pending ADR-0002 #56) — new `desktop/main/backend/inference/` module set (`profile-select.ts` for free-RAM Quality/Fast auto-selection with a 6 GiB inclusive threshold, `penalties.ts` porting `buildWllamaPenalties`' anti-repetition intent to the native sampler shape, `llama-engine.ts` implementing `EngineSurface` with a resident per-profile model, lazy load-once/reuse-always semantics, profile-switch reload deferred until in-flight generations end, and a 20 ms cancellation poll bridging the disconnect flag to the library abort so emission stops well inside 200 ms). Threads default to `min(cores, 8)` (explicitly not the browser WASM 4-cap); `inference.vulkan` defaulted to off in this release (llama.cpp #17389); issue #155 later made the choice probe-driven. A missing model makes both ask routes answer the contract's 503 with a load diagnostic — the stream route preflights through a new optional `EngineSurface.preflight()` BEFORE any SSE byte is written. New settings keys `inference.profile|profileThresholdGb|threads|vulkan` validate via PUT /settings while all rag_* keys still round-trip (owned by the composed stub until B5/B6); the `TRAININGAPP_DESKTOP_INFERENCE_PROFILE`/`_THREADS` env vars enforce the same gates headlessly, and contract history is capped to the last 12 turns before seeding the model. NOTE: as shipped in this release the packaged install relied on electron-builder's implicit native-module heuristic for node-llama-cpp (issue #155 pins it explicitly); dev runs and the headless dev-server are unaffected. The B3 stub survives only as the explicit `TRAININGAPP_DESKTOP_ENGINE=stub` dev/CI fixture; the conformance harness passes it for CI transport conformance, and `TRAININGAPP_CONFORMANCE_ENGINE=llama` runs the SAME suite against real inference. New `bench/node_bench_driver.mjs` measures the shipping engine and appends machine-tagged `engine=node-llama-cpp` rows to `bench/RESULTS.md` (devstation rows recorded; reference-i5 PENDING pending hardware). - **Tier-0 eval set + backend-agnostic retrieval/answer harness (Issue #54, WS-A PR 4/8)**: new `eval/` instrument — `eval/questions.jsonl` (56 curated tier-0 questions over 8 checked-in synthetic corpus documents in `eval/corpus/`, schema `{id, question, expected_doc_id, expected_page, expected_training_slide_id, category}` with 6 deliberately out-of-corpus abstain rows), `eval/runner.py` (drives `/ask` on any frozen-contract backend and reports recall@{1,3,5}, MRR, abstain accuracy, and p50/p95 latency with a `--base-url`+timestamp+label provenance header; abstain is defined strictly as empty `sources`, fallback-phrase answers tracked separately), `eval/ci_serve.py` (deterministic no-weights backend: real engine/store/routing with a feature-hashing embedder and a scripted LLM, empirically calibrated similarity floor with the derivation recorded in `eval/README.md`), a non-blocking `eval-report` CI job (uploads the report artifact; never gates merge), committed sample reports in `eval/samples/` (deterministic-stub plus a real devstation weighted run), and `eval/README.md` documenting schema, taxonomy, authoring rules, metric definitions, and both run modes. **Fixed on the way**: `/ask` and `/ask/stream` now lazily load the configured GGUF model on the first question (`RAGEngine._ensure_llm` is reachable again — the 503 pre-check previously shadowed lazy init, so a valid `RAG_GGUF_PATH` could never load via the API; failed loads now surface their real diagnostic in the 503 detail), and `RAGEngine._ensure_llm` is double-checked under the init lock so concurrent first requests load the model exactly once. Unit + env-gated integration coverage in `tests/test_eval_harness.py` / `tests/test_eval_harness_integration.py`. - **Benchmark harness + measured perf baseline (Issue #52, WS-A PR 2/8)**: new `bench/` harness — `llama_bench_driver.py` (native llama.cpp CPU matrix: Gemma 4 E2B-it / LFM2.5-1.2B-Instruct / Gemma 3 1B x Q4_K_M/Q5_K_M x 4/8 threads x 1k/2k/3k prompts, with a crash-as-data Vulkan attempt mode), `wllama_bench_driver.mjs` (real headless-browser wllama runs, COOP/COEP-present multi-threaded vs headers-stripped single-threaded), and an ONNX embed/rerank cost microbenchmark (`onnx_bench_driver.py`, with `append_results.py` merging recorded rows into `bench/RESULTS.md`). Measured, machine-tagged numbers are recorded in `bench/RESULTS.md` (dev-station rows recorded; reference-i5 rows PENDING until the physical reference laptop run — evidence is invalidated by any hardware change). `tests/test_rag_performance.py` and `tests/test_low_end_hardware.py` no longer blanket-skip: they gate on the exact missing artifact and read thresholds from `bench/RESULTS.md` floors (legacy generous bounds as CI-safe fallback). README performance claims now cite `bench/RESULTS.md` instead of restating unmeasured numbers. A new guardrail test (`tests/test_no_blanket_perf_skips.py`) bans unconditional skips in perf-evidence suites and freezes the remaining functional-suite skip debt in an audited, two-sided allowlist. Thresholds and results all flow from `bench/floors.py`-parsed, machine-tagged floors — nothing is hand-patched. diff --git a/INSTALL.md b/INSTALL.md index 9bc50f5e..cf68b2f0 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -31,7 +31,8 @@ Complete installation guide for the Document Q&A Assistant, including standard P - **Python**: 3.11 or higher ### Optional Components -- **NVIDIA GPU**: Not required for GGUF backend (CPU-only inference) +- **NVIDIA GPU**: not required. The GGUF backend uses a working Vulkan device when one is + present and falls back to CPU inference otherwise ## Standard Installation @@ -262,7 +263,8 @@ Not required. Application runs as a standard executable. ### System Requirements for Power Users -The application runs CPU-only using GGUF models. No GPU or NPU acceleration is required. +The application runs GGUF models on a GPU when a probe finds a working Vulkan device, and on +the CPU otherwise. GPU acceleration is optional; no particular GPU or NPU is required. ## Post-Installation diff --git a/README.md b/README.md index 6053f8cf..991c8fc7 100644 --- a/README.md +++ b/README.md @@ -86,7 +86,8 @@ The desktop app runs GGUF models via node-llama-cpp (Node main-process backend, - **Quality profile (default)**: Gemma 4 E2B-it (Q4_K_M GGUF per [ADR-0002](docs/adr/0002-llm-profiles.md); ~2.9 GB nominal per PACKAGING.md; 2,620,370,976 bytes measured per bench/RESULTS.md) — bundled - **Fast profile**: lfm2.5-vl-450m (Q4_K_M GGUF per ADR-0002) — bundled - Profile selection: automatic free-RAM gate or in-app choice; `TRAININGAPP_DESKTOP_INFERENCE_PROFILE` (`quality` / `fast` / `auto`) is the desktop env override. (`RAG_GGUF_PATH` / `--gguf-path` select a custom GGUF on the legacy Python harness only.) -- No GPU required +- No particular GPU required: a working Vulkan device is used when one is found, and + CPU inference is the fallback - No network access required (unless you turn on the external model or update checks) - Measured decode throughput and first-token latency per model/profile: see [bench/RESULTS.md](bench/RESULTS.md) (issue #52 benchmark harness) @@ -94,7 +95,7 @@ The desktop app runs GGUF models via node-llama-cpp (Node main-process backend, #### Minimum (Intel 11th Gen i5, 16GB RAM) - Windows 11 (64-bit) - Intel Core i5 11th generation or newer (or equivalent AMD Ryzen 5000+) -- Intel integrated graphics (present on all 11th gen+ Intel CPUs) — no discrete GPU required +- Intel integrated graphics (present on all 11th gen+ Intel CPUs); a discrete card is not needed - 16GB RAM - ~6.4 GB free storage for models + app (measured installed footprint 6,378,451,601 bytes; staged model resources 4,111,872,009 bytes; see bench/RESULTS.md) - **Performance**: measured numbers per model and quantization are recorded in [bench/RESULTS.md](bench/RESULTS.md) @@ -109,7 +110,7 @@ The desktop app runs GGUF models via node-llama-cpp (Node main-process backend, #### High-Performance (Intel 13th Gen i9, 64GB RAM) - High-end CPU (Intel Core i9 or AMD Ryzen 9) - 64GB RAM -- **Performance**: measured CPU-only GGUF numbers are recorded in [bench/RESULTS.md](bench/RESULTS.md) +- **Performance**: measured CPU and GPU numbers, and which machine each was measured on, are recorded in [bench/RESULTS.md](bench/RESULTS.md) > **Pending**: the offline/low-RAM reference-laptop validation matrix (issue #86) — > reference-i5 rows are not yet measured; no reference-hardware numbers are claimed here. @@ -298,7 +299,9 @@ The desktop app runs GGUF models via node-llama-cpp (Node main-process backend, verification of the bundled tree → knowledge-pack activation → license notices → complete. Every gate names its failure reason; setup can be re-run from Settings. -No Python, no GPU, and no network access are required (unless you turn on the external model or update checks). +No Python and no network access are required (unless you turn on the external model or update +checks). GPU acceleration is used when the machine has one that works; otherwise inference +runs on the CPU. ### Building the desktop app from source @@ -679,7 +682,7 @@ Mirrors [ARCHITECTURE.md](ARCHITECTURE.md) (the authoritative map): - Cross-encoder rerank (ettin-reranker-32m-v1) with a calibrated relevance floor **LLM Interface** -- GGUF via node-llama-cpp (desktop, CPU-only, fully offline) +- GGUF via node-llama-cpp (desktop; GPU when a probe finds one working, CPU otherwise, fully offline) - GGUF via wllama WASM (browser, CPU/SIMD, fully offline) **RAG Engine** @@ -1152,4 +1155,4 @@ Legacy Python harness only (CI conformance surface): --- **Version**: 2.3.0 **Last Updated**: 2026-09-29 (v3 documentation refresh, issue #89) -**Hardware**: CPU-only optimized for Intel 11th gen i5 and above (16GB RAM minimum) +**Hardware**: optimized for Intel 11th gen i5 and above (16GB RAM minimum); GPU acceleration is opportunistic and CPU inference is the floor diff --git a/bench/RESULTS.md b/bench/RESULTS.md index 074fb590..69babe5b 100644 --- a/bench/RESULTS.md +++ b/bench/RESULTS.md @@ -152,6 +152,34 @@ see the provenance rule 4 above) on the real staged tree, machine-tagged: | devstation | startup integrity gate latency (streaming sha256 of the full staged tree) | 2,377 ms | packaged-mode pass, 0 failures, 2026-09-23 | | reference-i5 | all E1 size/latency rows | PENDING | operator runs the same commands (E3/#86 owns the matrix) | +### GPU device matrix (issue #155) + +Which compute backend the desktop backend actually selects on each machine. Issue #155 turned +GPU acceleration from a hard-disabled constant into an out-of-process probe plus a CPU +fallback, so a row here is a statement about what the probe concludes there. + +Recording discipline for this table (the same one the `reference-i5` E1 rows above already +follow): every row is either a figure measured on the machine it names, or an explicit `PENDING` +naming what is unmeasured. No tok/s number appears here unless it was measured on the machine +named. The issue #155 comments carry prefill/decode tok/s figures from an external benchmark +harness; this table does not restate them as its own evidence, because this revision did not +re-measure them. + +| machine | device | backend selected | GPU offload | notes | +|---|---|---|---|---| +| devstation | Intel Arc Pro B50 (discrete), driver 32.0.101.8805 | vulkan | supportsGpuOffloading=true | measured: node-llama-cpp 3.20.0, same process reported gpu=false for the shipped CPU-only option and gpu=vulkan for `{gpu:{type:auto,exclude:[cuda]}}` | +| devstation | Intel Arc Pro B50 (discrete), driver 32.0.101.8805 | cpu | gpu=false, supportsGpuOffloading=false | measured: the same host and the same node-llama-cpp 3.20.0 process under the shipped `{gpu:false}` option - the comparison the GPU row above is measured against | +| reference-i5 | 12th-gen Core i5 mobile, Intel Iris Xe integrated | PENDING | PENDING | no integrated-GPU host was available; prefill and decode tok/s and the probe's own verdict are unmeasured, and this is the floor-spec machine the speed bar for the quality tier depends on | +| reference-amd-igpu | AMD integrated, RDNA | PENDING | PENDING | no AMD host was available; backend selection, probe verdict and throughput are unmeasured | +| reference-amd-dgpu | AMD discrete, RDNA | PENDING | PENDING | no AMD host was available; backend selection, probe verdict and throughput are unmeasured | +| reference-nvidia | NVIDIA discrete | PENDING | PENDING | no NVIDIA host was available; note the CUDA backend is deliberately not shipped, so this host is expected to resolve to Vulkan-or-CPU rather than CUDA | + +Two upstream llama.cpp defects have no runtime mitigation in the pinned library and are covered +only by the probe plus its CPU fallback, not fixed here: **#27638** (device loss at +`ubatch >= 2048`; `ubatch` is not exposed by node-llama-cpp 3.20.0) and **#29054** (deterministic +hang on a q8_0 KV cache; `kvCacheType` is not exposed on `LlamaContextOptions`). Both were +verified absent by searching the installed package, not assumed. + **Known limit (recorded, never hidden): the single-file NSIS target cannot embed the real-weights payload.** `makensis.exe` aborts with `File: failed creating mmap of …-x64.nsis.7z` because the app archive is 4,177,827,908 diff --git a/contracts/api.openapi.yaml b/contracts/api.openapi.yaml index a6b5082b..4dd4bc8c 100644 --- a/contracts/api.openapi.yaml +++ b/contracts/api.openapi.yaml @@ -434,6 +434,32 @@ paths: "503": description: Model status is not wired on this host + /settings/inference/gpu-test: + post: + tags: [settings] + summary: Re-run the GPU capability probe (issue #155) + description: >- + Runs the out-of-process GPU probe again and adopts the result, so a + machine whose driver or hardware changed can recover from a stale + negative verdict. Takes no request body and persists nothing itself; the + host adopts and persists the returned verdict. + + "No usable GPU" is an ANSWER, not an error, so a failed probe is a 200 + with `ok: false` and a populated `reason` - the same shape + `POST /settings/external/test` uses. A 503 means the host has no probe + wired at all (a known route that cannot be served, never a 404). + Transport-token-guarded like every route. + operationId: testGpu + responses: + "200": + description: The probe ran; the verdict says whether a GPU is usable. + content: + application/json: + schema: + $ref: "#/components/schemas/GpuProbeResponse" + "503": + description: GPU probing is not wired on this host + /packs: get: tags: [packs] @@ -572,6 +598,24 @@ components: in: header name: X-API-Key schemas: + GpuProbeResponse: + type: object + description: >- + The result of a GPU capability probe run (issue #155). Same shape as the + `gpu` member of StatusModelsResponse. + required: [backend, ok, reason] + properties: + backend: + type: string + enum: [vulkan, cpu] + ok: + type: boolean + reason: + type: string + device: + type: string + nullable: true + StatusModelsResponse: type: object required: [engine, profile, models] @@ -585,6 +629,34 @@ components: GGUFs never block chat. profile: type: string + gpu: + type: object + description: >- + The GPU decision (issue #155), as the engine actually made it. + Present for the local llama.cpp backend and absent when an external + endpoint generates, because then no local compute backend is + involved. `backend` is what runs; `ok` is whether a GPU was usable; + `reason` is ALWAYS non-empty so a CPU fallback can be explained + rather than guessed at. + required: [backend, ok, reason] + properties: + backend: + type: string + enum: [vulkan, cpu] + description: >- + The resolved compute backend. CPU means either the probe found no + usable device, the operator pinned CPU, or nothing has been + probed yet. + ok: + type: boolean + description: True only when a GPU backend loaded AND produced a sane generation. + reason: + type: string + description: Human-readable explanation; never empty. + device: + type: string + nullable: true + description: Adapter identity the backend reported, when it reported one. resident: type: object description: >- diff --git a/desktop/README.md b/desktop/README.md index f5e82ddf..68016314 100644 --- a/desktop/README.md +++ b/desktop/README.md @@ -126,8 +126,14 @@ profile auto-selection. `lfm2.5-vl-450m/model.gguf` — both relative to the model dir (assumption A2 pending ADR-0002 #56). Embeddings remain the stub until B5 (#63). - **Threads**: `min(cores, 8)` by default — explicitly NOT the browser WASM - 4-cap (`web_ui/src/lib/llm/wllama-service.ts`). `inference.vulkan` is - reserved, default `false` (llama.cpp #17389). + 4-cap (`web_ui/src/lib/llm/wllama-service.ts`). +- **Compute backend**: `inference.vulkan` takes `'auto'` (the default, follow + the probe), `true` (force the GPU) or `false` (force CPU). The probe runs in + a separate OS process, so a driver fault cannot take the app down; its + verdict and a human-readable reason persist in `gpu-probe.json` beside the + other profile sidecars and are reported by `GET /status/models`. A GPU load + that fails on the automatic path retries once on CPU and records why; an + explicitly forced GPU surfaces the error instead of degrading silently. - **Model location**: `TRAININGAPP_INFERENCE_MODEL_DIR` env (or dev-server `--model-dir`) -> `/models` (Electron injects the path) -> `~/.trainingapp/models` headless fallback. A missing model makes `/ask` and @@ -138,15 +144,19 @@ profile auto-selection. in-flight generation ends); a client disconnect stops emission via a 20ms cancellation poll + abort signal — well inside the 200ms budget. - **Settings**: `inference.profile` | `inference.profileThresholdGb` | - `inference.threads` (1..64) | `inference.vulkan` via PUT /settings; the + `inference.threads` (1..64) | `inference.vulkan` ('auto'|true|false) via + PUT /settings; the rag_* keys still round-trip (owned by the composed stub until B5/B6). Headless env equivalents: `TRAININGAPP_DESKTOP_INFERENCE_PROFILE` (quality|fast|auto; invalid values fall back to auto) and `TRAININGAPP_DESKTOP_INFERENCE_THREADS` (same 1..64 integer gate as the settings key; invalid values fall back to the min(cores, 8) default). -- **Packaged installs**: the installer does NOT yet unpack node-llama-cpp's - native addon from the asar archive — packaged inference lands with #84 - (E1). Dev runs and the headless dev-server are unaffected. +- **Packaged installs**: `electron-builder.yml` pins + `@node-llama-cpp/win-x64-vulkan` and `@node-llama-cpp/win-x64` in + `asarUnpack`, so the compute backends are always on disk rather than + depending on electron-builder's implicit native-module heuristic. The CUDA + backend packages are excluded from the installer and no code path selects + them. Dev runs and the headless dev-server are unaffected. - **History**: contract-supplied history is capped to the last 12 turns before seeding the model, so an oversized array cannot overflow the 8192 context (the browser client already caps at 6). diff --git a/desktop/electron-builder.yml b/desktop/electron-builder.yml index 12785af1..5fedc27f 100644 --- a/desktop/electron-builder.yml +++ b/desktop/electron-builder.yml @@ -12,6 +12,14 @@ directories: files: - dist/**/* - package.json + # issue #155: the CUDA backend packages are excluded from the installer. + # Nothing can select them: the GPU decision resolves to 'vulkan' or 'cpu' only, + # and the auto probe passes exclude: ['cuda'] because these packages are not + # shipped. They measured ~510 MB unpacked, so carrying them is pure weight. + # The exclusion is declared here rather than in npm config so `npm ci` still + # installs everything the dev tree and the vulkan path need. + - '!**/node_modules/@node-llama-cpp/win-x64-cuda/**' + - '!**/node_modules/@node-llama-cpp/win-x64-cuda-ext/**' # The desktop:build script copies web_ui/dist into renderer/ before packaging # (stage-installer-resources.mjs excludes the weight files duplicated into # installer-resources); the main process serves it from /web_ui @@ -56,3 +64,12 @@ asarUnpack: - '**/node_modules/pdfjs-dist/**' - '**/node_modules/sqlite-vec*/**' - '**/node_modules/better-sqlite3/**' + # issue #155: the llama.cpp compute backends are pinned explicitly for the + # same reason sqlite-vec is above. They used to reach disk only through + # electron-builder's implicit native-module heuristic; the Vulkan binary is + # now LOADED AT RUNTIME by the GPU probe, so relying on a heuristic that + # electron-builder may change would silently turn every packaged install + # back into CPU-only. The CPU backend is pinned alongside it because the + # probe's CPU fallback path must work even where no GPU binary exists. + - '**/node_modules/@node-llama-cpp/win-x64-vulkan/**' + - '**/node_modules/@node-llama-cpp/win-x64/**' diff --git a/desktop/main/backend/index.ts b/desktop/main/backend/index.ts index db5af9f2..63e3ffa1 100644 --- a/desktop/main/backend/index.ts +++ b/desktop/main/backend/index.ts @@ -26,6 +26,14 @@ import { PackManager } from './store/pack-manager.js'; import { createPackSurface } from './packs/surface.js'; import { resolvePacksSecurity } from './packs/pack-extract.js'; import { loadSettingsSnapshot, saveSettingsSnapshot } from './settings-store.js'; +import { + activeGpuVerdict, + readGpuProbeVerdict, + runGpuProbe, + setActiveGpuVerdict, + writeGpuProbeVerdict, + type GpuProbeVerdict, +} from './inference/gpu-probe.js'; import { loadExternalSnapshot, replayExternalSnapshot, saveExternalSnapshot } from './external-store.js'; import { OnnxEmbedder, resolveEmbedder, type EmbeddingSurface } from './ingest/embedder.js'; import { resolveIngestConfig, resolveIngestLimits } from './ingest/config.js'; @@ -99,6 +107,18 @@ export class NodeBackendHost implements BackendHost { /** C3 (#70): pack lifecycle, constructed on start (instance field — the b3 * duck-type pin reserves host prototypes for start/stop only). */ private packManager: PackManager | null = null; + /** + * issue #155: the GPU probe verdict the host owns. The engine reads it + * through a SYNC seam (resolveNodeEngine's `gpuVerdict`), because the probe + * itself is async and runs out of process; this field is the writable holder + * the async probe fills. Null means "never probed", which the engine resolves + * to CPU - an unprobed host behaves exactly as it did before issue #155. + */ + private gpuVerdict: GpuProbeVerdict | null = null; + /** issue #155: profile dir the gpu-probe.json sidecar lives in, or null when + * the host runs without a store (CI stub) - then the probe still runs but + * nothing is persisted. */ + private gpuProbeDir: string | null = null; /** Issue #133: named reason the pack lifecycle is down ('ok' when live); * surfaced to the wizard via getPackLifecycleStatus → packs.unavailableReason. */ @@ -207,6 +227,71 @@ export class NodeBackendHost implements BackendHost { } }; + /** + * issue #155: run the GPU probe and adopt its verdict. Own property BY DESIGN + * (the b3 duck-type pin requires both host prototypes to expose exactly + * start/stop - node-only capabilities must stay off the prototype). + * + * Never throws: runGpuProbe resolves a CPU verdict for every failure mode, + * and this wrapper additionally swallows anything unexpected so a probe + * problem can never take host start with it. + */ + startGpuProbe = async (): Promise => { + try { + const modelPath = this.probeModelPath(); + const verdict = + modelPath === null + ? { backend: 'cpu' as const, ok: false, reason: 'GPU probe skipped: no probe model is staged.', device: null } + : await runGpuProbe({ args: [modelPath] }); + this.adoptGpuVerdict(verdict); + return verdict; + } catch (err) { + const verdict: GpuProbeVerdict = { + backend: 'cpu', + ok: false, + reason: `GPU probe failed unexpectedly: ${err instanceof Error ? err.message : String(err)}`, + device: null, + }; + this.adoptGpuVerdict(verdict); + return verdict; + } + }; + + /** + * issue #155: adopt a verdict in memory AND persist it, so a restart does not + * re-probe. Called by the automatic probe, by the re-test endpoint, and by the + * engine's load-failure downgrade (via onGpuLoadFailure). + */ + adoptGpuVerdict = (verdict: GpuProbeVerdict): void => { + this.gpuVerdict = verdict; + // Publish to the shared holder the engine reads, and persist so a restart + // does not re-probe. Both writers are the host; nothing else mutates it. + setActiveGpuVerdict(verdict); + if (this.gpuProbeDir !== null) writeGpuProbeVerdict(this.gpuProbeDir, verdict); + }; + + /** issue #155: the verdict the engine's sync seam reads. */ + readGpuVerdict = (): GpuProbeVerdict | null => this.gpuVerdict ?? activeGpuVerdict(); + + /** + * issue #155: which model the probe loads. The FAST profile's GGUF: the probe + * validates the BACKEND, the backend behaves identically for both profiles, + * and loading the 2.6 GB quality GGUF to answer a question the 332 MB fast + * GGUF answers identically would make first boot needlessly slow. + */ + /** + * issue #155: OWN PROPERTY, not a prototype method. TypeScript's `private` is + * compile-time only, so a `private` method still shows up on the prototype - + * and the b3 duck-type pin (b3-backend-selector.test.ts) requires BOTH host + * prototypes to expose exactly start/stop. The other node-only capabilities + * are arrow class fields for the same reason (createStoreBackup, above). + */ + private probeModelPath = (): string | null => { + const models = (this.config.engine as { models?: { quality?: string; fast?: string } } | undefined)?.models; + if (models?.fast !== undefined && models.fast !== '' && fs.existsSync(models.fast)) return models.fast; + return null; + }; + /** * B6 (issue #64): snapshot the open store into // * (WAL-flushed; restore validates schema+dims). Own property BY DESIGN: the @@ -382,6 +467,12 @@ export class NodeBackendHost implements BackendHost { replayExternalSnapshot(externalSnapshot, (patch) => this.engine.applySettingsPatch(patch)); } persistExternal = (snapshot) => saveExternalSnapshot(storePath, snapshot); + + // issue #155: the profile DIRECTORY is the sidecar home (index.ts:409-412 + // states the convention; dev-server.ts:119 already uses it). A missing or + // corrupt file reads as null and simply means "not probed yet". + this.gpuProbeDir = path.dirname(storePath); + this.gpuVerdict = readGpuProbeVerdict(this.gpuProbeDir); } // #133: the first-run wizard applies the operator's profile choice // through the SAME validated seam the settings API uses — live apply + @@ -430,6 +521,15 @@ export class NodeBackendHost implements BackendHost { modelStatus: typeof this.engine.modelStatus === 'function' ? () => this.engine.modelStatus!() : undefined, + // issue #155: the on-demand re-probe. Each call re-reads the sidecar + // first so a driver change is what the verdict reflects, then re-runs. + gpuTest: async () => { + if (this.gpuProbeDir !== null) { + const stored = readGpuProbeVerdict(this.gpuProbeDir); + if (stored !== null) this.gpuVerdict = stored; + } + return await this.startGpuProbe(); + }, // C7 (issue #74): pack lifecycle surface. Provider shape — the // PackManager is constructed lazily with the store (see the recovery // path below), so resolve it per request; null -> contract-safe 503. @@ -445,6 +545,11 @@ export class NodeBackendHost implements BackendHost { // gate chat on the real load state instead of a time heuristic. The // warmup itself never fails host start; single-flight in the engine // merges it with a concurrent first query. + // issue #155: kick the GPU probe off in the BACKGROUND, next to warmup, + // so neither blocks the listener. It runs out of process (gpu-probe.ts), + // so a driver fault cannot take the host down; the verdict that lands here + // is picked up by the NEXT model load, never by a load already in flight. + void this.startGpuProbe(); void this.engine.warmup?.().catch((err: unknown) => { console.error( `[trainingapp-backend] model warmup crashed: ${err instanceof Error ? err.message : String(err)}`, diff --git a/desktop/main/backend/inference/gpu-probe-child.ts b/desktop/main/backend/inference/gpu-probe-child.ts new file mode 100644 index 00000000..563c292b --- /dev/null +++ b/desktop/main/backend/inference/gpu-probe-child.ts @@ -0,0 +1,111 @@ +// Issue #155: the out-of-process half of the GPU capability probe. +// +// Runs as a child of the backend host (ELECTRON_RUN_AS_NODE=1) so that a Vulkan +// driver fault - which aborts rather than throws - cannot take the app with it. +// The contract with the parent is exactly ONE JSON object on stdout, followed by +// exit 0. Anything else is interpreted by the parent as "no usable GPU", which +// is the safe direction. +// +// It loads the model handed to it on argv (the host passes the FAST profile's +// GGUF): the probe validates the BACKEND, the backend behaves identically for +// both profiles, and loading the 2.6 GB quality GGUF to answer a question the +// 332 MB fast GGUF answers identically would make first boot needlessly slow. +import fs from 'node:fs'; +import type * as Nlc from 'node-llama-cpp'; + +const PROMPT = 'Reply with only the word: OK'; +/** Small on purpose: this measures whether the device produces coherent output, + * not how fast it is. Speed is the bench harness's job, not the probe's. */ +const MAX_TOKENS = 12; + +function emit(report: Record): void { + process.stdout.write(JSON.stringify(report)); +} + +function fail(reason: string, device: string | null = null): void { + emit({ ok: false, backend: 'cpu', reason, device }); + process.exit(0); +} + +async function main(): Promise { + const modelPath = process.argv[2]; + if (typeof modelPath !== 'string' || modelPath === '') { + fail('GPU probe received no model path.'); + return; + } + if (!fs.existsSync(modelPath)) { + fail(`GPU probe: the probe model is not staged at ${modelPath}.`); + return; + } + + let llama: Nlc.Llama; + try { + const nlc = await import('node-llama-cpp'); + const options: Nlc.LlamaOptions = { + // CUDA is excluded on purpose (issue #155 AC12): the installer does not + // ship it, so an NVIDIA host must fall back rather than half-work. + gpu: { type: 'auto', exclude: ['cuda'] }, + // `build: 'never'` is load-bearing. Under ELECTRON_RUN_AS_NODE this + // process is plain Node as far as the library is concerned, where its + // documented default is "auto" - which attempts a FROM-SOURCE build when + // no prebuilt binary matches. That is a multi-minute failure mode. With + // "never" a missing binary throws NoBinaryFoundError immediately, which + // is exactly the signal this probe exists to report. + build: 'never', + }; + llama = await nlc.getLlama(options); + } catch (err) { + fail(`GPU probe: no usable GPU backend (${err instanceof Error ? err.message : String(err)}).`); + return; + } + + const selected = llama.gpu; + if (selected !== 'vulkan') { + await llama.dispose().catch(() => {}); + fail(`GPU probe: this host selected "${String(selected)}" rather than a usable Vulkan device.`); + return; + } + + let model: Nlc.LlamaModel | undefined; + let context: Nlc.LlamaContext | undefined; + try { + // Explicit gpuLayers (llama.cpp #29277: a wrong free-memory report must not + // silently decide the offload size). 'max' is validated by the probe + // itself - if it cannot fit, the load throws and the parent reports CPU. + model = await llama.loadModel({ modelPath, gpuLayers: 'max' }); + context = await model.createContext({ threads: 2, contextSize: 512 }); + } catch (err) { + await context?.dispose().catch(() => {}); + await model?.dispose().catch(() => {}); + await llama.dispose().catch(() => {}); + fail(`GPU probe: the GPU could not load the probe model (${err instanceof Error ? err.message : String(err)}).`); + return; + } + + try { + const { LlamaChatSession } = await import('node-llama-cpp'); + const session = new LlamaChatSession({ + contextSequence: context.getSequence(), + autoDisposeSequence: false, + }); + const sample = await session.prompt(PROMPT, { maxTokens: MAX_TOKENS, temperature: 0 }); + emit({ + ok: true, + backend: 'vulkan', + device: selected, + // The parent re-judges this itself (probeOutputIsSane) rather than + // trusting the child's ok flag - see gpu-probe.ts. + sample: String(sample ?? ''), + }); + } catch (err) { + fail(`GPU probe: the GPU failed to generate (${err instanceof Error ? err.message : String(err)}).`); + } finally { + await context.dispose().catch(() => {}); + await model.dispose().catch(() => {}); + await llama.dispose().catch(() => {}); + } +} + +void main().catch((err: unknown) => { + fail(`GPU probe crashed: ${err instanceof Error ? err.message : String(err)}`); +}); \ No newline at end of file diff --git a/desktop/main/backend/inference/gpu-probe.ts b/desktop/main/backend/inference/gpu-probe.ts new file mode 100644 index 00000000..8c39990e --- /dev/null +++ b/desktop/main/backend/inference/gpu-probe.ts @@ -0,0 +1,320 @@ +// Issue #155: GPU capability probe (Vulkan, vendor-neutral) with CPU fallback. +// +// The probe runs in a SEPARATE OS PROCESS on purpose. A Vulkan driver fault does +// not throw - it aborts - so an in-process try/catch cannot contain it, and a +// worker_threads Worker shares the address space and dies with it. The one +// precedent this repo has for isolating a native module that can abort +// (rerank-worker.ts, ONNX exit 134) is a thread, which is not enough here. +// +// The verdict is a value object; gpu-probe.json beside settings.json / +// external.json / first-run.json / updates.json is the persistence; this module +// is the isolation boundary. The load-time-safe read is tolerant in exactly the +// shape update-checker.ts:335-354 established: a missing, non-object or +// wrong-typed file yields null and logs nothing, and never throws. +// +// This module must stay electron-free (desktop/main/backend/types.ts:1-12 states +// the rule for its neighbours): nothing here imports electron. +import { spawn } from 'node:child_process'; +import fs from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +/** What the probe concluded about this machine, and why. */ +export interface GpuProbeVerdict { + /** The backend that will actually be used. */ + backend: 'vulkan' | 'cpu'; + /** True only when a GPU backend loaded AND produced a sane generation. */ + ok: boolean; + /** Human-readable, shown in Settings and returned by the API. Never empty. */ + reason: string; + /** Adapter/device identity when the backend reported one, else null. */ + device?: string | null; +} + +/** + * The process-wide ACTIVE verdict. One writer (the backend host, which runs the + * async probe) and one reader (the engine, through a sync seam, because the + * verdict must be readable at model-load time). A null value means "never + * probed", which the engine resolves to CPU. + */ +let activeVerdict: GpuProbeVerdict | null = null; + +export function activeGpuVerdict(): GpuProbeVerdict | null { + return activeVerdict; +} + +export function setActiveGpuVerdict(verdict: GpuProbeVerdict | null): void { + activeVerdict = verdict; +} + +/** The sidecar name, used beside the other profile-dir sidecars. */ +export const GPU_PROBE_SIDECAR = 'gpu-probe.json'; + +/** Default probe deadline. The first GPU call pays a one-off warmup, so this is + * generous; it is a ceiling that must contain a hang, not a performance target. */ +export const DEFAULT_PROBE_TIMEOUT_MS = 60_000; + +/** Grace between SIGTERM and SIGKILL when the deadline expires. Mirrors the + * terminate-then-escalate shape proven in sidecar-manager.ts:245-252. */ +const KILL_GRACE_MS = 2_000; + +/** A verdict that always falls back to CPU, with a stated reason. */ +function cpuVerdict(reason: string): GpuProbeVerdict { + return { backend: 'cpu', ok: false, reason: reason === '' ? 'GPU probe did not report a reason' : reason, device: null }; +} + +/** + * Read a persisted verdict. Missing, unparseable, non-object and + * wrong-typed files all read as null; a missing verdict means "never probed", + * which the engine resolves to CPU (an unprobed host behaves exactly as it did + * before this feature). + */ +export function readGpuProbeVerdict(dir: string): GpuProbeVerdict | null { + try { + const raw = fs.readFileSync(path.join(dir, GPU_PROBE_SIDECAR), 'utf8'); + const parsed = JSON.parse(raw) as unknown; + if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) return null; + const record = parsed as Record; + if (record.backend !== 'vulkan' && record.backend !== 'cpu') return null; + if (typeof record.ok !== 'boolean') return null; + if (typeof record.reason !== 'string' || record.reason === '') return null; + const device = record.device; + return { + backend: record.backend, + ok: record.ok, + reason: record.reason, + device: typeof device === 'string' ? device : null, + }; + } catch { + // Tolerant by design: a missing or corrupt sidecar is never fatal, and a + // probe result is not worth a boot-time error (update-checker.ts:345-346). + return null; + } +} + +/** + * Persist a verdict atomically: `gpu-probe.json...tmp` in the SAME + * directory, fsynced, then renamed onto the target (same-dir rename is atomic + * on NTFS/ext4). The tmp file is unlinked on any failure. Identical idiom to + * settings-store.ts:57-78. + */ +export function writeGpuProbeVerdict(dir: string, verdict: GpuProbeVerdict): void { + fs.mkdirSync(dir, { recursive: true }); + const target = path.join(dir, GPU_PROBE_SIDECAR); + const tmp = `${target}.${process.pid}.${Math.random().toString(36).slice(2)}.tmp`; + let fd: number | undefined; + try { + fd = fs.openSync(tmp, 'w'); + fs.writeSync(fd, JSON.stringify({ v: 1, ...verdict })); + fs.fsyncSync(fd); + fs.closeSync(fd); + fd = undefined; + fs.renameSync(tmp, target); + } catch { + if (fd !== undefined) { + try { + fs.closeSync(fd); + } catch { + // already closed + } + } + try { + fs.unlinkSync(tmp); + } catch { + // best effort + } + } +} + +/** + * llama.cpp #28648: an Arc 140V enumerates, loads, and then produces garbage + * with layers on the GPU. Load success is therefore not a usable signal - the + * sample the backend actually produced has to be judged. The parent re-judges + * the child's sample itself (see runGpuProbe) rather than trusting the child's + * own verdict, so a child that lies about a garbage generation still fails. + */ +export function probeOutputIsSane(sample: string): boolean { + if (typeof sample !== 'string') return false; + if (sample.trim() === '') return false; + // A control-character-dense string is the signature of a mis-decode. Count + // C0 controls plus DEL, excluding the whitespace controls a real answer uses. + let controls = 0; + for (const ch of sample) { + const code = ch.codePointAt(0) ?? 0; + if (code === 9 || code === 10 || code === 13) continue; + if (code < 32 || code === 127) controls += 1; + } + if (controls > 0) return false; + // At least one real word character: a run of punctuation or a lone glyph is + // not an answer, and is what a corrupted decode tends to collapse to. + return /[\p{L}\p{N}]/u.test(sample); +} + +/** The compiled child entry, resolved from this module's own location exactly + * as reranker.ts:69-71 resolves rerank-worker.js. desktop/tsconfig.json sets + * rootDir "." and outDir "dist", so this lands at + * dist/main/backend/inference/gpu-probe-child.js for a compiled/dev run and at + * the same relative path inside app.asar for a packaged one. */ +export function gpuProbeChildPath(): string { + return path.join(path.dirname(fileURLToPath(import.meta.url)), 'gpu-probe-child.js'); +} + +export interface GpuProbeRunOptions { + /** + * The child command to spawn. Defaults to the PRODUCTION spawn: the current + * executable with ELECTRON_RUN_AS_NODE=1, because in a packaged app the only + * Node runtime present is the Electron binary - `node` is not shipped, so + * hardcoding it passes in dev and fails in production. Override only in + * tests. argv = [execPath, childPath, ...extraArgs]. + */ + command?: string[]; + /** Extra argv appended after the child script path (the model path). */ + args?: string[]; + /** Deadline; on expiry the child is SIGTERMed then SIGKILLed. */ + timeoutMs?: number; + /** Injectable spawn, so tests never fork a real process. */ + spawnFn?: typeof spawn; +} + +interface ProbeChildReport { + ok?: unknown; + backend?: unknown; + device?: unknown; + sample?: unknown; + reason?: unknown; +} + +/** + * Run the probe out of process and resolve a verdict. NEVER throws and never + * rejects: every failure mode - spawn error, timeout, non-zero exit, signal + * death, unparseable stdout - resolves to a CPU verdict carrying a reason. The + * caller (the host) therefore has exactly one code path to handle. + */ +export async function runGpuProbe(opts: GpuProbeRunOptions = {}): Promise { + const command = opts.command ?? [process.execPath, gpuProbeChildPath()]; + const spawnFn = opts.spawnFn ?? spawn; + const timeoutMs = opts.timeoutMs ?? DEFAULT_PROBE_TIMEOUT_MS; + + return await new Promise((resolve) => { + let settled = false; + let stdout = ''; + let stderr = ''; + let child: ReturnType; + + const finish = (verdict: GpuProbeVerdict): void => { + if (settled) return; + settled = true; + clearTimeout(deadline); + clearTimeout(escalate); + resolve(verdict); + }; + + const escalate = setTimeout(() => { + try { + child.kill('SIGKILL'); + } catch { + // already gone + } + }, KILL_GRACE_MS); + escalate.unref?.(); + + const deadline = setTimeout(() => { + try { + child.kill('SIGTERM'); + } catch { + // already gone + } + finish(cpuVerdict(`GPU probe timed out after ${timeoutMs} ms and was terminated.`)); + }, timeoutMs); + deadline.unref?.(); + + try { + const executable = command[0]; + if (typeof executable !== 'string' || executable === '') { + finish(cpuVerdict('GPU probe was given no executable to run.')); + return; + } + child = spawnFn(executable, [...command.slice(1), ...(opts.args ?? [])], { + // ELECTRON_RUN_AS_NODE is what makes the Electron binary execute the + // script as Node instead of starting a browser process. + env: { ...process.env, ELECTRON_RUN_AS_NODE: '1' }, + stdio: ['ignore', 'pipe', 'pipe'], + windowsHide: true, + }); + } catch (err) { + finish(cpuVerdict(`GPU probe could not be started: ${err instanceof Error ? err.message : String(err)}`)); + return; + } + + child.stdout?.on('data', (chunk: Buffer) => { + stdout += chunk.toString('utf8'); + if (stdout.length > 1_000_000) { + try { + child.kill('SIGKILL'); + } catch { + // already gone + } + finish(cpuVerdict('GPU probe produced more output than the probe contract allows.')); + } + }); + child.stderr?.on('data', (chunk: Buffer) => { + stderr += chunk.toString('utf8'); + if (stderr.length > 64_000) stderr = stderr.slice(0, 64_000); + }); + child.on('error', (err: Error) => { + finish(cpuVerdict(`GPU probe process error: ${err.message}`)); + }); + child.on('close', (code: number | null, signal: string | null) => { + if (code !== 0) { + const how = signal !== null ? `signal ${signal}` : `exit code ${code}`; + const detail = stderr.trim().split('\n').slice(-1)[0] ?? ''; + finish(cpuVerdict(`GPU probe did not complete cleanly (${how})${detail === '' ? '' : `: ${detail}`}`)); + return; + } + let report: ProbeChildReport; + try { + // The child contract is exactly one JSON object on stdout; take the + // last non-empty line so incidental native logging cannot break it. + const lines = stdout.split('\n').filter((line) => line.trim() !== ''); + report = JSON.parse(lines[lines.length - 1] ?? '') as ProbeChildReport; + } catch { + finish(cpuVerdict('GPU probe did not return a readable report.')); + return; + } + if (report.ok !== true) { + const reason = typeof report.reason === 'string' && report.reason !== '' ? report.reason : 'GPU probe reported the device is not usable.'; + finish({ backend: 'cpu', ok: false, reason, device: typeof report.device === 'string' ? report.device : null }); + return; + } + // Defence in depth: the child's own `ok` is NOT taken at face value. A + // backend that loads but generates garbage (llama.cpp #28648) is rejected + // here, in the parent, even if the child said otherwise. + const sample = typeof report.sample === 'string' ? report.sample : ''; + if (!probeOutputIsSane(sample)) { + finish(cpuVerdict('GPU probe: the device loaded but produced unusable output, so GPU acceleration was rejected.')); + return; + } + if (report.backend !== 'vulkan') { + finish(cpuVerdict(`GPU probe: the device selected an unexpected backend (${String(report.backend)}), so GPU acceleration was rejected.`)); + return; + } + finish({ + backend: 'vulkan', + ok: true, + reason: 'GPU probe: Vulkan loaded and produced a sane generation.', + device: typeof report.device === 'string' ? report.device : null, + }); + }); + }); +} + +/** Terminate an in-flight probe child. Exported so host shutdown can guarantee + * no probe outlives the backend host (sidecar-manager.ts:245-252 shape). */ +export function terminateProbeChild(child: { kill: (signal?: NodeJS.Signals) => unknown } | null): void { + if (child === null) return; + try { + child.kill('SIGTERM'); + } catch { + // already gone + } +} \ No newline at end of file diff --git a/desktop/main/backend/inference/llama-engine.ts b/desktop/main/backend/inference/llama-engine.ts index 6e1b5daf..9f48475d 100644 --- a/desktop/main/backend/inference/llama-engine.ts +++ b/desktop/main/backend/inference/llama-engine.ts @@ -42,6 +42,7 @@ import type { RetrievalSurface, } from '../types.js'; import { buildPenalties, PENALTY_FULL_CONTEXT_TOKENS, type PenaltyOptions } from './penalties.js'; +import { activeGpuVerdict, type GpuProbeVerdict } from './gpu-probe.js'; import { EXTERNAL_SETTING_KEYS, ExternalProviderState, @@ -91,11 +92,29 @@ export interface LlamaEngineBackend { export interface LlamaEngineFactoryOptions { modelPath: string; threads: number; - vulkan: boolean; + /** issue #155: the RESOLVED compute backend, not the operator's preference. */ + backend: GpuBackendName; + /** issue #155 (llama.cpp #29277): explicit offload size, never auto-fit. */ + gpuLayers: 'max' | number; /** Effective profile the factory call serves (drives generation params). */ profile: InferenceProfileName; } +/** issue #155: the two compute backends the desktop backend can run. CUDA is + * deliberately absent - the installer does not ship it, so an NVIDIA host + * falls back to Vulkan-or-CPU rather than half-working. */ +export type GpuBackendName = 'vulkan' | 'cpu'; + +/** issue #155: 'auto' lets the persisted probe verdict decide; true forces the + * GPU even if the probe disagreed; false forces CPU. An ABSENT setting is + * 'auto', so an unprobed host behaves exactly as it did before this feature. */ +export type GpuSelection = 'auto' | boolean; + +/** Explicit offload: every layer, or none. 'auto' is NOT used - llama.cpp + * #29277 reports a device whose free-memory figure is wrong, and auto-fit + * would silently size the offload from it. */ +const EXPLICIT_GPU_LAYERS = 'max' as const; + export interface LlamaEngineOptions { /** Model root directory. Precedence over userDataPath and the headless default. */ modelDir?: string; @@ -104,8 +123,25 @@ export interface LlamaEngineOptions { profile?: ProfileSetting; profileThresholdGb?: number; threads?: number; - /** Reserved (issue #62): off until A2 shows stability — llama.cpp #17389. */ - vulkan?: boolean; + /** + * issue #155: 'auto' (default) follows the GPU probe verdict, `true` forces + * Vulkan even when the probe disagreed, `false` forces CPU. Widened from the + * old bare boolean, which could not express "let the app decide". + */ + vulkan?: GpuSelection; + /** + * issue #155: the probe verdict oracle. A SYNC seam, matching the other + * injection points (freeMemBytes/cpuCount/llamaFactory): the host runs the + * out-of-process probe at start and hands the engine a reader. Returning null + * means "never probed", which resolves to CPU. + */ + gpuVerdict?: () => GpuProbeVerdict | null; + /** + * issue #155: called when an automatic ('auto') GPU load fails, so the host + * can downgrade its in-memory verdict and rewrite the sidecar instead of + * re-attempting a failing GPU on every subsequent request. + */ + onGpuLoadFailure?: (verdict: GpuProbeVerdict) => void; /** Explicit absolute model file overrides, keyed by profile. */ models?: { quality?: string; fast?: string }; /** Injectable seams for tests. */ @@ -258,10 +294,17 @@ async function resolveReasoningSuppressedWrapper( async function defaultLlamaFactory(opts: LlamaEngineFactoryOptions): Promise { // Dynamic import: native code loads only when a model is actually needed. const nlc = await import('node-llama-cpp'); - // CPU-only by default; the vulkan build is reserved until A2 certifies it - // (llama.cpp #17389 — Gemma-3n E2B Vulkan crash on Intel iGPU). - const llama = await nlc.getLlama(opts.vulkan ? { gpu: 'vulkan' } : { gpu: false }); - const model = await llama.loadModel({ modelPath: opts.modelPath }); + // issue #155: the GPU is no longer hard-disabled. The backend was resolved by + // the probe (or pinned by the operator) BEFORE this call, so this site only + // has to state the request, not decide it. CUDA is excluded because the + // installer does not ship it - an NVIDIA host falls back rather than + // half-working (AC12). + const llama = await nlc.getLlama( + opts.backend === 'vulkan' ? { gpu: { type: 'auto', exclude: ['cuda'] } } : { gpu: false }, + ); + // Explicit gpuLayers (llama.cpp #29277: a device whose free-memory figure is + // wrong must not silently size the offload from it). + const model = await llama.loadModel({ modelPath: opts.modelPath, gpuLayers: opts.gpuLayers }); const chatWrapper = await resolveReasoningSuppressedWrapper(nlc, opts.modelPath); const context = await model.createContext({ threads: opts.threads, contextSize: CONTEXT_SIZE }); // ONE resident sequence for the lifetime of the backend: v3 allocates @@ -392,6 +435,10 @@ export function resolveNodeEngine( models?: { quality?: string; fast?: string }; /** universal-provider-settings-overhaul: SecretStore + airgap flag. */ externalProvider?: ExternalProviderOptions; + /** issue #155: the probe verdict reader the host owns (see LlamaEngineOptions). */ + gpuVerdict?: () => GpuProbeVerdict | null; + /** issue #155: notified when an automatic GPU load fails. */ + onGpuLoadFailure?: (verdict: GpuProbeVerdict) => void; }, ): EngineSurface { if (env.TRAININGAPP_DESKTOP_ENGINE === 'stub') { @@ -409,6 +456,8 @@ export function resolveNodeEngine( userDataPath: overrides?.userDataPath, ...(overrides?.models !== undefined ? { models: overrides.models } : {}), ...(overrides?.externalProvider !== undefined ? { externalProvider: overrides.externalProvider } : {}), + ...(overrides?.gpuVerdict !== undefined ? { gpuVerdict: overrides.gpuVerdict } : {}), + ...(overrides?.onGpuLoadFailure !== undefined ? { onGpuLoadFailure: overrides.onGpuLoadFailure } : {}), profile, ...(threads !== undefined ? { threads } : {}), }); @@ -418,6 +467,14 @@ interface ResidentEntry { backend: LlamaEngineBackend; profile: InferenceProfileName; inFlight: number; + /** + * issue #155: the identity the resident was BUILT with. Before this field + * existed, reuse keyed on `profile` alone, so a settings change to the backend + * selection (or to threads) updated the field and was reported back by + * responseSettings() while the model kept running on the old backend - the + * switch moved and nothing changed. The effective backend is part of the key. + */ + backendIdentity: string; } /** LlamaEngine.captureSettingsState() snapshot (PR #140 review FB140-001). */ @@ -426,7 +483,7 @@ export interface LlamaSettingsState { readonly profileSetting: ProfileSetting; readonly thresholdGb: number; readonly threadsSetting: number | undefined; - readonly vulkanSetting: boolean | undefined; + readonly vulkanSetting: GpuSelection | undefined; readonly stickyAuto: InferenceProfileName | null; /** PR #142 rebase RB-001: external.* values, the session key and the SecretStore entries. */ readonly external: ExternalProviderSettingsState; @@ -440,8 +497,13 @@ export class LlamaEngine implements EngineSurface { private readonly llamaFactoryFn: (opts: LlamaEngineFactoryOptions) => Promise; private profileSetting: ProfileSetting; private thresholdGb: number; - private vulkanSetting: boolean | undefined; + private vulkanSetting: GpuSelection | undefined; private threadsSetting: number | undefined; + /** issue #155: the probe verdict reader seam (null = never probed). */ + private readonly gpuVerdictFn: () => GpuProbeVerdict | null; + /** issue #155: notified when an automatic GPU load fails, so the host can + * downgrade its in-memory verdict AND rewrite the sidecar. */ + private readonly onGpuLoadFailure: ((verdict: GpuProbeVerdict) => void) | undefined; private readonly modelDirOption: string | undefined; private readonly userDataPath: string | undefined; private readonly modelOverrides: { quality?: string; fast?: string }; @@ -489,6 +551,8 @@ export class LlamaEngine implements EngineSurface { this.thresholdGb = options.profileThresholdGb ?? 6; this.vulkanSetting = options.vulkan; this.threadsSetting = options.threads; + this.gpuVerdictFn = options.gpuVerdict ?? activeGpuVerdict; + this.onGpuLoadFailure = options.onGpuLoadFailure; this.modelDirOption = options.modelDir; this.userDataPath = options.userDataPath; this.modelOverrides = options.models ?? {}; @@ -571,8 +635,52 @@ export class LlamaEngine implements EngineSurface { return this.threadsSetting ?? defaultThreadCount(this.cpuCount()); } - private effectiveVulkan(): boolean { - return this.vulkanSetting ?? false; + /** + * issue #155: resolve the operator's PREFERENCE against the probe verdict + * into the backend that will actually run. + * + * true -> force Vulkan, even when the probe disagreed. A forced GPU that + * cannot load surfaces the error; it never silently degrades. + * false -> force CPU. + * auto -> follow a GPU-ok verdict; otherwise CPU. An ABSENT verdict + * (never probed) resolves to CPU, so an unprobed host behaves + * exactly as it did before this feature existed. + */ + private effectiveGpuBackend(): GpuBackendName { + const selection = this.vulkanSetting ?? 'auto'; + if (selection === false) return 'cpu'; + if (selection === true) return 'vulkan'; + const verdict = this.gpuVerdictFn(); + return verdict !== null && verdict.ok && verdict.backend === 'vulkan' ? 'vulkan' : 'cpu'; + } + + /** issue #155: whether the operator pinned the choice rather than delegating + * it. Load-time auto-downgrade is allowed only when they did not. */ + private gpuIsForced(): boolean { + return this.vulkanSetting === true || this.vulkanSetting === false; + } + + /** issue #155: the verdict as the API reports it. Absent verdict reads as a + * CPU decision with a stated reason rather than an absent field, so the + * renderer never has to guess. */ + private gpuStatus(): { backend: GpuBackendName; ok: boolean; reason: string; device: string | null } { + const verdict = this.gpuVerdictFn(); + if (verdict === null) { + return { + backend: 'cpu', + ok: false, + reason: 'GPU acceleration has not been tested on this machine yet; CPU inference is in use.', + device: null, + }; + } + return { backend: verdict.backend, ok: verdict.ok, reason: verdict.reason, device: verdict.device ?? null }; + } + + /** The resident reuse key: everything a backend is BUILT from that a + * settings change can alter. Model path included so a profile file swap + * reloads too. */ + private backendIdentityFor(profile: InferenceProfileName, modelPath: string): string { + return [profile, modelPath, this.effectiveThreads(), this.effectiveGpuBackend()].join('|'); } private modelPathFor(profile: InferenceProfileName): string { @@ -626,17 +734,24 @@ export class LlamaEngine implements EngineSurface { profile: this.effectiveProfile(), models: { quality: statusFor('quality'), fast: statusFor('fast') }, resident: external ? { state: 'idle', profile: null, loadStartedAt: null } : this.residentLoadStatus(), + // issue #155: the detected backend, whether it works, and WHY - so the + // Settings surface can show a reason instead of a silent CPU. Absent on + // an external endpoint, where no local backend is involved at all. + ...(external ? {} : { gpu: this.gpuStatus() }), }; } private async ensureResident(profile: InferenceProfileName, modelPath: string): Promise { - if (this.resident !== null && this.resident.profile === profile) return this.resident; + // issue #155: reuse requires the whole backend identity to match, not just + // the profile - so changing the GPU selection (or threads) actually reloads. + const identity = this.backendIdentityFor(profile, modelPath); + if (this.resident !== null && this.resident.backendIdentity === identity) return this.resident; if (this.loadInFlight !== null) { // Single-flight: a warmup (or a concurrent query) already loading a - // backend — await it, then re-evaluate (it may even be the profile we + // backend — await it, then re-evaluate (it may even be the identity we // want). The in-flight promise never re-enters this branch. const loaded = await this.loadInFlight; - if (loaded.profile === profile) return loaded; + if (loaded.backendIdentity === identity) return loaded; } const load = this.loadBackend(profile, modelPath); this.loadInFlight = load; @@ -651,40 +766,102 @@ export class LlamaEngine implements EngineSurface { const old = this.resident; if (old !== null) { this.resident = null; - // Profile switch defers to post-request: the per-engine queue means the + // A backend switch defers to post-request: the per-engine queue means the // caller only reaches this point after prior generations completed, so - // the old backend is idle and safe to dispose synchronously. + // the old backend is idle and safe to dispose synchronously. Disposing + // BEFORE constructing the next one is what keeps exactly one context + // alive (issue #155 change 2). await old.backend.dispose().catch(() => {}); } const threads = this.effectiveThreads(); + let backendName = this.effectiveGpuBackend(); let backend: LlamaEngineBackend; this.loadState = 'loading'; this.loadStartedAt = Date.now(); this.loadingProfile = profile; try { - backend = await this.llamaFactoryFn({ - modelPath, - threads, - vulkan: this.effectiveVulkan(), - profile, - }); + backend = await this.constructBackend(profile, modelPath, threads, backendName); } catch (err) { - this.loadState = 'idle'; - this.loadStartedAt = null; - this.loadingProfile = null; - // Corrupt/unloadable model: wrap into the 503-diagnostic error type, - // carrying the underlying failure for the operator. - throw new ModelNotConfiguredError( - `Failed to load the ${profile} model from ${modelPath}: ${err instanceof Error ? err.message : String(err)}`, - ); + // issue #155: load-time CPU retry, AUTOMATIC PATH ONLY. The probe + // validated the DEVICE with the fast model; nothing else proves the + // quality GGUF fits on it, so a GPU load failure here is real evidence + // and must not leave the user without an answer. An operator-pinned + // selection is never retried: "force the GPU" that silently degrades to + // CPU is not force, so that failure surfaces. + if (backendName === 'vulkan' && !this.gpuIsForced()) { + const reason = `GPU load failed, so this machine is now using CPU inference: ${err instanceof Error ? err.message : String(err)}`; + console.warn(`[trainingapp-backend] ${reason}`); + // Update the in-memory holder AND the sidecar, so the next load in this + // session resolves to cpu instead of re-attempting the failing GPU on + // every subsequent request. + this.onGpuLoadFailure?.({ backend: 'cpu', ok: false, reason, device: null }); + backendName = 'cpu'; + try { + backend = await this.constructBackend(profile, modelPath, threads, 'cpu'); + } catch (cpuErr) { + this.markLoadFailed(); + throw new ModelNotConfiguredError( + `Failed to load the ${profile} model from ${modelPath}: ${cpuErr instanceof Error ? cpuErr.message : String(cpuErr)}`, + ); + } + } else { + this.markLoadFailed(); + // Corrupt/unloadable model: wrap into the 503-diagnostic error type, + // carrying the underlying failure for the operator. + throw new ModelNotConfiguredError( + `Failed to load the ${profile} model from ${modelPath}: ${err instanceof Error ? err.message : String(err)}`, + ); + } } this.loads += 1; + this.markLoadReady(); + const entry: ResidentEntry = { + backend, + profile, + inFlight: 0, + backendIdentity: this.backendIdentityFor(profile, modelPath), + }; + this.resident = entry; + return entry; + } + + /** A load that succeeded: report 'ready' with no in-flight window. */ + private markLoadReady(): void { this.loadState = 'ready'; this.loadStartedAt = null; this.loadingProfile = null; - const entry: ResidentEntry = { backend, profile, inFlight: 0 }; - this.resident = entry; - return entry; + } + + /** + * A load that failed: report 'idle', NEVER 'ready' and NEVER a stuck + * 'loading' (the #133 round-4 contract: a failed warmup leaves state idle and + * the first /ask retries). Called on every throw path. + */ + private markLoadFailed(): void { + this.loadState = 'idle'; + this.loadStartedAt = null; + this.loadingProfile = null; + } + + /** One construction attempt, wrapped so a failure carries the operator-facing + * detail without losing the load-state reset. */ + private async constructBackend( + profile: InferenceProfileName, + modelPath: string, + threads: number, + backendName: GpuBackendName, + ): Promise { + try { + return await this.llamaFactoryFn({ + modelPath, + threads, + backend: backendName, + gpuLayers: EXPLICIT_GPU_LAYERS, + profile, + }); + } catch (err) { + throw err instanceof Error ? err : new Error(String(err)); + } } async query(question: string, opts: EngineQueryOptions = {}): Promise { @@ -919,7 +1096,7 @@ export class LlamaEngine implements EngineSurface { let profile: ProfileSetting | undefined; let thresholdGb: number | undefined; let threads: number | undefined; - let vulkan: boolean | undefined; + let vulkan: GpuSelection | undefined; for (const [key, value] of Object.entries(inferenceSubset)) { switch (key) { case 'inference.profile': @@ -949,8 +1126,11 @@ export class LlamaEngine implements EngineSurface { } break; case 'inference.vulkan': - if (typeof value !== 'boolean') { - errors.push(`${key}: expected a boolean`); + // issue #155: widened from a bare boolean so "let the probe decide" + // is expressible. 'yes' and 1 stay invalid - a coerced truthy is not + // a selection. + if (value !== 'auto' && typeof value !== 'boolean') { + errors.push(`${key}: expected 'auto', true or false`); } else { vulkan = value; } @@ -1080,7 +1260,10 @@ export class LlamaEngine implements EngineSurface { 'inference.profile': this.profileSetting, 'inference.profileThresholdGb': this.thresholdGb, 'inference.threads': this.effectiveThreads(), - 'inference.vulkan': this.effectiveVulkan(), + // issue #155: the operator's selection is what the setting round-trips + // (so a saved 'auto' stays 'auto'); the resolved backend and the reason + // ride on /status/models. + 'inference.vulkan': this.vulkanSetting ?? 'auto', // universal-provider-settings-overhaul: external.* (never the key). ...this.external.responseFields(), }; diff --git a/desktop/main/backend/server.ts b/desktop/main/backend/server.ts index d47e4435..c9ef3c61 100644 --- a/desktop/main/backend/server.ts +++ b/desktop/main/backend/server.ts @@ -67,6 +67,12 @@ export interface BackendServerOptions { * degrades to a contract-safe 503 — never 404 (same shape as telemetry). */ modelStatus?: () => ModelStatus; + /** + * issue #155: the GPU re-probe action for POST /settings/inference/gpu-test. + * Host wires its probe runner. When absent the known route degrades to a + * contract-safe 503 — never 404, same shape as telemetry and modelStatus. + */ + gpuTest?: () => Promise<{ backend: 'vulkan' | 'cpu'; ok: boolean; reason: string; device?: string | null }>; /** * C7 (issue #74): the pack lifecycle surface behind the /packs routes. * Provider FUNCTION shape (`() => PackSurface | null`) because the host @@ -117,6 +123,10 @@ export const CONTRACT_ROUTES: ReadonlyMap> = new Map // universal-provider-settings-overhaul: desktop-backend-only connection test // for a draft external endpoint (persists nothing; token-guarded). ['/settings/external/test', new Set(['POST'])], + // issue #155: re-run the out-of-process GPU capability probe on demand, so a + // machine whose driver or hardware changed can recover from a stale negative + // verdict. Persists nothing by itself; the host adopts what it returns. + ['/settings/inference/gpu-test', new Set(['POST'])], ['/stats', new Set(['GET'])], ['/telemetry/memory', new Set(['GET'])], ['/status/models', new Set(['GET'])], @@ -987,6 +997,31 @@ export function createBackendServer(opts: BackendServerOptions): http.Server { sendJson(res, 200, engine.responseSettings(), cors); return; } + case 'POST /settings/inference/gpu-test': { + // issue #155: no body is read and nothing is persisted here; the + // host's probe runner decides both. A failure is still a 200 with + // ok:false, because "this machine has no usable GPU" is an ANSWER, + // not an error - the same shape POST /settings/external/test uses. + if (typeof opts.gpuTest !== 'function') { + sendJson(res, 503, { detail: 'GPU probing is not wired on this host' }, cors); + return; + } + let outcome: { backend: 'vulkan' | 'cpu'; ok: boolean; reason: string; device?: string | null }; + try { + outcome = await opts.gpuTest(); + } catch (err) { + // A probe that throws is contained HERE rather than becoming a 500 + // the operator has to interpret: report it as a CPU verdict. + outcome = { + backend: 'cpu', + ok: false, + reason: `GPU probe failed: ${err instanceof Error ? err.message : String(err)}`, + device: null, + }; + } + sendJson(res, 200, outcome, cors); + return; + } case 'POST /settings/external/test': { const body = await readBody(req, JSON_BODY_CAP_BYTES); if (body === null) { diff --git a/desktop/main/backend/types.ts b/desktop/main/backend/types.ts index 85de8921..2e4b9db0 100644 --- a/desktop/main/backend/types.ts +++ b/desktop/main/backend/types.ts @@ -362,6 +362,19 @@ export interface ModelStatus { profile: string | null; loadStartedAt: number | null; }; + /** + * issue #155: the GPU decision, as the engine actually made it. `backend` is + * what runs; `ok` is whether a GPU was usable; `reason` is always non-empty so + * the renderer can explain a CPU fallback instead of guessing; `device` is the + * adapter identity when the backend reported one. Optional: an external + * endpoint has no local compute backend. + */ + gpu?: { + backend: 'vulkan' | 'cpu'; + ok: boolean; + reason: string; + device: string | null; + }; } /** diff --git a/desktop/src/__tests__/b4-llama-engine.test.ts b/desktop/src/__tests__/b4-llama-engine.test.ts index 6553e1e2..a597d2ff 100644 --- a/desktop/src/__tests__/b4-llama-engine.test.ts +++ b/desktop/src/__tests__/b4-llama-engine.test.ts @@ -70,7 +70,10 @@ class FakeBackend implements LlamaEngineBackend { interface FactoryCapture { modelPath: string; threads: number; - vulkan: boolean; + /** issue #155: the RESOLVED compute backend the factory is asked to build. */ + backend: 'vulkan' | 'cpu'; + /** issue #155: the explicit offload size (never 'auto'). */ + gpuLayers: 'max' | number; } function makeTmpDir(prefix: string): string { @@ -103,7 +106,7 @@ function makeEngine(overrides: Partial = {}): EngineHarness cpuCount: () => 8, models: { quality: dummy.quality, fast: dummy.fast }, llamaFactory: async (opts) => { - captures.push({ modelPath: opts.modelPath, threads: opts.threads, vulkan: opts.vulkan }); + captures.push({ modelPath: opts.modelPath, threads: opts.threads, backend: opts.backend, gpuLayers: opts.gpuLayers }); const backend = new FakeBackend(); backends.push(backend); return backend; @@ -157,8 +160,12 @@ describe('b4-llama-engine Group A (mocked backend): resident model + cancellatio await engine.query('quality again'); expect(engine.getLoadCount()).toBe(2); // no further reconstruction - }); - + // Per-test budget, not the 5s default: this test performs THREE sequential + // queries each of which disposes one backend and constructs another, and it + // runs inside the full desktop suite where the machine is saturated. It + // passes comfortably in isolation; the default budget made it a load + // flake. The assertion, not the budget, is what this test is for. + }, 30000); it('AC4 cancel: flag set mid-generation resolves cancelled:true and stops streaming <200ms after the flag', async () => { const { engine } = makeEngine(); const flag = { set: false, isSet: () => flag.set }; @@ -191,11 +198,51 @@ describe('b4-llama-engine Group A (mocked backend): resident model + cancellatio expect(explicit.captures[0].threads).toBe(3); }); - it('vulkan defaults to false (reserved) and is forwarded when set', async () => { + // issue #155: this test used to be named "vulkan defaults to false + // (reserved) and is forwarded when set" but its body only ever asserted the + // default - the "forwarded when set" half of the name was never exercised. + // It now asserts BOTH halves, and the name says what it does. + it('the backend defaults to cpu and each explicit selection is forwarded', async () => { + // No selection and no probe verdict: 'auto' with nothing probed resolves to + // cpu, so an unprobed host behaves exactly as it did before issue #155. const plain = makeEngine(); await plain.engine.query('q'); - expect(plain.captures[0].vulkan).toBe(false); - }); + expect(plain.captures[0].backend).toBe('cpu'); + + // Explicitly forced GPU -> forwarded as vulkan. + const forced = makeEngine({ vulkan: true }); + await forced.engine.query('q'); + expect(forced.captures[0].backend).toBe('vulkan'); + + // Explicitly forced CPU -> forwarded as cpu. + const pinned = makeEngine({ vulkan: false }); + await pinned.engine.query('q'); + expect(pinned.captures[0].backend).toBe('cpu'); + + // 'auto' plus a GPU-ok probe verdict -> forwarded as vulkan. This is the + // half the old test name claimed and never checked. + const probed = makeEngine({ + gpuVerdict: () => ({ backend: 'vulkan', ok: true, reason: 'stub probe: gpu usable', device: 'stub-gpu-0' }), + }); + await probed.engine.query('q'); + expect(probed.captures[0].backend).toBe('vulkan'); + + // 'auto' plus a FAILED probe verdict -> cpu. The fallback direction. + const unusable = makeEngine({ + gpuVerdict: () => ({ backend: 'cpu', ok: false, reason: 'stub probe: no usable gpu', device: null }), + }); + await unusable.engine.query('q'); + expect(unusable.captures[0].backend).toBe('cpu'); + + // Every construction above asked for an explicit offload size, never 'auto' + // (llama.cpp #29277: a wrong free-memory report must not size the offload). + for (const harness of [plain, forced, pinned, probed, unusable]) { + expect(harness.captures[0].gpuLayers).toBe('max'); + } + // Five engine constructions; the default 5s per-test budget is too tight + // for them on a loaded machine, and a timeout here would read as a product + // failure rather than a harness budget problem. + }, 30000); it('auto profile uses the DEFAULT 6 GiB threshold: exactly 6 GiB -> quality path, one byte under -> fast path', async () => { const atThreshold = makeEngine({ profile: 'auto', freeMemBytes: () => 6 * GB }); @@ -227,14 +274,14 @@ describe('b4-llama-engine Group A (mocked backend): resident model + cancellatio 'inference.profile': 'quality', 'inference.profileThresholdGb': 8, 'inference.threads': 4, - 'inference.vulkan': false, + 'inference.vulkan': 'auto', }); expect(patch.ok).toBe(true); const settings = engine.responseSettings(); expect(settings['inference.profile']).toBe('quality'); expect(settings['inference.profileThresholdGb']).toBe(8); expect(settings['inference.threads']).toBe(4); - expect(settings['inference.vulkan']).toBe(false); + expect(settings['inference.vulkan']).toBe('auto'); }); it('settings patch: unknown keys and wrong-typed/out-of-bounds values are rejected with ok:false', () => { @@ -252,6 +299,8 @@ describe('b4-llama-engine Group A (mocked backend): resident model + cancellatio { 'inference.profileThresholdGb': 'big' }, { 'inference.vulkan': 'yes' }, { 'inference.vulkan': 1 }, + { 'inference.vulkan': 'off' }, + { 'inference.vulkan': null }, ]; for (const patch of badPatches) { const result = engine.applySettingsPatch(patch); diff --git a/desktop/src/__tests__/t155-gpu-permanent.test.ts b/desktop/src/__tests__/t155-gpu-permanent.test.ts new file mode 100644 index 00000000..0a3e7c04 --- /dev/null +++ b/desktop/src/__tests__/t155-gpu-permanent.test.ts @@ -0,0 +1,368 @@ +// issue #155: permanent regression tests for the GPU backend selection. +// +// These do NOT need a GPU, staged weights, or a network. They use the same +// injection seams the frozen acceptance checks use, plus a plain Node child +// script. They exist because the frozen checks create TEMPORARY files that +// delete themselves: nothing they assert survives as a guard. +// +// Plan-critic Round 1 (CRITICAL): every frozen probe check injects an explicit +// `command`, so the PRODUCTION default spawn was never executed by any check. +// A packaged app that could not spawn its probe would silently always fall back +// to CPU while all fourteen checks stayed green. `default spawn` below is the +// test that closes that hole. +import { EventEmitter } from 'node:events'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { afterEach, describe, expect, it, vi } from 'vitest'; + +import { + GPU_PROBE_SIDECAR, + activeGpuVerdict, + gpuProbeChildPath, + probeOutputIsSane, + readGpuProbeVerdict, + runGpuProbe, + setActiveGpuVerdict, + writeGpuProbeVerdict, + type GpuProbeRunOptions, +} from '../../main/backend/inference/gpu-probe.js'; +import { LlamaEngine, type LlamaEngineBackend } from '../../main/backend/inference/llama-engine.js'; + +const GB = 1024 ** 3; + +/** A spawn stand-in that records how it was called and can be made to answer. */ +function fakeSpawn(reply?: Record, exitCode = 0) { + const calls: Array<{ file: string; args: string[]; env: NodeJS.ProcessEnv | undefined }> = []; + const spawnFn = ((file: string, args: string[], opts: { env?: NodeJS.ProcessEnv }) => { + calls.push({ file, args, env: opts?.env }); + const child = new EventEmitter() as EventEmitter & { + stdout: EventEmitter | null; + stderr: EventEmitter | null; + kill: (signal?: string) => boolean; + killed: boolean; + }; + child.stdout = new EventEmitter(); + child.stderr = new EventEmitter(); + child.killed = false; + child.kill = () => { + child.killed = true; + return true; + }; + setImmediate(() => { + if (reply !== undefined) child.stdout?.emit('data', Buffer.from(JSON.stringify(reply))); + child.emit('close', exitCode, null); + }); + return child; + }) as unknown as NonNullable; + return { calls, spawnFn }; +} + +function stageModels(): { quality: string; fast: string; dir: string } { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 't155-perm-')); + const quality = path.join(dir, 'quality.gguf'); + const fast = path.join(dir, 'fast.gguf'); + fs.writeFileSync(quality, 'x'); + fs.writeFileSync(fast, 'x'); + return { quality, fast, dir }; +} + +const GPU_OK = { backend: 'vulkan' as const, ok: true, reason: 'stub probe: gpu usable', device: 'stub-gpu-0' }; +const CPU_FAIL = { backend: 'cpu' as const, ok: false, reason: 'stub probe: no usable gpu on this host', device: null }; + +afterEach(() => { + setActiveGpuVerdict(null); +}); + +describe('issue #155: the PRODUCTION probe spawn (what the frozen checks never execute)', () => { + it('the default command is this process executable plus the compiled child module', () => { + // The frozen checks always inject `command`, so nothing else proves the + // packaged-app default. In a packaged Electron app the only Node runtime + // present is the Electron binary: hardcoding `node` passes in dev and + // fails in production. + const { calls, spawnFn } = fakeSpawn({ ok: true, backend: 'vulkan', device: 'd', sample: 'OK' }); + return runGpuProbe({ spawnFn, args: ['/models/fast.gguf'] }).then(() => { + expect(calls).toHaveLength(1); + expect(calls[0]?.file).toBe(process.execPath); + const childPath = gpuProbeChildPath(); + expect(calls[0]?.args[0]).toBe(childPath); + // The model path is handed to the child, which loads the FAST profile's + // GGUF rather than the 2.6 GB quality one. + expect(calls[0]?.args).toContain('/models/fast.gguf'); + }); + }); + + it('the child env carries ELECTRON_RUN_AS_NODE, without which the child is a browser process', async () => { + const { calls, spawnFn } = fakeSpawn({ ok: false, backend: 'cpu', reason: 'no device' }); + await runGpuProbe({ spawnFn }); + expect(calls[0]?.env?.ELECTRON_RUN_AS_NODE).toBe('1'); + }); + + it('the resolved child module path exists next to the compiled probe module', () => { + // desktop/tsconfig.json sets rootDir "." and outDir "dist", so the sibling + // .js must exist beside this compiled module or the packaged app can never + // spawn its probe. Under vitest this runs from source, so assert the SOURCE + // sibling exists and that the resolved path is derived from this module. + const resolved = gpuProbeChildPath(); + expect(path.basename(resolved)).toBe('gpu-probe-child.js'); + // Compare through path.normalize so the assertion is about the DIRECTORY, + // not about whether this host spells separators with / or \. + // gpuProbeChildPath() resolves from the PROBE module's own location, so + // the assertion below is about that directory - not this test's. + const probeDir = path.dirname(resolved); + expect(probeDir.endsWith(path.join('main', 'backend', 'inference'))).toBe(true); + // The compiled child must ship beside the compiled probe: the path is + // derived from the probe module's own directory, so a sibling that is not + // there is a missing file at spawn time in the packaged app - exactly the + // failure mode the plan critic Round 1 raised. Under vitest the probe runs + // from source, so assert the SOURCE sibling is in that same directory. + const sourceSibling = path.join(probeDir, 'gpu-probe-child.ts'); + expect( + fs.existsSync(sourceSibling), + `the probe child module must ship beside the probe module; missing: ${sourceSibling}`, + ).toBe(true); + }); + + it('a probe child that never answers is terminated and resolves to CPU, never a throw', async () => { + // The hang case: no close event ever arrives. + const spawnFn = (() => { + const child = new EventEmitter() as EventEmitter & { stdout: EventEmitter; stderr: EventEmitter; kill: () => boolean }; + child.stdout = new EventEmitter(); + child.stderr = new EventEmitter(); + child.kill = () => true; + return child; + }) as unknown as NonNullable; + const verdict = await runGpuProbe({ spawnFn, timeoutMs: 60 }); + expect(verdict.backend).toBe('cpu'); + expect(verdict.ok).toBe(false); + expect(verdict.reason).not.toBe(''); + }); +}); + +describe('issue #155: the probe judges output, and never trusts the child', () => { + it('a child claiming success with garbage output is rejected by the PARENT', async () => { + const { spawnFn } = fakeSpawn({ ok: true, backend: 'vulkan', device: 'd', sample: ' \u0001\u0002\u0003 zz' }); + const verdict = await runGpuProbe({ spawnFn }); + expect(verdict.ok).toBe(false); + expect(verdict.backend).toBe('cpu'); + expect(verdict.reason).not.toBe(''); + }); + + it('probeOutputIsSane separates an answer from a mis-decode', () => { + expect(probeOutputIsSane(' The capital of France is Paris.')).toBe(true); + expect(probeOutputIsSane('OK')).toBe(true); + expect(probeOutputIsSane('')).toBe(false); + expect(probeOutputIsSane(' \n\t ')).toBe(false); + expect(probeOutputIsSane('\u0000\u0001\u0002')).toBe(false); + expect(probeOutputIsSane('...,,,;;;')).toBe(false); + }); + + it('a child that never exits cleanly is reported with its exit state', async () => { + const { spawnFn } = fakeSpawn(undefined, 134); + const verdict = await runGpuProbe({ spawnFn }); + expect(verdict.backend).toBe('cpu'); + expect(verdict.reason).toContain('134'); + }); +}); + +describe('issue #155: the verdict sidecar', () => { + it('round-trips, and reads missing and corrupt files as null without throwing', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 't155-sidecar-')); + expect(readGpuProbeVerdict(dir)).toBeNull(); + writeGpuProbeVerdict(dir, CPU_FAIL); + expect(fs.existsSync(path.join(dir, GPU_PROBE_SIDECAR))).toBe(true); + const reloaded = readGpuProbeVerdict(dir); + expect(reloaded?.backend).toBe('cpu'); + expect(reloaded?.reason).toBe(CPU_FAIL.reason); + fs.writeFileSync(path.join(dir, GPU_PROBE_SIDECAR), '{ not json', 'utf8'); + expect(readGpuProbeVerdict(dir)).toBeNull(); + fs.rmSync(dir, { recursive: true, force: true }); + }); + + it('the process-wide holder is the seam the engine reads', () => { + expect(activeGpuVerdict()).toBeNull(); + setActiveGpuVerdict(GPU_OK); + expect(activeGpuVerdict()?.backend).toBe('vulkan'); + }); +}); + +describe('issue #155: forced GPU never silently degrades', () => { + it('an explicit GPU whose factory throws surfaces the error and does NOT retry on CPU', async () => { + const models = stageModels(); + let attempts = 0; + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + vulkan: true, + gpuVerdict: () => CPU_FAIL, + llamaFactory: async () => { + attempts += 1; + throw new Error('stub: the GPU could not load the model'); + }, + }); + await expect(engine.query('q')).rejects.toThrow(/could not load/i); + // Exactly ONE attempt: a forced GPU that quietly retries on CPU is not + // "force", and a silent degrade is the failure mode this pins. + expect(attempts).toBe(1); + const status = engine.modelStatus() as unknown as { gpu?: { ok?: boolean; reason?: string } }; + expect(status.gpu?.ok).toBe(false); + expect(String(status.gpu?.reason ?? '')).not.toBe(''); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); +}); + +describe('issue #155: the automatic path falls back and records why', () => { + it('a GPU load failure on auto retries once on CPU and downgrades the verdict', async () => { + const models = stageModels(); + const backends: Array<{ backend: string; disposed: boolean }> = []; + const downgrades: Array<{ backend: string; ok: boolean; reason: string }> = []; + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + gpuVerdict: () => GPU_OK, + onGpuLoadFailure: (v) => { + downgrades.push(v); + }, + llamaFactory: async (opts) => { + if (opts.backend === 'vulkan') throw new Error('stub: out of device memory on the GPU'); + const entry = { backend: opts.backend, disposed: false }; + backends.push(entry); + return { + async generate() { + return { answer: 'cpu answer', cancelled: false }; + }, + async dispose() { + entry.disposed = true; + }, + } as LlamaEngineBackend; + }, + }); + const result = await engine.query('q'); + expect(result.answer).toBe('cpu answer'); + expect(backends).toHaveLength(1); + expect(backends[0]?.backend).toBe('cpu'); + // The host is told, so its in-memory holder and sidecar stop re-attempting + // a failing GPU on every later request. + expect(downgrades).toHaveLength(1); + expect(downgrades[0]?.backend).toBe('cpu'); + expect(String(downgrades[0]?.reason ?? '')).toContain('CPU'); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); + + it('an unprobed host resolves to CPU, exactly as before the feature existed', async () => { + const models = stageModels(); + const seen: string[] = []; + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + gpuVerdict: () => null, + llamaFactory: async (opts) => { + seen.push(opts.backend); + return { + async generate() { + return { answer: 'ok', cancelled: false }; + }, + async dispose() { + return undefined; + }, + } as LlamaEngineBackend; + }, + }); + await engine.query('q'); + expect(seen).toEqual(['cpu']); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); +}); + +describe('issue #155: the resident reuse key honours backend identity', () => { + it('changing only the thread count reloads the model', async () => { + // The same class of bug as the GPU selection: anything the backend is BUILT + // from must be part of the reuse key, or a settings switch moves without + // changing anything. + const models = stageModels(); + let loads = 0; + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + threads: 4, + models: { quality: models.quality, fast: models.fast }, + llamaFactory: async () => { + loads += 1; + return { + async generate() { + return { answer: 'ok', cancelled: false }; + }, + async dispose() { + return undefined; + }, + } as LlamaEngineBackend; + }, + }); + await engine.query('q'); + expect(loads).toBe(1); + expect(engine.applySettingsPatch({ 'inference.threads': 6 }).ok).toBe(true); + await engine.query('q'); + expect(loads).toBe(2); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); + + it('an unprobed status still explains itself rather than reporting nothing', () => { + const models = stageModels(); + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + gpuVerdict: () => null, + }); + const status = engine.modelStatus() as unknown as { gpu?: { backend?: string; ok?: boolean; reason?: string } }; + expect(status.gpu?.backend).toBe('cpu'); + expect(status.gpu?.ok).toBe(false); + expect(String(status.gpu?.reason ?? '').length).toBeGreaterThan(0); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); +}); + +describe('issue #155: the settings domain is closed', () => { + it("accepts 'auto' and booleans, rejects everything else without coercing", () => { + const models = stageModels(); + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + }); + expect(engine.applySettingsPatch({ 'inference.vulkan': 'auto' }).ok).toBe(true); + expect(engine.applySettingsPatch({ 'inference.vulkan': true }).ok).toBe(true); + expect(engine.applySettingsPatch({ 'inference.vulkan': false }).ok).toBe(true); + for (const bad of ['yes', 1, 'off', null, 'true', 0]) { + const result = engine.applySettingsPatch({ 'inference.vulkan': bad }); + expect(result.ok, `${JSON.stringify(bad)} must be rejected, not coerced`).toBe(false); + } + fs.rmSync(models.dir, { recursive: true, force: true }); + }); + + it('a rejected patch leaves the previous selection in force', () => { + const models = stageModels(); + const engine = new LlamaEngine({ + profile: 'fast', + freeMemBytes: () => 8 * GB, + cpuCount: () => 8, + models: { quality: models.quality, fast: models.fast }, + }); + expect(engine.applySettingsPatch({ 'inference.vulkan': true }).ok).toBe(true); + expect(engine.applySettingsPatch({ 'inference.vulkan': 'yes' }).ok).toBe(false); + expect(engine.responseSettings()['inference.vulkan']).toBe(true); + fs.rmSync(models.dir, { recursive: true, force: true }); + }); +}); + +// Keep the import used even if a future edit drops the fake spawn's typing. +void vi; \ No newline at end of file diff --git a/desktop/vitest.config.ts b/desktop/vitest.config.ts index 556275ae..7d5eab3a 100644 --- a/desktop/vitest.config.ts +++ b/desktop/vitest.config.ts @@ -27,5 +27,11 @@ export default defineConfig({ test: { environment: 'node', include: ['src/__tests__/**/*.test.ts'], + // Several inference tests perform a model construct/dispose cycle per + // query and sit near vitest's 5s default once the whole suite is loaded on + // one machine, so a ceiling - not a product defect - decides pass or fail. + // Raised for the whole suite rather than patched per offender, because the + // offenders move with load; assertions are unchanged. + testTimeout: 30_000, }, }); diff --git a/web_ui/src/components/ExternalModelSection.tsx b/web_ui/src/components/ExternalModelSection.tsx index 47fb251b..963eb3f2 100644 --- a/web_ui/src/components/ExternalModelSection.tsx +++ b/web_ui/src/components/ExternalModelSection.tsx @@ -290,6 +290,14 @@ export function ExternalModelSection({ id, builtIn, notice }: ExternalModelSecti const writeSeqRef = useRef(0); // Desktop: the backend is the source of truth for the external settings. + // Key the read on the session's IDENTITY STRING (baseUrl), not on the + // wrapper object's identity (issue #155): a provider that returns a fresh + // session value re-fires an identity-keyed effect on every render, and each + // pass re-applies the snapshot into NEW draft/keyState objects — a state + // update per pass, so the component re-renders and the effect re-fires + // again: an unbounded update loop. The baseUrl changes exactly when the + // backend session does. (Same fix as SettingsPage's settings read.) + const sessionBaseUrl = session?.baseUrl ?? null; useEffect(() => { if (!desktop || session === null) return; let cancelled = false; @@ -307,7 +315,7 @@ export function ExternalModelSection({ id, builtIn, notice }: ExternalModelSecti return () => { cancelled = true; }; - }, [desktop, session, applyDesktopSettings]); + }, [desktop, sessionBaseUrl, applyDesktopSettings]); /** Policy check with the airgap rule this app enforces. */ const checkUrl = useCallback( diff --git a/web_ui/src/lib/api/client.ts b/web_ui/src/lib/api/client.ts index f02c702b..df8b2646 100644 --- a/web_ui/src/lib/api/client.ts +++ b/web_ui/src/lib/api/client.ts @@ -23,6 +23,7 @@ import type { StatsResponse, ExternalProbeRequest, ExternalProbeResponse, + GpuProbeResult, } from './types'; import { getToken } from './auth'; @@ -522,6 +523,22 @@ export class ApiClient { return response.json(); } + /** + * Desktop backend only (issue #155): re-run the out-of-process GPU + * capability probe (POST /settings/inference/gpu-test). Takes no body. A + * machine with no usable GPU is a 200 with ok:false, not an error. + */ + async testGpu(): Promise { + const response = await fetch(`${this.baseUrl}/settings/inference/gpu-test`, { + method: 'POST', + headers: this.requestHeaders(), + }); + if (!response.ok) { + throw new ApiError(response.status, await parseErrorResponse(response)); + } + return response.json(); + } + /** * Stats Operations */ diff --git a/web_ui/src/lib/api/types.ts b/web_ui/src/lib/api/types.ts index ee67cabe..c2589a21 100644 --- a/web_ui/src/lib/api/types.ts +++ b/web_ui/src/lib/api/types.ts @@ -279,6 +279,24 @@ export interface ModelStatus { profile: string | null; loadStartedAt: number | null; }; + /** issue #155: the GPU decision the engine made. Absent when an external + * endpoint generates. `reason` is always non-empty so a CPU fallback can be + * explained rather than guessed at. */ + gpu?: { + backend: 'vulkan' | 'cpu'; + ok: boolean; + reason: string; + device?: string | null; + }; +} + +/** issue #155: the result of a GPU capability probe run, the same shape as + * the `gpu` member of ModelStatus. */ +export interface GpuProbeResult { + backend: 'vulkan' | 'cpu'; + ok: boolean; + reason: string; + device?: string | null; } export interface StatsResponse { diff --git a/web_ui/src/pages/SettingsPage.tsx b/web_ui/src/pages/SettingsPage.tsx index c2eaa4ea..21da9f55 100644 --- a/web_ui/src/pages/SettingsPage.tsx +++ b/web_ui/src/pages/SettingsPage.tsx @@ -450,6 +450,11 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti const desktopApp = isElectron(); const [desktopStatus, setDesktopStatus] = useState(null); const [desktopProfile, setDesktopProfile] = useState<'quality' | 'fast' | 'auto' | ''>(''); + // issue #155: the operator's GPU preference. 'auto' delegates to the probe; + // 'gpu'/'cpu' pin it. Empty means "not read from the backend yet". + const [desktopGpu, setDesktopGpu] = useState<'auto' | 'gpu' | 'cpu' | ''>(''); + const [gpuTestPending, setGpuTestPending] = useState(false); + const [gpuTestNote, setGpuTestNote] = useState(null); const [desktopSettingsError, setDesktopSettingsError] = useState(null); const { themePreference, setTheme } = useTheme(); @@ -511,12 +516,19 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti // backend's settings when the session appears (mount) AND whenever the app // switches into api mode, and derive the Response Quality display state // from them — never a silent PUT. - const desktopReadRef = useRef<{ session: typeof desktopSession; mode: string | null }>({ session: null, mode: null }); + // Key the read on the session's IDENTITY STRING, not on the wrapper object's + // identity. The wrapper is a fresh object whenever its provider returns a new + // value, and an effect that depends on that identity re-fires on every render: + // each pass then re-reads the backend, sets state, re-renders, and repeats - + // an unbounded loop that starves the event loop. The baseUrl changes exactly + // when the session does, and comparing a string makes the dependency honest. + const desktopSessionKey = desktopSession?.baseUrl ?? null; + const desktopReadRef = useRef<{ key: string | null; mode: string | null }>({ key: null, mode: null }); useEffect(() => { const previous = desktopReadRef.current; - desktopReadRef.current = { session: desktopSession, mode }; + desktopReadRef.current = { key: desktopSessionKey, mode }; if (!electronMode || !desktopSession) return; - const sessionChanged = previous.session !== desktopSession; + const sessionChanged = previous.key !== desktopSessionKey; const enteredApi = mode === 'api' && previous.mode !== 'api'; if (!sessionChanged && !enteredApi) return; const ticket = ++presetTicketRef.current; @@ -528,6 +540,11 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti if (profile === 'quality' || profile === 'fast' || profile === 'auto') { setDesktopProfile(profile); } + // issue #155: the backend reports the selection in its widened form. + const gpu = settings['inference.vulkan']; + if (gpu === 'auto' || gpu === true || gpu === false) { + setDesktopGpu(gpu === 'auto' ? 'auto' : gpu === true ? 'gpu' : 'cpu'); + } if (ticket === presetTicketRef.current) applyDesktopSettings(settings); } catch (err) { if (isMountedRef.current) setDesktopSettingsError(err instanceof Error ? err.message : String(err)); @@ -540,7 +557,7 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti if (isMountedRef.current) setDesktopStatus(null); } })(); - }, [electronMode, desktopSession, mode, applyDesktopSettings]); + }, [electronMode, desktopSessionKey, mode, applyDesktopSettings]); // B9: persist an inference-profile override to the backend (AC3 — survives // restart through the backend's settings sidecar). @@ -559,6 +576,46 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti [desktopSession] ); + // issue #155: persist the GPU preference. The backend reloads the resident + // model on an effective change, so this takes effect without a restart; the + // status snapshot is re-read so the displayed backend follows the decision. + const handleDesktopGpuChange = useCallback( + (value: 'auto' | 'gpu' | 'cpu') => { + if (!desktopSession) return; + setDesktopGpu(value); + setGpuTestNote(null); + desktopSession.apiClient + .updateSettings({ + 'inference.vulkan': value === 'auto' ? 'auto' : value === 'gpu', + }) + .then(() => notifyDesktopModelsChanged()) + .then(() => fetchModelStatus(desktopSession)) + .then((status) => setDesktopStatus(status)) + .catch((err) => setDesktopSettingsError(err instanceof Error ? err.message : String(err))); + }, + [desktopSession] + ); + + // issue #155: re-run the out-of-process probe on demand, so a machine whose + // driver changed can recover from a stale negative verdict without a restart. + const handleGpuTest = useCallback(() => { + if (!desktopSession) return; + setGpuTestPending(true); + setGpuTestNote(null); + desktopSession.apiClient + .testGpu() + .then((result) => { + setGpuTestNote( + result.ok + ? `GPU usable: ${result.device ?? result.backend}` + : `GPU not usable, CPU inference stays on. ${result.reason}` + ); + return fetchModelStatus(desktopSession).then((status) => setDesktopStatus(status)); + }) + .catch((err) => setGpuTestNote(`GPU test failed: ${err instanceof Error ? err.message : String(err)}`)) + .finally(() => setGpuTestPending(false)); + }, [desktopSession]); + // settings-wiring-honesty (AC2/AC3; user decision 2026-09-30, reversing PR // #138's rag_n_results-only mirror): with a desktop session, a preset change // PUTs the preset's full patch — result count, reranking, max tokens and @@ -959,7 +1016,17 @@ function SettingsPageInner({ initialSection, sectionRequest, reloadPage }: Setti { value: 'browser-local', label: 'In this window', - description: 'Runs in this app window with wllama (CPU) or WebLLM (WebGPU), chosen below.', + // issue #155: keep "GPU"/"Vulkan" wording OUT of this card's + // text. The card's description lives inside its