From 77fadaca9e32c90082fcf8f0e98fa0b5a290b7dc Mon Sep 17 00:00:00 2001 From: Jaswant Panchumarti Date: Tue, 4 Aug 2026 10:26:03 -0400 Subject: [PATCH] ci: use gpu runner for linux --- .github/workflows/test_and_release.yml | 82 +++++++++++++++++--- tests/check_gpu.py | 102 +++++++++++++++++++++++++ tests/conftest.py | 38 ++++++++- tests/test_cone.py | 4 +- tests/test_multi_view.py | 9 +-- tests/test_volume_rendering.py | 4 +- 6 files changed, 219 insertions(+), 20 deletions(-) create mode 100644 tests/check_gpu.py diff --git a/.github/workflows/test_and_release.yml b/.github/workflows/test_and_release.yml index 16146a9b..ffa29783 100644 --- a/.github/workflows/test_and_release.yml +++ b/.github/workflows/test_and_release.yml @@ -28,14 +28,25 @@ jobs: pytest: name: Pytest ${{ matrix.config.name }} runs-on: ${{ matrix.config.os }} + container: + image: ${{ matrix.config.container }} + options: ${{ matrix.config.container-options }} strategy: fail-fast: false matrix: python-version: ["3.13"] config: - - { name: "Linux", os: ubuntu-latest } - - { name: "MacOSX", os: macos-latest } - - { name: "Windows", os: windows-latest } + - { + name: "Linux", + os: [self-hosted, gpu], + container: "mcr.microsoft.com/playwright:v1.61.0-noble", + # Default caps are compute,utility, which mount no rendering + # driver at all. "all" pulls in graphics/display so + # libEGL_nvidia and libGLX_nvidia reach the container. + container-options: "--gpus all -e NVIDIA_DRIVER_CAPABILITIES=all -e NVIDIA_VISIBLE_DEVICES=all", + } + - { name: "MacOSX", os: [macos-latest], container: "", container-options: "" } + - { name: "Windows", os: [windows-latest], container: "", container-options: "" } defaults: run: @@ -47,6 +58,56 @@ jobs: UV_PYTHON: ${{ matrix.python-version }} steps: + # The Playwright image already ships libvulkan1/libegl1/libgles2 (webkit + # pulls them in), so no apt step is needed. The NVIDIA libraries those + # loaders dispatch to are mounted by the container toolkit, not installed. + - name: Enable GPU tests + if: matrix.config.name == 'Linux' + run: echo "TRAME_TEST_GPU=1" >> "$GITHUB_ENV" + + - name: Report GPU + if: matrix.config.name == 'Linux' + run: nvidia-smi + + # Write manifests when they are missing. + # without the Vulkan one, ANGLE's vulkan backend finds no device and + # WebGPU silently drops to SwiftShader (undesired) + - name: Backfill GPU ICD manifests + if: matrix.config.name == 'Linux' + run: | + if ! ls /usr/share/vulkan/icd.d/*nvidia*.json /etc/vulkan/icd.d/*nvidia*.json >/dev/null 2>&1; then + if ldconfig -p | grep -q libGLX_nvidia; then + api=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1) + echo "no Vulkan ICD found; writing one for driver $api" + mkdir -p /usr/share/vulkan/icd.d + cat > /usr/share/vulkan/icd.d/nvidia_icd.json <<'JSON' + { + "file_format_version": "1.0.0", + "ICD": { + "library_path": "libGLX_nvidia.so.0", + "api_version": "1.3.277" + } + } + JSON + else + echo "libGLX_nvidia is not mounted -- the graphics capability is not reaching the container" + fi + fi + if ! ls /usr/share/glvnd/egl_vendor.d/*nvidia*.json >/dev/null 2>&1; then + if ldconfig -p | grep -q libEGL_nvidia; then + echo "no EGL vendor manifest found; writing one" + mkdir -p /usr/share/glvnd/egl_vendor.d + cat > /usr/share/glvnd/egl_vendor.d/10_nvidia.json <<'JSON' + { + "file_format_version": "1.0.0", + "ICD": { + "library_path": "libEGL_nvidia.so.0" + } + } + JSON + fi + fi + - name: Checkout uses: actions/checkout@v6 @@ -56,7 +117,10 @@ jobs: version: "0.11.24" enable-cache: true + # The Playwright image already ships node 24, which is what this action + # would install. Keep the pin in sync with the image's NODE_VERSION. - name: Set Up Node + if: matrix.config.name != 'Linux' uses: actions/setup-node@v6 with: node-version: 24 @@ -70,14 +134,8 @@ jobs: - name: Install the project run: uv sync --all-extras --dev - - name: Install OSMesa for Linux - if: matrix.config.os == 'ubuntu-latest' - run: | - sudo apt update - sudo apt-get install -y libosmesa6-dev - - name: Install OSMesa for Windows - if: matrix.config.os == 'windows-latest' + if: matrix.config.name == 'Windows' # 25.0.7 is the last mesa-dist-win release that ships osmesa.dll # (removed in 25.1.0). PATH is searched by LoadLibrary, which is how # VTK's vtkOSOpenGLRenderWindow locates osmesa.dll at runtime. @@ -91,6 +149,10 @@ jobs: uv sync --all-extras --dev uv run playwright install + - name: Verify browser GPU + if: matrix.config.name == 'Linux' + run: uv run python tests/check_gpu.py + - name: Run tests run: uv run pytest -s ./tests --cov=src --cov-report=xml diff --git a/tests/check_gpu.py b/tests/check_gpu.py new file mode 100644 index 00000000..0d4d647f --- /dev/null +++ b/tests/check_gpu.py @@ -0,0 +1,102 @@ +"""Fail the CI job when the browser is not actually rendering on the GPU. + +This runs probe on a secure context. Ask for highPerformance webgpu adapter +since VTK asks for it. +""" + +import asyncio +import sys +import threading +from functools import partial +from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +from playwright.async_api import async_playwright + +sys.path.insert(0, str(Path(__file__).parent)) + +from conftest import chromium_launch, webgpu_args # noqa: E402 + +SOFTWARE = ("swiftshader", "llvmpipe", "software", "basic render", "warp") + +PROBE = """async () => { + const out = {}; + const gl = document.createElement('canvas').getContext('webgl2'); + const ext = gl && gl.getExtension('WEBGL_debug_renderer_info'); + out.webgl = ext ? gl.getParameter(ext.UNMASKED_RENDERER_WEBGL) : null; + + if (!navigator.gpu) { + out.webgpu = null; + out.reason = 'navigator.gpu is undefined (WebGPU not enabled in this build/flags)'; + return out; + } + let adapter = null; + let reason = 'requestAdapter resolved to null (no adapter matched)' + try { + adapter = await navigator.gpu.requestAdapter(); + } catch (e) { + reason = 'requestAdapter threw: ' + e; + } + if (!adapter) { + try { + adapter = await navigator.gpu.requestAdapter({ + powerPreference: 'high-performance', + }); + } catch (e) { + reason = 'requestAdapter with high-performance threw: ' + e; + } + } + if (!adapter) { + out.webgpu = null; + out.reason = reason; + return out; + } + const i = adapter.info || {}; + out.webgpu = [i.vendor, i.architecture, i.device, i.description] + .filter(Boolean) + .join(' '); + return out; +}""" + + +async def main(): + handler = partial(SimpleHTTPRequestHandler, directory=str(Path(__file__).parent)) + server = ThreadingHTTPServer(("127.0.0.1", 0), handler) + threading.Thread(target=server.serve_forever, daemon=True).start() + + try: + async with async_playwright() as p: + browser = await chromium_launch(p, webgpu_args()) + page = await browser.new_page() + await page.goto(f"http://127.0.0.1:{server.server_port}/") + info = await page.evaluate(PROBE) + await browser.close() + finally: + server.shutdown() + + print("WebGL :", info.get("webgl")) + print("WebGPU :", info.get("webgpu") or f"unavailable -- {info.get('reason')}") + + failures = [] + webgl = (info.get("webgl") or "").lower() + if not webgl: + failures.append("WebGL2 context unavailable") + elif any(bad in webgl for bad in SOFTWARE): + failures.append(f"WebGL is on a software rasterizer: {info['webgl']}") + + webgpu = (info.get("webgpu") or "").lower() + if not webgpu: + failures.append( + f"WebGPU unavailable, so webgpu tests will skip: {info.get('reason')}" + ) + elif any(bad in webgpu for bad in SOFTWARE): + failures.append(f"WebGPU adapter is a software fallback: {info['webgpu']}") + + if failures: + for f in failures: + print("FAIL:", f) + sys.exit(1) + print("OK: browser is rendering on the GPU") + + +asyncio.run(main()) diff --git a/tests/conftest.py b/tests/conftest.py index 6455a323..655caa33 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,3 +1,4 @@ +import os import sys from pathlib import Path @@ -16,6 +17,15 @@ HELPER = FixtureHelper(ROOT_PATH) +async def chromium_launch(p, args=None): + """Launch headless Chromium""" + args = list(args or []) + if os.environ.get("TRAME_TEST_GPU") == "1": + # The container runs as root, where the setuid sandbox refuses to start. + args += ["--no-sandbox", "--ignore-gpu-blocklist"] + return await p.chromium.launch(args=args, headless=True) + + def webgpu_args(): """Chromium flags needed to obtain a working WebGPU adapter, per platform. @@ -29,11 +39,22 @@ def webgpu_args(): Chromium build ships a dxil.dll it cannot load (EnsureDXCLibraries -> "DynamicLib.Open: dxil.dll Windows Error: 87"). Disabling the use_dxc Dawn feature falls back to the FXC shader compiler, which needs no external DLL. + + On Linux, use the headless-WebGPU flag set from + https://developer.chrome.com/blog/supercharge-web-ai-testing: + --enable-features=Vulkan turns on Chrome's Vulkan path, and + --disable-vulkan-surface makes it present via bit-blit instead of a + VK_KHR_surface swapchain, which headless has no window to back. Without + the latter Vulkan init fails and Dawn silently drops to SwiftShader. + (--enable-features=Vulkan,VulkanFromANGLE from the gpuweb wiki kills the + GPU process outright here: WebGL and WebGPU both go dark.) """ backend = {"darwin": "metal", "win32": "d3d11"}.get(sys.platform, "vulkan") args = [f"--use-angle={backend}", "--enable-unsafe-webgpu"] if sys.platform == "win32": args.append("--disable-dawn-features=use_dxc") + elif sys.platform.startswith("linux"): + args += ["--enable-features=Vulkan", "--disable-vulkan-surface"] return args @@ -47,11 +68,26 @@ async def webgpu_hardware_available(page): cannot present with, so the canvas stays blank. macOS runners have a real Metal GPU and render correctly. Detect the fallback case so webgpu configs can skip where no real GPU exists instead of failing on a blank frame. + + Request with powerPreference high-performance because that is what VTK + requests (vtkWebGPUConfiguration defaults PowerPreference to + HighPerformance). The two are not equivalent: on the Linux/NVIDIA GPU + runner the preference-less request resolves to null while the + high-performance one returns the hardware adapter, so probing without + the preference would skip tests that VTK can actually run. """ return await page.evaluate( """async () => { if (!navigator.gpu) return false; - const a = await navigator.gpu.requestAdapter(); + // First call after GPU-process startup can resolve null while + // Dawn initializes (Linux/NVIDIA); retry briefly before deciding. + let a = null; + for (let attempt = 0; attempt < 20 && !a; attempt++) { + if (attempt > 0) await new Promise((r) => setTimeout(r, 250)); + a = await navigator.gpu.requestAdapter({ + powerPreference: 'high-performance', + }); + } if (!a) return false; if (a.isFallbackAdapter) return false; const i = a.info || {}; diff --git a/tests/test_cone.py b/tests/test_cone.py index 07a0adec..7f159e22 100644 --- a/tests/test_cone.py +++ b/tests/test_cone.py @@ -4,7 +4,7 @@ import pytest from playwright.async_api import async_playwright, expect -from conftest import webgpu_args, webgpu_hardware_available +from conftest import chromium_launch, webgpu_args, webgpu_hardware_available BASELINES = [ Path(__file__).with_name("assets") / "cone" / name @@ -46,7 +46,7 @@ async def test_cone(ConeApp, utils, config): async with async_playwright() as p: args = webgpu_args() if wasm_rendering == "webgpu" else [] - browser = await p.chromium.launch(headless=True, args=args) + browser = await chromium_launch(p, args) page = await browser.new_page() await page.set_viewport_size({"width": 300, "height": 300}) diff --git a/tests/test_multi_view.py b/tests/test_multi_view.py index 9fc91257..1eff2a92 100644 --- a/tests/test_multi_view.py +++ b/tests/test_multi_view.py @@ -4,6 +4,8 @@ import pytest from playwright.async_api import async_playwright, expect +from conftest import chromium_launch + BASELINES = [ Path(__file__).with_name("assets") / "multi_view" / name for name in [ @@ -17,10 +19,7 @@ async def test_multi_view(MultiViewApp, utils): """Two LocalViews sharing one WASM session must both render (issues/76, /77). Catches the updateAsync-serialization regression (a second view's update - being swallowed leaves it black). It does NOT catch the Size-property leak - on Linux/OSMesa: there the vtkOSOpenGLRenderWindow -> vtkRenderWindow remap - makes even the old skip match, so that bug only reproduces on a platform - whose render window serializes under its own child class (e.g. macOS). + being swallowed leaves it black). """ app = MultiViewApp("multi-view") task = app.server.start(exec_mode="task", port=0) @@ -30,7 +29,7 @@ async def test_multi_view(MultiViewApp, utils): valid_image_comparisons = [] async with async_playwright() as p: - browser = await p.chromium.launch(headless=True) + browser = await chromium_launch(p) page = await browser.new_page() await page.set_viewport_size({"width": 600, "height": 300}) diff --git a/tests/test_volume_rendering.py b/tests/test_volume_rendering.py index 81afff68..99eb1da4 100644 --- a/tests/test_volume_rendering.py +++ b/tests/test_volume_rendering.py @@ -5,7 +5,7 @@ from trame_vtklocal.module.wasm import wasm_downloaded -from conftest import webgpu_args +from conftest import chromium_launch, webgpu_args BASELINES = [ Path(__file__).with_name("assets") / "volume" / name @@ -69,7 +69,7 @@ async def test_volume_rendering(VolumeApp, utils, config, mapper_type): async with async_playwright() as p: args = webgpu_args() if wasm_rendering == "webgpu" else [] - browser = await p.chromium.launch(headless=True, args=args) + browser = await chromium_launch(p, args) page = await browser.new_page() await page.set_viewport_size({"width": 300, "height": 300})