diff --git a/dashboard/views/render.ts b/dashboard/views/render.ts index d7885a5..53f3d32 100644 --- a/dashboard/views/render.ts +++ b/dashboard/views/render.ts @@ -201,6 +201,38 @@ function layout(title: string, content: string, orgName: string = "", activeNav: .check-prod-btn:disabled { opacity: 0.5; cursor: wait; } .check-prod-result { font-size: 11px; margin-left: 8px; color: #aaa; } + /* Export dropdown on each repo card. Filename + format picker lives + here; the actual download is a GET to /export/:owner/:name/:format + with the currently selected branch from the combobox. */ + .export-combo { position: relative; display: inline-block; } + .export-btn { + font-family: var(--font-pixel); font-size: 8px; letter-spacing: 1px; + background: #050505; border: 1px solid #00ffff; color: #00ffff; + padding: 7px 12px; cursor: pointer; transition: all 0.15s; + } + .export-btn:hover { background: #00ffff; color: #0a0a0a; } + .export-menu { + position: absolute; right: 0; top: 100%; margin-top: 2px; + list-style: none; background: #050505; border: 1px solid #00ffff; + min-width: 240px; z-index: 50; + box-shadow: 0 0 12px rgba(0,255,255,0.25); display: none; + } + .export-combo.open .export-menu { display: block; } + .export-menu li { + font-family: var(--font-mono); font-size: 12px; color: #aaa; + padding: 7px 12px; cursor: pointer; white-space: nowrap; + } + .export-menu li:hover:not(.export-header) { background: #0a0a0a; color: #00ffff; } + .export-menu li.export-header { + font-family: var(--font-pixel); font-size: 7px; color: #666; + letter-spacing: 2px; padding: 6px 12px 4px; cursor: default; + border-top: 1px solid #1a1a1a; + } + .export-menu li.export-header:first-child { border-top: none; } + /* Org export sits above the repo list, right-aligned. */ + .org-export-combo { float: right; margin-top: -58px; margin-bottom: 10px; } + .org-export-combo .export-menu { right: 0; } + /* Tabs */ .tab-bar { display: flex; gap: 0; margin-bottom: 0; border-bottom: 2px solid #333; flex-wrap: wrap; } .tab { @@ -580,6 +612,53 @@ function layout(title: string, content: string, orgName: string = "", activeNav: btn.disabled = false; }); } + + function toggleExportMenu(repoId) { + // Close every other open export menu first so we don't stack them. + document.querySelectorAll('.export-combo.open').forEach(function(el) { + if (el.querySelector('#export-menu-' + repoId) === null) { + el.classList.remove('open'); + } + }); + var menu = document.getElementById('export-menu-' + repoId); + if (!menu) return; + menu.parentElement.classList.toggle('open'); + } + + function downloadExport(evt, owner, name, repoId, format) { + evt.stopPropagation(); + // Honor the branch selected in the combobox for this repo card. Falls + // back to no ?branch= so the server picks main/master. + var combo = document.getElementById('branch-' + repoId); + var branch = combo ? combo.getAttribute('data-value') : ''; + var url = '/export/' + owner + '/' + name + '/' + format; + if (branch) url += '?branch=' + encodeURIComponent(branch); + // Close the menu immediately so the dropdown doesn't linger while the + // download starts. + var parent = document.getElementById('export-menu-' + repoId); + if (parent) parent.parentElement.classList.remove('open'); + window.location.href = url; + } + + // Close export menus when clicking anywhere else on the page. + document.addEventListener('click', function(evt) { + if (!evt.target.closest('.export-combo')) { + document.querySelectorAll('.export-combo.open').forEach(function(el) { + el.classList.remove('open'); + }); + } + }); + + function toggleOrgExportMenu() { + var el = document.querySelector('.org-export-combo'); + if (el) el.classList.toggle('open'); + } + function downloadOrgExport(evt, format) { + evt.stopPropagation(); + var el = document.querySelector('.org-export-combo'); + if (el) el.classList.remove('open'); + window.location.href = '/export/all/' + format; + } `; @@ -622,7 +701,24 @@ export function renderDashboard(summaries: RepoSummary[], branchesPerRepo: Map
Leaks
${secretsCount}
`; - const searchHtml = ``; + // Org-level audit-bundle export — main/master only across all repos. + // Deliberately separate from the per-repo dropdown (which respects the + // branch dropdown) because auditors want production state, not WIP. + const orgExportHtml = ` +
+ + +
`; + + const searchHtml = `${orgExportHtml}`; const reposHtml = summaries.map(r => { const [owner, name] = r.repo.split("/"); @@ -687,6 +783,20 @@ export function renderDashboard(summaries: RepoSummary[], branchesPerRepo: Map ${hasSiteUrl ? `` : ""} +
+ +
    +
  • MACHINE-READABLE
  • +
  • JSON (full state)
  • +
  • SARIF (code scanning)
  • +
  • OSCAL (assessment results)
  • +
  • CSV
  • +
  • NIST CSF controls
  • +
  • EU AI Act articles
  • +
  • Risk register
  • +
  • Vulnerabilities
  • +
+
OVERVIEW
diff --git a/dashboard/worker.ts b/dashboard/worker.ts index 229943e..5c7626e 100644 --- a/dashboard/worker.ts +++ b/dashboard/worker.ts @@ -3,6 +3,12 @@ import { parse } from "yaml"; import type { Manifest, } from "../scanner/types.js"; import { evaluateFramework } from "../scanner/generators/framework-report.js"; import { evaluateEUAIAct, calcAIComplianceScore } from "../scanner/frameworks/eu-ai-act.js"; +import { assessRisks } from "../scanner/generators/risk-assessment.js"; +import { generateJsonExport } from "../scanner/generators/exports/json.js"; +import { generateCsvNistCsf, generateCsvEuAiAct, generateCsvRisks, generateCsvVulnerabilities } from "../scanner/generators/exports/csv.js"; +import { concatCsv } from "../scanner/generators/exports/concat.js"; +import { generateSarifExport } from "../scanner/generators/exports/sarif.js"; +import { generateOscalExport } from "../scanner/generators/exports/oscal.js"; import { renderDashboard, renderRepoDetail, renderNistView, renderBranchComparison, renderTrendChart, renderAIComplianceView, renderInventoryView } from "./views/render.js"; import { verifyGitHubOidc, assertRepositoryMatches, AuthError, DEFAULT_AUDIENCE } from "./auth.js"; @@ -698,4 +704,157 @@ app.get("/api/inventory.csv", async (c) => { }); }); +// ─── Phase 9 Sub-phase A: machine-readable exports ───────────────────────── + +/** + * Compute the framework + risk artifacts for one manifest. The scanner itself + * passes authFindings (source-level auth middleware misses) into assessRisks, + * but those aren't stored in the manifest so we call assessRisks with [] + * here. The risk set is nearly identical — only file-specific auth misses + * drop out. + */ +function exportPayload(manifest: Manifest, siteUrl: string | undefined) { + const nist = evaluateFramework(manifest); + const eu = evaluateEUAIAct(manifest); + const config = { + siteName: manifest.repo, + siteUrl: siteUrl ?? "", + ownerName: manifest.repo.split("/")[0] ?? "Unknown", + contactEmail: "", + securityContact: "", + logRetentionDays: 90, + jurisdiction: ["gdpr"], + preferredLanguages: ["en"], + outputDir: "docs/policies", + policyUrls: {}, + ai: { enabled: false, provider: "anthropic" as const }, + aiSystemOverrides: [], + }; + const risks = assessRisks(manifest, config, []); + return { nist, eu, risks }; +} + +function stemFor(manifest: Manifest): string { + const safeBranch = manifest.branch.replace(/[^\w.-]/g, "-"); + return `${manifest.repo.replace(/\//g, "-")}-${safeBranch}-${manifest.commit.slice(0, 7)}`; +} + +function download(body: string, filename: string, contentType: string): Response { + return new Response(body, { + headers: { + "content-type": contentType, + "content-disposition": `attachment; filename="${filename}"`, + }, + }); +} + +// Per-repo exports — honor the ?branch= query param so consumers can +// download the state of any scanned branch, not just main. Falls back to the +// repo's default (main/master/first-scanned) via findByBranch. +app.get("/export/:owner/:name/:format{.+}", async (c) => { + const repoName = `${c.req.param("owner")}/${c.req.param("name")}`; + const format = c.req.param("format"); + const branch = c.req.query("branch") || undefined; + const all = await getManifests(c.env.GRC_KV); + const entry = findByBranch(all.filter(m => m.manifest.repo === repoName), branch); + if (!entry) return c.json({ error: "Repo not found" }, 404); + + const { manifest, siteUrl } = entry; + const { nist, eu, risks } = exportPayload(manifest, siteUrl); + const stem = stemFor(manifest); + + switch (format) { + case "manifest.json": + return download(generateJsonExport(manifest, nist, eu, risks), `${stem}.json`, "application/json; charset=utf-8"); + case "findings.sarif": + return download(generateSarifExport(manifest), `${stem}.sarif`, "application/sarif+json; charset=utf-8"); + case "assessment.oscal.json": + return download(generateOscalExport(manifest, nist, eu), `${stem}.oscal.json`, "application/json; charset=utf-8"); + case "nist-csf.csv": + return download(generateCsvNistCsf(manifest, nist), `${stem}-nist-csf.csv`, "text/csv; charset=utf-8"); + case "eu-ai-act.csv": + return download(generateCsvEuAiAct(manifest, eu), `${stem}-eu-ai-act.csv`, "text/csv; charset=utf-8"); + case "risks.csv": + return download(generateCsvRisks(manifest, risks), `${stem}-risks.csv`, "text/csv; charset=utf-8"); + case "vulnerabilities.csv": + return download(generateCsvVulnerabilities(manifest), `${stem}-vulnerabilities.csv`, "text/csv; charset=utf-8"); + default: + return c.json({ error: `Unknown export format: ${format}. Try manifest.json, findings.sarif, assessment.oscal.json, nist-csf.csv, eu-ai-act.csv, risks.csv, vulnerabilities.csv.` }, 400); + } +}); + +/** + * Pick the main/master entry for each repo — the org-level export is an + * "audit bundle" so we deliberately leave feature branches out. + * + * If a repo has no main/master manifest in KV (e.g. only feature-branch + * scans have landed so far), it is excluded entirely rather than silently + * substituting an arbitrary branch. The endpoint's documented semantics + * are "main/master only"; the audit bundle must honor that literally or + * it leaks WIP state into what auditors believe is production evidence. + */ +function mainEntryPerRepo(all: StoredManifest[]): StoredManifest[] { + const byRepo = new Map(); + for (const entry of all) { + const repo = entry.manifest.repo; + const isMain = entry.manifest.branch === "main" || entry.manifest.branch === "master"; + if (!isMain) continue; + const existing = byRepo.get(repo); + // Prefer "main" over "master" if a repo somehow has both. Otherwise + // just keep whichever we saw first; main and master aren't both + // expected in the same repo. + if (!existing || entry.manifest.branch === "main") { + byRepo.set(repo, entry); + } + } + return [...byRepo.values()]; +} + +// Org-level aggregate — main/master only across all repos. JSON + 4 CSVs. +// SARIF + OSCAL aggregation is skipped for now; they require careful merge +// semantics (dedup of common rules, UUID stability) and nobody's asked. +app.get("/export/all/:format{.+}", async (c) => { + const all = await getManifests(c.env.GRC_KV); + const entries = mainEntryPerRepo(all); + const format = c.req.param("format"); + const dateStem = new Date().toISOString().slice(0, 10); + + switch (format) { + case "manifest.json": { + const payload = entries.map(e => { + const { nist, eu, risks } = exportPayload(e.manifest, e.siteUrl); + return JSON.parse(generateJsonExport(e.manifest, nist, eu, risks)); + }); + return download(JSON.stringify(payload, null, 2) + "\n", `grc-org-${dateStem}.json`, "application/json; charset=utf-8"); + } + case "nist-csf.csv": { + const csvs = entries.map(e => { + const { nist } = exportPayload(e.manifest, e.siteUrl); + return generateCsvNistCsf(e.manifest, nist); + }); + return download(concatCsv(csvs), `grc-org-${dateStem}-nist-csf.csv`, "text/csv; charset=utf-8"); + } + case "eu-ai-act.csv": { + const csvs = entries.map(e => { + const { eu } = exportPayload(e.manifest, e.siteUrl); + return generateCsvEuAiAct(e.manifest, eu); + }); + return download(concatCsv(csvs), `grc-org-${dateStem}-eu-ai-act.csv`, "text/csv; charset=utf-8"); + } + case "risks.csv": { + const csvs = entries.map(e => { + const { risks } = exportPayload(e.manifest, e.siteUrl); + return generateCsvRisks(e.manifest, risks); + }); + return download(concatCsv(csvs), `grc-org-${dateStem}-risks.csv`, "text/csv; charset=utf-8"); + } + case "vulnerabilities.csv": { + const csvs = entries.map(e => generateCsvVulnerabilities(e.manifest)); + return download(concatCsv(csvs), `grc-org-${dateStem}-vulnerabilities.csv`, "text/csv; charset=utf-8"); + } + default: + return c.json({ error: `Unknown org export format: ${format}. Try manifest.json, nist-csf.csv, eu-ai-act.csv, risks.csv, vulnerabilities.csv.` }, 400); + } +}); + export default app; diff --git a/docs/implementation-checklist.md b/docs/implementation-checklist.md index 6515ded..a77961a 100644 --- a/docs/implementation-checklist.md +++ b/docs/implementation-checklist.md @@ -365,18 +365,19 @@ Recommended build order: **Design principle: no new action runtime.** We don't want to bloat the action's scan time. Integration work happens via (a) exports the user downloads or (b) on-demand dashboard actions. Automatic push-on-scan is deliberately deferred. -### Sub-phase A: Standard export formats — MVP -- [ ] **SARIF-format output** for security findings (format compatibility only — producing a conformant SARIF file; this does NOT by itself populate GitHub's PR Security tab) -- [ ] **SARIF upload step** in the action — use `github/codeql-action/upload-sarif@v3` (or POST to `/repos/:owner/:repo/code-scanning/sarifs`) so findings actually appear in GitHub's code scanning UI. Requires `security-events: write` in the consuming workflow's permissions block. Must be documented alongside the existing `contents: write` requirement. -- [ ] **OSCAL export** (NIST SP 800-53 / Open Security Controls Assessment Language) — JSON/YAML/XML standard that Drata, Hyperproof, and an increasing number of GRC platforms can ingest -- [ ] **Enhanced JSON export** — structured scan data with all findings, for custom ingestion or scripting -- [ ] **CSV export** — flat finding list for spreadsheet ingestion (audit workpapers, remediation tracking) -- [ ] **Export dropdown on each repo card** — choose format, download file -- [ ] **Org-level export** — aggregated across all scanned repos in one bundle -- [ ] Exports land in `.grc/exports/` when run via CLI; in-browser download from the dashboard -- [ ] No credentials required for downloads; SARIF upload uses `GITHUB_TOKEN` only (no vendor keys) +### Sub-phase A: Standard export formats — MVP — DONE +- [x] **SARIF-format output** — conformant 2.1.0 with per-secret, per-CVE, per-AI-system results. Vulnerabilities map CVSS severity onto SARIF level (critical → error, moderate → warning, low → note). Fingerprints populated so the Security tab can dedupe across runs. +- [ ] **SARIF upload step** in the action — deferred. Would require consumers to add `security-events: write` to their workflow permissions block; want to gather signal on whether anyone actually wants the Security-tab integration before adding another required permission. +- [x] **OSCAL Assessment Results export** — OSCAL v1.1.2 JSON with one result per framework (NIST CSF always; EU AI Act when evaluated). Each evaluated control becomes an observation; non-pass non-NA results also produce findings. Cross-refs preserved as OSCAL props. +- [x] **Enhanced JSON export** — `GRCExport` envelope: manifest + NIST CSF evaluation + EU AI Act evaluation + full risk register in one blob. Schema-versioned (`schema: grc-export`, `schemaVersion: 1.0`). +- [x] **CSV export** — four flat tables (NIST CSF controls, EU AI Act articles, risk register, dependency vulnerabilities) with RFC 4180 quoting. Repo/branch/commit/scan_date columns included on every row. +- [x] **Export dropdown on each repo card** — respects the branch combobox. Per-format filename like `---.sarif` so org-level bundles don't collide. +- [x] **Org-level export** — main/master only across all repos. JSON + 4 CSVs. SARIF/OSCAL org aggregation deliberately deferred (they need careful merge semantics; no demand yet). +- [x] **Per-CVE capture in the scanner** — new `DependencyVulnerability[]` field on the manifest carrying advisory id, package, severity, CVSS score, range, fix availability, paths. Stable severity-then-name sort. Required for meaningful SARIF / CSV vuln output. +- [x] Exports land in `.grc/exports/` when run via CLI; `/export/:owner/:name/:format` and `/export/all/:format` routes on the dashboard for in-browser download. +- [x] No credentials required for downloads. - **GRC concept:** Structured evidence formats; OSCAL as the emerging interchange standard; SARIF as the de-facto security findings format -- **Known gotcha:** SARIF-on-disk is not the same as findings in the Security tab. Producing the file and uploading it are two separate pieces of work — ship both. +- **Known gotcha:** SARIF-on-disk is not the same as findings in the Security tab. The export is conformant; the upload step is still open. ### Sub-phase B: Auditor evidence packaging (was Phase 5 Tier 3) - [ ] Generate PDF/ZIP evidence package per framework diff --git a/scanner/generators/exports/concat.test.ts b/scanner/generators/exports/concat.test.ts new file mode 100644 index 0000000..d09ae63 --- /dev/null +++ b/scanner/generators/exports/concat.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, it } from "vitest"; +import { concatCsv, splitHeaderAndBody } from "./concat.js"; + +describe("splitHeaderAndBody", () => { + it("splits at the first unquoted newline", () => { + const csv = "a,b,c\nr1c1,r1c2,r1c3\nr2c1,r2c2,r2c3\n"; + const { header, body } = splitHeaderAndBody(csv); + expect(header).toBe("a,b,c"); + expect(body).toBe("r1c1,r1c2,r1c3\nr2c1,r2c2,r2c3\n"); + }); + + it("ignores newlines inside quoted fields when finding the header boundary", () => { + // Quoted cell on the header row's second column contains a newline. + const csv = `"col\nwith\nnewlines",b\nr1c1,r1c2\n`; + const { header, body } = splitHeaderAndBody(csv); + expect(header).toBe(`"col\nwith\nnewlines",b`); + expect(body).toBe("r1c1,r1c2\n"); + }); + + it("handles CRLF line endings", () => { + const csv = "a,b\r\nr1c1,r1c2\r\n"; + const { header, body } = splitHeaderAndBody(csv); + expect(header).toBe("a,b"); + expect(body).toBe("r1c1,r1c2\r\n"); + }); + + it("handles doubled-quote escape without flipping in-quotes state", () => { + const csv = `a,b\n"field with ""embedded"" quotes",normal\n`; + const { header, body } = splitHeaderAndBody(csv); + expect(header).toBe("a,b"); + expect(body).toBe(`"field with ""embedded"" quotes",normal\n`); + }); +}); + +describe("concatCsv", () => { + it("preserves quoted rows that contain embedded newlines (regression: PR #32 Codex P1)", () => { + // Two per-repo CSVs whose bodies include quoted cells with real \n + // inside. The pre-fix split("\n") implementation would turn each of + // those quoted cells into multiple broken rows. + const a = `repo,evidence\nalpha,"line one\nline two"\n`; + const b = `repo,evidence\nbeta,"line three\nline four"\n`; + const out = concatCsv([a, b]); + expect(out).toBe( + `repo,evidence\nalpha,"line one\nline two"\nbeta,"line three\nline four"\n`, + ); + }); + + it("keeps exactly one header regardless of input count", () => { + const a = "h1,h2\nr1a,r1b\n"; + const b = "h1,h2\nr2a,r2b\n"; + const c = "h1,h2\nr3a,r3b\n"; + const out = concatCsv([a, b, c]); + expect(out.split("\n").filter(l => l === "h1,h2").length).toBe(1); + }); + + it("returns an empty string for no inputs", () => { + expect(concatCsv([])).toBe(""); + }); + + it("drops inputs with no body (header-only CSVs) without leaving blank rows", () => { + const empty = "repo,evidence\n"; + const data = `repo,evidence\nbeta,ok\n`; + const out = concatCsv([empty, data]); + expect(out).toBe("repo,evidence\nbeta,ok\n"); + }); +}); diff --git a/scanner/generators/exports/concat.ts b/scanner/generators/exports/concat.ts new file mode 100644 index 0000000..614aa9b --- /dev/null +++ b/scanner/generators/exports/concat.ts @@ -0,0 +1,43 @@ +/** + * CSV concatenation helper used by the dashboard's org-level export endpoints. + * Lives in the scanner/generators/exports/ module so it sits next to the + * emitters it operates on — and so it can be unit-tested independently of + * the Cloudflare Worker runtime. + */ + +/** + * Split a CSV document into (header, body) at the first UNQUOTED newline. + * RFC 4180 allows raw `\n` inside quoted fields — risk-register and + * evidence columns commonly contain them. A naive `split("\n")` corrupts + * those rows by turning one logical record into several broken ones, so + * we walk the string character-by-character tracking in-quotes state. + */ +export function splitHeaderAndBody(csv: string): { header: string; body: string } { + let inQuotes = false; + for (let i = 0; i < csv.length; i++) { + const ch = csv[i]; + if (ch === '"') { + // RFC 4180 escapes an embedded quote by doubling it. Skip the next + // quote so we stay in the same in-quotes state across `""`. + if (inQuotes && csv[i + 1] === '"') { i++; continue; } + inQuotes = !inQuotes; + } else if ((ch === "\n" || ch === "\r") && !inQuotes) { + const header = csv.slice(0, i); + const skip = ch === "\r" && csv[i + 1] === "\n" ? 2 : 1; + return { header, body: csv.slice(i + skip) }; + } + } + return { header: csv, body: "" }; +} + +/** Concatenate multiple per-repo CSVs into one table keeping a single header. */ +export function concatCsv(csvs: string[]): string { + if (csvs.length === 0) return ""; + const first = splitHeaderAndBody(csvs[0]!); + const bodies: string[] = [first.body]; + for (let i = 1; i < csvs.length; i++) { + bodies.push(splitHeaderAndBody(csvs[i]!).body); + } + const cleaned = bodies.map(b => b.replace(/\n+$/, "")).filter(b => b.length > 0); + return first.header + "\n" + cleaned.join("\n") + "\n"; +} diff --git a/scanner/generators/exports/csv.ts b/scanner/generators/exports/csv.ts new file mode 100644 index 0000000..0020134 --- /dev/null +++ b/scanner/generators/exports/csv.ts @@ -0,0 +1,117 @@ +import type { Manifest } from "../../types.js"; +import type { ControlResult } from "../framework-report.js"; +import type { Risk } from "../risk-assessment.js"; +import type { AIComplianceResult } from "../../types.js"; + +/** + * CSV export — one row per finding, flattened. Covers the four "finding" + * tables an auditor workflow cares about: + * + * - `nist-csf.csv`: per-control status + evidence + * - `eu-ai-act.csv`: per-article status + evidence + * - `risks.csv`: the full risk register (security + ai-compliance categories) + * - `vulnerabilities.csv`: per-CVE dependency advisories + * + * Flat-file audit workpapers expect this shape; most spreadsheet tools + * drop CSVs straight into a pivot table. + */ + +function csvEscape(value: unknown): string { + const s = value === undefined || value === null ? "" : String(value); + // RFC 4180: quote anything containing a comma, quote, or newline; + // double-up internal quotes. + if (/[",\n\r]/.test(s)) { + return `"${s.replace(/"/g, '""')}"`; + } + return s; +} + +function toCsv(header: string[], rows: Array>): string { + const lines = [header.map(csvEscape).join(",")]; + for (const row of rows) lines.push(row.map(csvEscape).join(",")); + return lines.join("\n") + "\n"; +} + +export function generateCsvNistCsf(manifest: Manifest, results: ControlResult[]): string { + return toCsv( + ["repo", "branch", "commit", "scan_date", "control_id", "function", "category", "subcategory", "status", "evidence", "soc2", "iso27001"], + results.map(r => [ + manifest.repo, + manifest.branch, + manifest.commit, + manifest.scanDate, + r.control.id, + r.control.function, + r.control.category, + r.control.subcategory, + r.status, + r.evidence, + r.soc2.join("; "), + r.iso27001.join("; "), + ]), + ); +} + +export function generateCsvEuAiAct(manifest: Manifest, results: AIComplianceResult[]): string { + return toCsv( + ["repo", "branch", "commit", "scan_date", "article_id", "article", "title", "phase", "status", "evidence", "nist_ai_rmf", "iso42001"], + results.map(r => [ + manifest.repo, + manifest.branch, + manifest.commit, + manifest.scanDate, + r.articleId, + r.article, + r.title, + r.phase, + r.status, + r.evidence, + r.nistAiRmf.join("; "), + r.iso42001.join("; "), + ]), + ); +} + +export function generateCsvRisks(manifest: Manifest, risks: Risk[]): string { + return toCsv( + ["repo", "branch", "commit", "scan_date", "risk_id", "category", "title", "severity", "likelihood", "impact", "status", "mitigation", "frameworks", "description"], + risks.map(r => [ + manifest.repo, + manifest.branch, + manifest.commit, + manifest.scanDate, + r.id, + r.category, + r.title, + r.severity, + r.likelihood, + r.impact, + r.status, + r.mitigation, + r.framework.join("; "), + r.description, + ]), + ); +} + +export function generateCsvVulnerabilities(manifest: Manifest): string { + const vulns = manifest.vulnerabilities ?? []; + return toCsv( + ["repo", "branch", "commit", "scan_date", "advisory_id", "package", "severity", "title", "range", "cvss_score", "is_direct", "fix_available", "url"], + vulns.map(v => [ + manifest.repo, + manifest.branch, + manifest.commit, + manifest.scanDate, + v.advisoryId, + v.package, + v.severity, + v.title, + v.range, + v.cvssScore, + v.isDirect ? "true" : "false", + v.fixAvailable ? "true" : "false", + v.url, + ]), + ); +} diff --git a/scanner/generators/exports/exports.test.ts b/scanner/generators/exports/exports.test.ts new file mode 100644 index 0000000..30b207c --- /dev/null +++ b/scanner/generators/exports/exports.test.ts @@ -0,0 +1,210 @@ +import { describe, expect, it } from "vitest"; +import type { Manifest, AIComplianceResult } from "../../types.js"; +import type { ControlResult } from "../framework-report.js"; +import type { Risk } from "../risk-assessment.js"; +import { generateJsonExport } from "./json.js"; +import { generateCsvNistCsf, generateCsvRisks, generateCsvVulnerabilities } from "./csv.js"; +import { generateSarifExport } from "./sarif.js"; +import { generateOscalExport } from "./oscal.js"; + +function baseManifest(overrides: Partial = {}): Manifest { + return { + repo: "test/repo", + scanDate: "2026-04-19T00:00:00Z", + branch: "main", + commit: "abc1234", + dataCollection: [], + thirdPartyServices: [], + securityHeaders: null, + https: null, + dependencies: null, + secretsScan: { detected: false, findings: [] }, + artifacts: { + privacyPolicy: "generated", + termsOfService: "generated", + securityTxt: "present", + vulnerabilityDisclosure: "present", + incidentResponsePlan: "present", + }, + accessControls: { branchProtection: true, requiredReviews: 1, signedCommits: false }, + aiSystems: [], + ...overrides, + }; +} + +const emptyNist: ControlResult[] = []; +const emptyEu: AIComplianceResult[] = []; +const emptyRisks: Risk[] = []; + +describe("JSON export", () => { + it("produces a well-formed GRCExport envelope with schema + schemaVersion", () => { + const out = generateJsonExport(baseManifest(), emptyNist, emptyEu, emptyRisks); + const parsed = JSON.parse(out); + expect(parsed.schema).toBe("grc-export"); + expect(parsed.schemaVersion).toBe("1.0"); + expect(parsed.manifest.repo).toBe("test/repo"); + expect(parsed.generatedAt).toMatch(/^\d{4}-\d{2}-\d{2}T/); + expect(Array.isArray(parsed.nistCsf)).toBe(true); + expect(Array.isArray(parsed.euAiAct)).toBe(true); + expect(Array.isArray(parsed.risks)).toBe(true); + }); +}); + +describe("CSV exports", () => { + it("quotes fields containing commas, quotes, and newlines", () => { + // Use a risk whose description contains each hazard character. + const risks: Risk[] = [{ + id: "R1", + category: "governance", + title: 'Policy "thing"', + description: "Line one,\nline two", + likelihood: "low", + impact: "low", + severity: "low", + status: "open", + mitigation: "fix it", + framework: [], + }]; + const out = generateCsvRisks(baseManifest(), risks); + const header = out.split("\n")[0]; + expect(header).toContain("repo"); + expect(header).toContain("title"); + // The title column must be double-quoted and contain escaped inner quotes. + expect(out).toContain('"Policy ""thing"""'); + // Description with a newline and comma must also be quoted. + expect(out).toContain('"Line one,\nline two"'); + }); + + it("emits one row per control + a header", () => { + const nist: ControlResult[] = [ + { + control: { id: "ID.GV-1", function: "Identify", category: "Governance", subcategory: "x", description: "y", check: () => "pass", evidence: () => "" }, + status: "pass", evidence: "ok", soc2: ["CC1.1"], iso27001: ["A.5"], + }, + ]; + const out = generateCsvNistCsf(baseManifest(), nist); + const lines = out.trim().split("\n"); + expect(lines.length).toBe(2); // header + 1 row + expect(lines[1]).toContain("ID.GV-1"); + expect(lines[1]).toContain("CC1.1"); + }); + + it("gracefully handles a manifest missing the optional vulnerabilities field", () => { + // Important for pre-Phase-9 manifests in KV. + const out = generateCsvVulnerabilities(baseManifest()); + const lines = out.trim().split("\n"); + expect(lines.length).toBe(1); // header only + }); +}); + +describe("SARIF export", () => { + it("produces a valid SARIF envelope with $schema and version", () => { + const out = generateSarifExport(baseManifest()); + const parsed = JSON.parse(out); + expect(parsed.version).toBe("2.1.0"); + expect(parsed.$schema).toContain("sarif-schema-2.1.0"); + expect(Array.isArray(parsed.runs)).toBe(true); + expect(parsed.runs.length).toBe(1); + expect(parsed.runs[0].tool.driver.name).toBe("grc-observability-dashboard"); + }); + + it("emits a dependency-vulnerability result per CVE with SARIF level mapped from severity", () => { + const manifest = baseManifest({ + vulnerabilities: [ + { package: "lodash", advisoryId: "1", severity: "critical", title: "RCE", range: "<4.0", url: "https://x", cvssScore: 9.8, isDirect: true, fixAvailable: true, paths: [] }, + { package: "qs", advisoryId: "2", severity: "moderate", title: "ReDoS", range: "<1.0", url: "https://y", cvssScore: 5.1, isDirect: false, fixAvailable: false, paths: [] }, + ], + }); + const parsed = JSON.parse(generateSarifExport(manifest)); + const results = parsed.runs[0].results; + expect(results.length).toBe(2); + expect(results[0].ruleId).toBe("grc/dependency-vulnerability"); + expect(results[0].level).toBe("error"); // critical → error + expect(results[1].level).toBe("warning"); // moderate → warning + }); + + it("emits ai-prohibited and ai-high-risk results when systems match", () => { + const manifest = baseManifest({ + aiSystems: [ + { provider: "OpenAI", sdk: "openai", location: "package.json", category: "inference", dataFlows: [], riskTier: "high", riskTierSource: "heuristic", usageLocations: ["src/hiring/screen.ts"], euMarket: true }, + { provider: "OpenAI", sdk: "openai", location: "package.json", category: "inference", dataFlows: [], riskTier: "prohibited", riskTierSource: "heuristic", usageLocations: ["src/social-score/rank.ts"], euMarket: true }, + ], + }); + const parsed = JSON.parse(generateSarifExport(manifest)); + const ruleIds = parsed.runs[0].results.map((r: { ruleId: string }) => r.ruleId); + expect(ruleIds).toContain("grc/ai-high-risk"); + expect(ruleIds).toContain("grc/ai-prohibited"); + }); + + it("only emits rules that actually produced results (keeps output focused)", () => { + const parsed = JSON.parse(generateSarifExport(baseManifest())); + expect(parsed.runs[0].tool.driver.rules).toEqual([]); + expect(parsed.runs[0].results).toEqual([]); + }); + + it("extracts the file path from the scanner's prose secret finding (regression: PR #32 Codex P1)", () => { + const manifest = baseManifest({ + secretsScan: { + detected: true, + findings: [ + "OpenAI API key found in src/config.ts", + "AWS access key found in lib/creds.ts:42", + ], + }, + }); + const parsed = JSON.parse(generateSarifExport(manifest)); + const results = parsed.runs[0].results; + expect(results.length).toBe(2); + // The URI must be just the path, not the prose label. + expect(results[0].locations[0].physicalLocation.artifactLocation.uri).toBe("src/config.ts"); + expect(results[0].message.text).toContain("OpenAI API key"); + expect(results[0].message.text).not.toContain("found in"); + + expect(results[1].locations[0].physicalLocation.artifactLocation.uri).toBe("lib/creds.ts"); + expect(results[1].locations[0].physicalLocation.region.startLine).toBe(42); + }); +}); + +describe("OSCAL export", () => { + it("produces a valid assessment-results envelope with OSCAL 1.1.2", () => { + const out = generateOscalExport(baseManifest(), emptyNist, emptyEu); + const parsed = JSON.parse(out); + const ar = parsed["assessment-results"]; + expect(ar.metadata["oscal-version"]).toBe("1.1.2"); + expect(ar.metadata.title).toContain("test/repo"); + expect(ar.metadata.title).toContain("main"); + expect(typeof ar.uuid).toBe("string"); + // UUID v4 pattern, not strict but should match the generic shape. + expect(ar.uuid).toMatch(/^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/); + }); + + it("emits one OSCAL result per framework — NIST always, EU AI Act only when results present", () => { + const nistResults: ControlResult[] = [ + { control: { id: "ID.GV-1", function: "Identify", category: "Governance", subcategory: "x", description: "y", check: () => "pass", evidence: () => "" }, status: "pass", evidence: "ok", soc2: [], iso27001: [] }, + ]; + + // No EU AI Act results → only one OSCAL result in the output. + const outNoAi = JSON.parse(generateOscalExport(baseManifest(), nistResults, [])); + expect(outNoAi["assessment-results"].results.length).toBe(1); + + // With EU AI Act results → two results. + const euResults: AIComplianceResult[] = [ + { articleId: "ART-5", article: 5, title: "Prohibited", phase: "Map", description: "", status: "pass", evidence: "ok", nistAiRmf: [], iso42001: [] }, + ]; + const outWithAi = JSON.parse(generateOscalExport(baseManifest(), nistResults, euResults)); + expect(outWithAi["assessment-results"].results.length).toBe(2); + }); + + it("emits one finding per non-pass non-applicable observation", () => { + const nistResults: ControlResult[] = [ + { control: { id: "A", function: "Identify", category: "", subcategory: "", description: "", check: () => "pass", evidence: () => "" }, status: "pass", evidence: "", soc2: [], iso27001: [] }, + { control: { id: "B", function: "Protect", category: "", subcategory: "", description: "", check: () => "fail", evidence: () => "" }, status: "fail", evidence: "", soc2: [], iso27001: [] }, + { control: { id: "C", function: "Detect", category: "", subcategory: "", description: "", check: () => "partial", evidence: () => "" }, status: "partial", evidence: "", soc2: [], iso27001: [] }, + { control: { id: "D", function: "Respond", category: "", subcategory: "", description: "", check: () => "not-applicable", evidence: () => "" }, status: "not-applicable", evidence: "", soc2: [], iso27001: [] }, + ]; + const out = JSON.parse(generateOscalExport(baseManifest(), nistResults, [])); + const result = out["assessment-results"].results[0]; + expect(result.observations.length).toBe(4); // one per control + expect(result.findings.length).toBe(2); // fail + partial only + }); +}); diff --git a/scanner/generators/exports/json.ts b/scanner/generators/exports/json.ts new file mode 100644 index 0000000..197dd0c --- /dev/null +++ b/scanner/generators/exports/json.ts @@ -0,0 +1,41 @@ +import type { Manifest, AIComplianceResult } from "../../types.js"; +import type { ControlResult } from "../framework-report.js"; +import type { Risk } from "../risk-assessment.js"; + +/** + * "Enhanced JSON" export — the raw manifest plus everything computed off + * it (NIST CSF evaluation, EU AI Act evaluation, risk register). Intended + * as a single well-formed blob that downstream tooling or custom scripts + * can ingest without re-running the scanner. + * + * Shape is deliberately stable: top-level keys are snake-case'd via the + * renameKey step below so CSV / OSCAL / SARIF exports can reference the + * same canonical field names regardless of the TypeScript interface. + */ +export interface GRCExport { + schema: "grc-export"; + schemaVersion: "1.0"; + generatedAt: string; + manifest: Manifest; + nistCsf: ControlResult[]; + euAiAct: AIComplianceResult[]; + risks: Risk[]; +} + +export function generateJsonExport( + manifest: Manifest, + nistCsf: ControlResult[], + euAiAct: AIComplianceResult[], + risks: Risk[], +): string { + const payload: GRCExport = { + schema: "grc-export", + schemaVersion: "1.0", + generatedAt: new Date().toISOString(), + manifest, + nistCsf, + euAiAct, + risks, + }; + return JSON.stringify(payload, null, 2) + "\n"; +} diff --git a/scanner/generators/exports/oscal.ts b/scanner/generators/exports/oscal.ts new file mode 100644 index 0000000..b268b29 --- /dev/null +++ b/scanner/generators/exports/oscal.ts @@ -0,0 +1,250 @@ +import type { Manifest, AIComplianceResult } from "../../types.js"; +import type { ControlResult } from "../framework-report.js"; + +/** + * OSCAL Assessment Results export (model: ar, schema version 1.1.2). + * + * Two assessment results, one per framework: + * 1. NIST CSF 2.0 evaluation — cites framework subcategory IDs directly + * (e.g. "ID.GV-1") since NIST CSF is the primary framework. + * 2. EU AI Act evaluation — cites article identifiers ("ART-5") with + * regulation metadata in props. + * + * Each ControlResult / AIComplianceResult becomes one observation. Non-pass + * results also produce a finding referencing the observation, which is the + * OSCAL convention for assessment output that needs downstream remediation + * tracking. + * + * Reference: https://pages.nist.gov/OSCAL/reference/latest/assessment-results/ + * Reference JSON schema: https://github.com/usnistgov/OSCAL + * + * Note: many GRC platforms are still catching up to OSCAL — Hyperproof, + * Drata's custom-control import, and some Vanta paths accept it in partial + * form. Don't expect a lossless round-trip against every vendor. + */ + +// UUID v4 via Web Crypto where available (Node 20 + Workers both expose it). +function uuid(): string { + // Fall back to a simple v4 generator only if crypto.randomUUID isn't present. + if (typeof crypto !== "undefined" && typeof crypto.randomUUID === "function") { + return crypto.randomUUID(); + } + // Deterministic fallback — acceptable because exports are regenerated + // on every scan; no one is supposed to cross-reference UUIDs across runs. + const hex = Array.from({ length: 16 }, () => Math.floor(Math.random() * 256).toString(16).padStart(2, "0")); + hex[6] = ((parseInt(hex[6]!, 16) & 0x0f) | 0x40).toString(16).padStart(2, "0"); + hex[8] = ((parseInt(hex[8]!, 16) & 0x3f) | 0x80).toString(16).padStart(2, "0"); + return `${hex.slice(0, 4).join("")}-${hex.slice(4, 6).join("")}-${hex.slice(6, 8).join("")}-${hex.slice(8, 10).join("")}-${hex.slice(10).join("")}`; +} + +function oscalStatus( + status: "pass" | "fail" | "partial" | "not-applicable", +): "satisfied" | "not-satisfied" { + // OSCAL's assessment model only distinguishes satisfied / not-satisfied + // on a finding. We surface partial / not-applicable detail via props so + // downstream tooling can re-hydrate the nuance when it cares. + return status === "pass" ? "satisfied" : "not-satisfied"; +} + +interface OscalObservation { + uuid: string; + title: string; + description: string; + methods: string[]; + collected: string; + subjects: Array<{ type: string; "subject-uuid": string; title?: string }>; + props?: Array<{ name: string; value: string; ns?: string }>; + "relevant-evidence"?: Array<{ href: string; description: string }>; +} + +interface OscalFinding { + uuid: string; + title: string; + description: string; + "related-observations": Array<{ "observation-uuid": string }>; + target: { + "target-id": string; + type: "statement-id" | "objective-id"; + status: { state: "satisfied" | "not-satisfied" }; + }; +} + +interface OscalResult { + uuid: string; + title: string; + description: string; + start: string; + "reviewed-controls": { + "control-selections": Array<{ + description: string; + "include-controls": Array<{ "control-id": string }>; + }>; + }; + observations: OscalObservation[]; + findings: OscalFinding[]; +} + +interface OscalAssessmentResults { + "assessment-results": { + uuid: string; + metadata: { + title: string; + "last-modified": string; + version: string; + "oscal-version": "1.1.2"; + }; + "import-ap": { href: string }; + results: OscalResult[]; + }; +} + +function nistResultToResult( + manifest: Manifest, + results: ControlResult[], +): OscalResult { + // Stable subject UUID per repo so observations across scans can be + // correlated if a consumer stores them. Deliberately not using + // manifest.commit — subject is "the repo", not "this commit". + const repoSubjectUuid = uuid(); + + const observations: OscalObservation[] = results.map(r => ({ + uuid: uuid(), + title: `${r.control.id}: ${r.control.subcategory}`, + description: r.evidence, + methods: ["AUTOMATED"], + collected: manifest.scanDate, + subjects: [{ + type: "resource", + "subject-uuid": repoSubjectUuid, + title: manifest.repo, + }], + props: [ + { name: "control-id", value: r.control.id, ns: "https://nist.gov/ns/oscal/grc/nist-csf" }, + { name: "function", value: r.control.function, ns: "https://nist.gov/ns/oscal/grc/nist-csf" }, + { name: "status", value: r.status, ns: "https://nist.gov/ns/oscal/grc/status" }, + ...r.soc2.map(id => ({ name: "cross-ref-soc2", value: id })), + ...r.iso27001.map(id => ({ name: "cross-ref-iso27001", value: id })), + ], + })); + + const findings: OscalFinding[] = results + .filter(r => r.status !== "pass" && r.status !== "not-applicable") + .map(r => { + const obs = observations.find(o => o.title.startsWith(`${r.control.id}:`))!; + return { + uuid: uuid(), + title: `Gap — ${r.control.id}`, + description: r.evidence, + "related-observations": [{ "observation-uuid": obs.uuid }], + target: { + "target-id": r.control.id, + type: "statement-id", + status: { state: oscalStatus(r.status) }, + }, + }; + }); + + return { + uuid: uuid(), + title: "NIST Cybersecurity Framework 2.0 assessment", + description: `Automated scan of ${manifest.repo} at commit ${manifest.commit} (branch ${manifest.branch}) against the 18 NIST CSF 2.0 subcategories this scanner evaluates.`, + start: manifest.scanDate, + "reviewed-controls": { + "control-selections": [{ + description: "NIST CSF 2.0 subcategories evaluated by the scanner.", + "include-controls": results.map(r => ({ "control-id": r.control.id })), + }], + }, + observations, + findings, + }; +} + +function euAiActResultToResult( + manifest: Manifest, + results: AIComplianceResult[], +): OscalResult { + const repoSubjectUuid = uuid(); + + const observations: OscalObservation[] = results.map(r => ({ + uuid: uuid(), + title: `${r.articleId}: ${r.title}`, + description: r.evidence, + methods: ["AUTOMATED"], + collected: manifest.scanDate, + subjects: [{ + type: "resource", + "subject-uuid": repoSubjectUuid, + title: manifest.repo, + }], + props: [ + { name: "article-id", value: r.articleId, ns: "https://ec.europa.eu/ns/oscal/grc/eu-ai-act" }, + { name: "article-number", value: String(r.article), ns: "https://ec.europa.eu/ns/oscal/grc/eu-ai-act" }, + { name: "phase", value: r.phase, ns: "https://www.nist.gov/ns/oscal/grc/ai-rmf" }, + { name: "status", value: r.status, ns: "https://nist.gov/ns/oscal/grc/status" }, + ...r.nistAiRmf.map(id => ({ name: "cross-ref-nist-ai-rmf", value: id })), + ...r.iso42001.map(id => ({ name: "cross-ref-iso42001", value: id })), + ], + })); + + const findings: OscalFinding[] = results + .filter(r => r.status !== "pass" && r.status !== "not-applicable") + .map(r => { + const obs = observations.find(o => o.title.startsWith(`${r.articleId}:`))!; + return { + uuid: uuid(), + title: `Gap — EU AI Act ${r.articleId}`, + description: r.evidence, + "related-observations": [{ "observation-uuid": obs.uuid }], + target: { + "target-id": r.articleId, + type: "statement-id", + status: { state: oscalStatus(r.status) }, + }, + }; + }); + + return { + uuid: uuid(), + title: "EU AI Act assessment", + description: `Automated scan of ${manifest.repo} at commit ${manifest.commit} (branch ${manifest.branch}) against 13 EU AI Act articles.`, + start: manifest.scanDate, + "reviewed-controls": { + "control-selections": [{ + description: "EU AI Act articles evaluated by the scanner.", + "include-controls": results.map(r => ({ "control-id": r.articleId })), + }], + }, + observations, + findings, + }; +} + +export function generateOscalExport( + manifest: Manifest, + nistCsf: ControlResult[], + euAiAct: AIComplianceResult[], +): string { + const payload: OscalAssessmentResults = { + "assessment-results": { + uuid: uuid(), + metadata: { + title: `GRC Assessment — ${manifest.repo} @ ${manifest.branch} (${manifest.commit})`, + "last-modified": new Date().toISOString(), + version: "1.0", + "oscal-version": "1.1.2", + }, + // Assessment Plans are a separate OSCAL artifact; we don't ship one. + // Reference a well-known placeholder URI so consumers know the plan is + // implied by the scanner's own documentation rather than an upstream + // file they need to fetch. + "import-ap": { href: "https://github.com/shipstuff/GRC-Observability-Dashboard#assessment-plan" }, + results: [ + nistResultToResult(manifest, nistCsf), + ...(euAiAct.length > 0 ? [euAiActResultToResult(manifest, euAiAct)] : []), + ], + }, + }; + + return JSON.stringify(payload, null, 2) + "\n"; +} diff --git a/scanner/generators/exports/sarif.ts b/scanner/generators/exports/sarif.ts new file mode 100644 index 0000000..a628d6b --- /dev/null +++ b/scanner/generators/exports/sarif.ts @@ -0,0 +1,306 @@ +import type { Manifest } from "../../types.js"; + +/** + * Minimal SARIF 2.1.0 generator. The spec is huge (dozens of optional + * fields, taxonomies, invocations, fingerprints, flow graphs…) but GitHub's + * code-scanning endpoint only strictly requires a small subset. We emit the + * required fields plus the ones that make the output useful in the GitHub + * Security tab: ruleId, severity, message, physicalLocation with an + * artifactLocation URI relative to the repo root. + * + * SARIF reference: https://docs.oasis-open.org/sarif/sarif/v2.1.0/ + * GitHub-specific notes: https://docs.github.com/code-security/code-scanning/ + * integrating-with-code-scanning/sarif-support-for-code-scanning + */ + +// Only the fields we actually populate — a local structural subset. +interface SarifRegion { + startLine: number; +} + +interface SarifResult { + ruleId: string; + level: "error" | "warning" | "note" | "none"; + message: { text: string }; + locations: Array<{ + physicalLocation: { + artifactLocation: { uri: string }; + region?: SarifRegion; + }; + }>; + partialFingerprints?: Record; + properties?: Record; +} + +interface SarifRule { + id: string; + name: string; + shortDescription: { text: string }; + fullDescription?: { text: string }; + helpUri?: string; + defaultConfiguration?: { level: "error" | "warning" | "note" | "none" }; + properties?: { tags?: string[] }; +} + +const SCANNER_NAME = "grc-observability-dashboard"; +const SCANNER_INFO_URI = "https://github.com/shipstuff/GRC-Observability-Dashboard"; + +// --- Rule catalog --------------------------------------------------------- + +const RULES: SarifRule[] = [ + { + id: "grc/secret-leak", + name: "SecretLeak", + shortDescription: { text: "A credential-like string was detected in source." }, + fullDescription: { text: "API keys, bearer tokens, OAuth secrets, or private keys embedded in committed source. Rotate the credential and move it to a secret store." }, + helpUri: `${SCANNER_INFO_URI}#secrets`, + defaultConfiguration: { level: "error" }, + properties: { tags: ["security", "secret"] }, + }, + { + id: "grc/dependency-vulnerability", + name: "DependencyVulnerability", + shortDescription: { text: "A dependency has a published security advisory." }, + fullDescription: { text: "A package in the dependency tree has a known CVE or advisory. Consult the advisory URL and upgrade to a patched version where available." }, + helpUri: `${SCANNER_INFO_URI}#dependencies`, + defaultConfiguration: { level: "warning" }, + properties: { tags: ["security", "vulnerability"] }, + }, + { + id: "grc/unprotected-route", + name: "UnprotectedRoute", + shortDescription: { text: "An admin or sensitive route appears to lack authentication." }, + fullDescription: { text: "A route matching an admin / destructive pattern was found with no auth middleware detected in the same file. Verify that access is restricted before shipping." }, + helpUri: `${SCANNER_INFO_URI}#access-controls`, + defaultConfiguration: { level: "warning" }, + properties: { tags: ["security", "authentication"] }, + }, + { + id: "grc/ai-prohibited", + name: "AIProhibited", + shortDescription: { text: "An AI system was flagged as potentially prohibited under EU AI Act Article 5." }, + fullDescription: { text: "Path keywords matched social-scoring, subliminal manipulation, or vulnerability exploitation contexts. Verify the use case; if misclassified, override risk_tier in .grc/config.yml." }, + helpUri: `${SCANNER_INFO_URI}#eu-ai-act`, + defaultConfiguration: { level: "error" }, + properties: { tags: ["ai-compliance", "eu-ai-act", "art-5"] }, + }, + { + id: "grc/ai-high-risk", + name: "AIHighRisk", + shortDescription: { text: "An AI system was classified as high-risk under EU AI Act Annex III." }, + fullDescription: { text: "Path keywords suggest the system operates in employment, credit, healthcare, education, or biometric domains. Article 9/11/12/13/14/15/27 obligations likely apply." }, + helpUri: `${SCANNER_INFO_URI}#eu-ai-act`, + defaultConfiguration: { level: "warning" }, + properties: { tags: ["ai-compliance", "eu-ai-act", "annex-iii"] }, + }, +]; + +const RULE_INDEX = new Map(RULES.map((r, i) => [r.id, i])); + +// --- Location helpers ----------------------------------------------------- + +/** + * Many scanner findings ship their location as "file.ts" or "file.ts:42" — + * split into a URI + optional region. Paths are kept repo-relative because + * GitHub expects that; we'd need a base URI config to produce absolute URIs. + */ +function parseLocation(raw: string): { + uri: string; + region?: SarifRegion; +} { + // Strip any leading ./ and trim. + const trimmed = raw.trim().replace(/^\.\//, ""); + const match = trimmed.match(/^(.+?):(\d+)(?::\d+)?$/); + if (match) { + return { uri: match[1]!, region: { startLine: Number(match[2]) } }; + } + return { uri: trimmed }; +} + +/** + * The scanner's secretsScan.findings field stores human-readable strings + * like "OpenAI API key found in src/foo.ts" or + * "AWS access key found in config/keys.ts:42". We need to reach into that + * string to pull out the filename (and optional line number) for SARIF's + * physicalLocation, and keep the label ("OpenAI API key") for the message. + */ +function parseSecretFinding(raw: string): { label: string; uri: string; region?: SarifRegion } { + const match = raw.match(/^(.+?) found in (.+)$/); + if (!match) { + // Shouldn't happen given the current scanner format, but fall back to + // the whole string as a label with no location rather than putting + // prose into artifactLocation.uri. + return { label: raw, uri: "unknown" }; + } + const label = match[1]!.trim(); + const loc = parseLocation(match[2]!); + return { label, uri: loc.uri, region: loc.region }; +} + +function buildSecretsResults(manifest: Manifest): SarifResult[] { + const findings = manifest.secretsScan?.findings ?? []; + return findings.map(finding => { + const { label, uri, region } = parseSecretFinding(finding); + return { + ruleId: "grc/secret-leak", + level: "error" as const, + message: { text: `${label} detected in ${uri}.` }, + locations: [{ + physicalLocation: { + artifactLocation: { uri }, + ...(region ? { region } : {}), + }, + }], + partialFingerprints: { + "grc.secret/primary": `${uri}:${region?.startLine ?? "0"}:${label}`, + }, + }; + }); +} + +function buildVulnerabilityResults(manifest: Manifest): SarifResult[] { + const vulns = manifest.vulnerabilities ?? []; + return vulns.map(v => { + const level: SarifResult["level"] = + v.severity === "critical" ? "error" : + v.severity === "high" ? "error" : + v.severity === "moderate" ? "warning" : + "note"; + return { + ruleId: "grc/dependency-vulnerability", + level, + message: { + text: `${v.package} ${v.range}: ${v.title} (${v.severity}${v.cvssScore > 0 ? `, CVSS ${v.cvssScore}` : ""})${v.fixAvailable ? " — fix available" : ""}. See ${v.url}.`, + }, + // npm doesn't give a source-file location for the advisory — attach + // it to package.json so the finding has somewhere to land in the UI. + locations: [{ + physicalLocation: { + artifactLocation: { uri: "package.json" }, + }, + }], + partialFingerprints: { + "grc.advisory/id": v.advisoryId, + "grc.advisory/package": v.package, + }, + properties: { + "security-severity": v.cvssScore > 0 ? String(v.cvssScore) : v.severity, + advisoryId: v.advisoryId, + package: v.package, + isDirect: v.isDirect, + fixAvailable: v.fixAvailable, + }, + }; + }); +} + +function buildAccessControlResults(manifest: Manifest): SarifResult[] { + // accessControls aggregate doesn't carry per-finding locations; the real + // per-finding data lives in authFindings which isn't on the manifest. + // We synthesise one SARIF result when branchProtection is disabled so + // GitHub Security surfaces the governance gap in the PR Security tab. + const ac = manifest.accessControls; + if (ac?.branchProtection === false) { + return [{ + ruleId: "grc/unprotected-route", + level: "warning", + message: { text: "Branch protection is disabled on this repository. Enabling branch protection with required reviewers is a baseline governance control for NIST CSF PR.AC-4 and SOC 2 CC6.1." }, + locations: [{ + physicalLocation: { artifactLocation: { uri: ".github/branch-protection.md" } }, + }], + partialFingerprints: { "grc.governance/primary": `${manifest.repo}/branch-protection` }, + }]; + } + return []; +} + +function buildAIResults(manifest: Manifest): SarifResult[] { + const out: SarifResult[] = []; + for (const s of manifest.aiSystems ?? []) { + if (!s.riskTier) continue; + if (s.riskTier !== "prohibited" && s.riskTier !== "high") continue; + + const ruleId = s.riskTier === "prohibited" ? "grc/ai-prohibited" : "grc/ai-high-risk"; + // Emit one result per source path where the system is used; fall back + // to the primary `location` if no usageLocations were recorded. + const paths = s.usageLocations && s.usageLocations.length > 0 + ? s.usageLocations + : [s.location]; + + for (const path of paths) { + out.push({ + ruleId, + level: s.riskTier === "prohibited" ? "error" : "warning", + message: { + text: `${s.provider} (${s.sdk}) classified as ${s.riskTier}${s.riskTierSource === "override" ? " (user override)" : ""}. ${s.riskReasoning ?? ""}`.trim(), + }, + locations: [{ physicalLocation: { artifactLocation: { uri: path } } }], + partialFingerprints: { + "grc.ai/primary": `${s.provider}@${path}`, + }, + properties: { + provider: s.provider, + category: s.category, + riskTier: s.riskTier, + euMarket: s.euMarket ?? false, + }, + }); + } + } + return out; +} + +// --- Public entry point --------------------------------------------------- + +export function generateSarifExport(manifest: Manifest): string { + const results: SarifResult[] = [ + ...buildSecretsResults(manifest), + ...buildVulnerabilityResults(manifest), + ...buildAccessControlResults(manifest), + ...buildAIResults(manifest), + ]; + + // Only include rules that actually produced at least one result; GitHub + // doesn't require this, but it keeps the SARIF file focused and matches + // common generator output. + const usedRuleIds = new Set(results.map(r => r.ruleId)); + const rules = RULES.filter(r => usedRuleIds.has(r.id)); + // Rewire ruleIndex now that we've filtered. + const indexMap = new Map(rules.map((r, i) => [r.id, i])); + const resultsWithIndex = results.map(r => ({ + ...r, + ruleIndex: indexMap.get(r.ruleId), + })); + + // Avoid an unused-var complaint while keeping RULE_INDEX around for + // documentation + potential future use when we surface it from a helper. + void RULE_INDEX; + + const sarif = { + $schema: "https://raw.githubusercontent.com/oasis-tcs/sarif-spec/master/Schemata/sarif-schema-2.1.0.json", + version: "2.1.0", + runs: [{ + tool: { + driver: { + name: SCANNER_NAME, + informationUri: SCANNER_INFO_URI, + version: "1.0.0", + rules, + }, + }, + invocations: [{ + executionSuccessful: true, + startTimeUtc: manifest.scanDate, + endTimeUtc: manifest.scanDate, + }], + versionControlProvenance: [{ + repositoryUri: `https://github.com/${manifest.repo}`, + revisionId: manifest.commit, + branch: manifest.branch, + }], + results: resultsWithIndex, + }], + }; + + return JSON.stringify(sarif, null, 2) + "\n"; +} diff --git a/scanner/index.ts b/scanner/index.ts index 16b8634..8a58b04 100644 --- a/scanner/index.ts +++ b/scanner/index.ts @@ -22,6 +22,10 @@ import { assessRisks, generateRiskAssessment } from "./generators/risk-assessmen import { evaluateFramework, generateFrameworkReport } from "./generators/framework-report.js"; import { evaluateEUAIAct, calcAIComplianceScore } from "./frameworks/eu-ai-act.js"; import { generateAIComplianceReport } from "./generators/ai-compliance-report.js"; +import { generateJsonExport } from "./generators/exports/json.js"; +import { generateCsvNistCsf, generateCsvEuAiAct, generateCsvRisks, generateCsvVulnerabilities } from "./generators/exports/csv.js"; +import { generateSarifExport } from "./generators/exports/sarif.js"; +import { generateOscalExport } from "./generators/exports/oscal.js"; import { runAIEnhancements } from "./ai/enhance.js"; import { generateAIReport } from "./ai/report.js"; import { execFile } from "node:child_process"; @@ -82,7 +86,7 @@ export async function scan(repoPath: string, siteUrl: string | null): Promise r.status !== "not-applicable").length; console.log(`📄 EU AI Act report written to ${aiComplianceReportPath} (${aiComplianceScore}% across ${aiApplicable} applicable articles)`); + // Phase 9 Sub-phase A: machine-readable exports for GRC-platform ingestion. + // All exports go to <.grc>/exports/ — regenerated every scan, always + // gitignored. The filename encodes branch + short commit so the files + // don't collide across branches on a local workstation. + const exportsDir = resolve(grcDir, "exports"); + await mkdir(exportsDir, { recursive: true }); + const safeBranch = manifest.branch.replace(/[^\w.-]/g, "-"); + const stem = `${manifest.repo.replace(/\//g, "-")}-${safeBranch}-${manifest.commit.slice(0, 7)}`; + const [jsonOut, sarifOut, oscalOut, csvNist, csvAI, csvRisks, csvVulns] = [ + generateJsonExport(manifest, frameworkResults, aiComplianceResults, risks), + generateSarifExport(manifest), + generateOscalExport(manifest, frameworkResults, aiComplianceResults), + generateCsvNistCsf(manifest, frameworkResults), + generateCsvEuAiAct(manifest, aiComplianceResults), + generateCsvRisks(manifest, risks), + generateCsvVulnerabilities(manifest), + ]; + await Promise.all([ + writeFile(resolve(exportsDir, `${stem}.json`), jsonOut, "utf-8"), + writeFile(resolve(exportsDir, `${stem}.sarif`), sarifOut, "utf-8"), + writeFile(resolve(exportsDir, `${stem}.oscal.json`), oscalOut, "utf-8"), + writeFile(resolve(exportsDir, `${stem}-nist-csf.csv`), csvNist, "utf-8"), + writeFile(resolve(exportsDir, `${stem}-eu-ai-act.csv`), csvAI, "utf-8"), + writeFile(resolve(exportsDir, `${stem}-risks.csv`), csvRisks, "utf-8"), + writeFile(resolve(exportsDir, `${stem}-vulnerabilities.csv`), csvVulns, "utf-8"), + ]); + console.log(`📦 Exports written to ${exportsDir}/${stem}.{json,sarif,oscal.json} + 4 CSVs`); + // Run AI enhancements (optional — graceful degradation) const aiEnhancements = await runAIEnhancements(config, manifest, risks, frameworkResults); if (aiEnhancements) { diff --git a/scanner/rules/dependencies.ts b/scanner/rules/dependencies.ts index 0f7eb6e..1ee448d 100644 --- a/scanner/rules/dependencies.ts +++ b/scanner/rules/dependencies.ts @@ -1,11 +1,94 @@ import { join } from "node:path"; -import { ScanContext, ThirdPartyService, DependencyInfo } from "../types.js"; +import { ScanContext, ThirdPartyService, DependencyInfo, DependencyVulnerability } from "../types.js"; import { readFileContent, fileExists } from "../utils.js"; import { execFile } from "node:child_process"; import { promisify } from "node:util"; const exec = promisify(execFile); +/** + * Shape of one `via` entry in npm audit's vulnerabilities object when the + * entry is an actual advisory (as opposed to a string referencing another + * package in the dep tree). npm may add fields over time; we pluck only + * what the exporters need. + */ +interface NpmAuditVia { + source?: number; + name?: string; + title?: string; + url?: string; + severity?: string; + cvss?: { score?: number }; + range?: string; +} + +interface NpmAuditVulnEntry { + name?: string; + severity?: string; + isDirect?: boolean; + via?: Array; + range?: string; + nodes?: string[]; + fixAvailable?: boolean | { name?: string; version?: string }; +} + +/** + * Flatten npm audit's nested structure into one DependencyVulnerability per + * (package, advisory) pair. `via` entries that are strings are inter-package + * references — we skip those; they show up attached to whichever leaf + * advisory eventually surfaces. + */ +function parseNpmAudit(audit: { + vulnerabilities?: Record; +}): DependencyVulnerability[] { + const vulnerabilities = audit.vulnerabilities; + if (!vulnerabilities || typeof vulnerabilities !== "object") return []; + + const out: DependencyVulnerability[] = []; + const seenAdvisoryIds = new Set(); + + for (const [pkgName, entry] of Object.entries(vulnerabilities)) { + const severity = (entry.severity ?? "low") as DependencyVulnerability["severity"]; + if (!["critical", "high", "moderate", "low"].includes(severity)) continue; + + const viaAdvisories = (entry.via ?? []).filter( + (v): v is NpmAuditVia => typeof v === "object" && v !== null, + ); + if (viaAdvisories.length === 0) continue; + + for (const via of viaAdvisories) { + const advisoryId = via.source != null + ? String(via.source) + : `${pkgName}:${via.title ?? "unknown"}`; + if (seenAdvisoryIds.has(advisoryId)) continue; + seenAdvisoryIds.add(advisoryId); + + out.push({ + package: pkgName, + advisoryId, + severity: (via.severity as DependencyVulnerability["severity"]) ?? severity, + title: via.title ?? "Unspecified advisory", + range: via.range ?? entry.range ?? "*", + url: via.url ?? `https://www.npmjs.com/advisories?search=${encodeURIComponent(pkgName)}`, + cvssScore: via.cvss?.score ?? 0, + isDirect: entry.isDirect === true, + fixAvailable: entry.fixAvailable != null && entry.fixAvailable !== false, + paths: (entry.nodes ?? []).slice(0, 5), // cap to keep the manifest compact + }); + } + } + + // Stable ordering — critical first, then by package name. Exports that + // iterate in order (CSV rows, SARIF results) benefit from reproducibility. + const severityRank: Record = { critical: 0, high: 1, moderate: 2, low: 3 }; + out.sort((a, b) => { + const s = (severityRank[a.severity] ?? 9) - (severityRank[b.severity] ?? 9); + if (s !== 0) return s; + return a.package.localeCompare(b.package); + }); + return out; +} + // Known third-party services and their data implications const KNOWN_SERVICES: Record = { // Email @@ -46,9 +129,14 @@ const KNOWN_SERVICES: Record { +export async function scanDependencies(ctx: ScanContext): Promise<{ + services: ThirdPartyService[]; + deps: DependencyInfo | null; + vulnerabilities: DependencyVulnerability[]; +}> { const services: ThirdPartyService[] = []; let deps: DependencyInfo | null = null; + let vulnerabilities: DependencyVulnerability[] = []; // Check package.json (Node.js) const pkgPath = join(ctx.repoPath, "package.json"); @@ -75,36 +163,39 @@ export async function scanDependencies(ctx: ScanContext): Promise<{ services: Th // Run npm audit if package-lock.json exists const lockPath = join(ctx.repoPath, "package-lock.json"); if (await fileExists(lockPath)) { + // Prefer parsing whatever npm audit returns even on non-zero exit — + // the command returns non-zero whenever advisories exist, which is + // the common case we want to capture. + let auditJson: unknown = null; try { const { stdout } = await exec("npm", ["audit", "--json"], { cwd: ctx.repoPath, timeout: 30000, }); - const audit = JSON.parse(stdout); - const vuln = audit.metadata?.vulnerabilities || {}; - deps = { - criticalVulnerabilities: vuln.critical || 0, - highVulnerabilities: vuln.high || 0, - mediumVulnerabilities: vuln.moderate || 0, - outdatedPackages: 0, - lastAudit: new Date().toISOString().split("T")[0], - }; + auditJson = JSON.parse(stdout); } catch (e: any) { - // npm audit exits non-zero when vulnerabilities are found try { - const audit = JSON.parse(e.stdout || "{}"); - const vuln = audit.metadata?.vulnerabilities || {}; - deps = { - criticalVulnerabilities: vuln.critical || 0, - highVulnerabilities: vuln.high || 0, - mediumVulnerabilities: vuln.moderate || 0, - outdatedPackages: 0, - lastAudit: new Date().toISOString().split("T")[0], - }; + auditJson = JSON.parse(e.stdout || "{}"); } catch { - // Can't parse audit output, skip + auditJson = null; } } + + if (auditJson && typeof auditJson === "object") { + const audit = auditJson as { + metadata?: { vulnerabilities?: { critical?: number; high?: number; moderate?: number } }; + vulnerabilities?: Record; + }; + const vuln = audit.metadata?.vulnerabilities ?? {}; + deps = { + criticalVulnerabilities: vuln.critical ?? 0, + highVulnerabilities: vuln.high ?? 0, + mediumVulnerabilities: vuln.moderate ?? 0, + outdatedPackages: 0, + lastAudit: new Date().toISOString().split("T")[0]!, + }; + vulnerabilities = parseNpmAudit(audit); + } } } @@ -120,5 +211,5 @@ export async function scanDependencies(ctx: ScanContext): Promise<{ services: Th // Future: add Go module scanning } - return { services, deps }; + return { services, deps, vulnerabilities }; } diff --git a/scanner/types.ts b/scanner/types.ts index 1b05a69..a6614d2 100644 --- a/scanner/types.ts +++ b/scanner/types.ts @@ -36,6 +36,34 @@ export interface DependencyInfo { lastAudit: string; } +/** + * One advisory affecting a dependency. Sourced from `npm audit --json` in the + * scanner, flattened one record per (package, advisory) pair so SARIF and + * CSV exports can emit them individually. Older manifests predate this field + * — consumers must treat the whole array as optional. + */ +export interface DependencyVulnerability { + /** Package name the advisory is published against. */ + package: string; + /** GitHub Advisory DB numeric id where available — otherwise a stable hash. */ + advisoryId: string; + severity: "critical" | "high" | "moderate" | "low"; + /** Plain-English title from the advisory source. */ + title: string; + /** Affected semver range. */ + range: string; + /** Link to the GitHub advisory page (or npm advisory URL). */ + url: string; + /** CVSS v3 base score when published. 0 when unknown. */ + cvssScore: number; + /** True if the affected package is a direct dep of this repo. */ + isDirect: boolean; + /** True when npm believes a non-breaking upgrade would fix this. */ + fixAvailable: boolean; + /** Dependency paths (node_modules/... strings) — for SARIF location. */ + paths: string[]; +} + export interface SecretsFindings { detected: boolean; findings: string[]; @@ -138,6 +166,13 @@ export interface Manifest { artifacts: ArtifactStatus; accessControls: AccessControls; aiSystems: AISystem[]; + /** + * Per-advisory detail for dependency vulnerabilities. Optional for + * backward compatibility with older stored manifests that only carried + * the aggregate counts in `dependencies`. SARIF, CSV, and OSCAL + * exports rely on this field to produce per-finding output. + */ + vulnerabilities?: DependencyVulnerability[]; policyUrls?: PolicyUrlsManifest; }