diff --git a/.github/workflows/delphi-characterization.yml b/.github/workflows/delphi-characterization.yml index 78d3d6349c..a657e47def 100644 --- a/.github/workflows/delphi-characterization.yml +++ b/.github/workflows/delphi-characterization.yml @@ -64,6 +64,8 @@ jobs: run: | python -m pip install --quiet pyyaml python -m unittest tests.test_delphi_storage_codec -v + python -m unittest tests.test_delphi_postgres_results -v + python -m unittest tests.test_delphi_result_resource -v - name: Codec golden test (Node reads the files Python wrote, writes the cross file) working-directory: server diff --git a/.github/workflows/dynamo-removal.yml b/.github/workflows/dynamo-removal.yml new file mode 100644 index 0000000000..0618553f7f --- /dev/null +++ b/.github/workflows/dynamo-removal.yml @@ -0,0 +1,63 @@ +name: Dynamo removal + +on: + pull_request: + paths: + - 'delphi/**' + - 'queue-rs/**' + - 'server/**' + - 'client-report/**' + - 'scripts/test-dynamo-removal.sh' + - 'scripts/prove-dynamo-report.sh' + - 'ci/dynamo-removal/**' + - '.github/workflows/dynamo-removal.yml' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: dynamo-removal-${{ github.ref }} + cancel-in-progress: true + +jobs: + postgres-report: + runs-on: ubuntu-24.04 + timeout-minutes: 150 + env: + COMPOSE_PROJECT_NAME: polis-graph-test-dynamo-${{ github.run_id }}-${{ github.run_attempt }} + POLIS_RECOVERY_PG_PORT: '55449' + RECOVERY_PG_PORT: '55449' + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + persist-credentials: false + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5 + with: + python-version: '3.12' + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: '22' + - name: Queue toolchain + working-directory: queue-rs + run: rustup show active-toolchain + - name: Install uv + run: python -m pip install uv==0.9.2 + - name: Queue, Postgres results, importer and report with DynamoDB stopped + run: bash scripts/test-dynamo-removal.sh + - name: Retain proof logs and report screenshots + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: dynamo-removal-proof + retention-days: 7 + include-hidden-files: true + path: | + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/*.log + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/demo-result.json + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/sql-source-sha256.json + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/install/results.json + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/results-sql/results.json + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/report/*.log + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/report/browser.json + .dynamo-proof/${{ env.COMPOSE_PROJECT_NAME }}/report/*.png diff --git a/.github/workflows/dynamo-writers.yml b/.github/workflows/dynamo-writers.yml new file mode 100644 index 0000000000..df38b24aac --- /dev/null +++ b/.github/workflows/dynamo-writers.yml @@ -0,0 +1,27 @@ +name: Postgres Delphi writers +on: + pull_request: + paths: ['delphi/**', 'server/**', 'queue-rs/**', 'scripts/test-dynamo-writers.sh', '.github/workflows/dynamo-writers.yml'] + workflow_dispatch: +permissions: + contents: read +jobs: + writers: + runs-on: ubuntu-24.04 + timeout-minutes: 90 + env: + COMPOSE_PROJECT_NAME: polis-graph-test-writers-${{ github.run_id }}-${{ github.run_attempt }} + POLIS_RECOVERY_PG_PORT: '55453' + RECOVERY_PG_PORT: '55453' + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 + with: + persist-credentials: false + - run: bash scripts/test-dynamo-writers.sh + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: delphi-writers-proof + path: .writer-proof/**/*.log + include-hidden-files: true + retention-days: 7 diff --git a/ci/dynamo-removal/report-nginx.conf b/ci/dynamo-removal/report-nginx.conf new file mode 100644 index 0000000000..ccd01eb882 --- /dev/null +++ b/ci/dynamo-removal/report-nginx.conf @@ -0,0 +1,10 @@ +events {} +http { + include /etc/nginx/mime.types; + server { + listen 8080; + root /report; + location /api/ { proxy_pass http://astra-dynamo1424-server:5000; proxy_read_timeout 180s; } + location / { try_files $uri /index_report.html; } + } +} diff --git a/ci/dynamo-removal/report-proof.cjs b/ci/dynamo-removal/report-proof.cjs new file mode 100644 index 0000000000..94c680d20b --- /dev/null +++ b/ci/dynamo-removal/report-proof.cjs @@ -0,0 +1,78 @@ +/* Read-only browser proof: use generated local reports, never a production URL. */ +const fs=require('node:fs'); +const path=require('node:path'); +const assert=require('node:assert/strict'); +const {assertFixtureReports,assertReportCoverage}=require('./report-quality.cjs'); +const playwright=process.env.DYNAMO_PROOF_PLAYWRIGHT; +assert(playwright,'DYNAMO_PROOF_PLAYWRIGHT must point to an installed local Playwright package'); +const {chromium}=require(playwright); +const base=process.env.DYNAMO_PROOF_REPORT_URL; +assert(/^http:\/\/(127\.0\.0\.1|localhost):\d+$/.test(base),'local browser proof URL required'); +const rid=process.env.DYNAMO_PROOF_REPORT_ID; +assert(/^rlocal[a-z0-9]+$/.test(rid),'generated local report ID required'); +const output=process.env.DYNAMO_PROOF_OUTPUT; +assert(output,'DYNAMO_PROOF_OUTPUT required');fs.mkdirSync(output,{recursive:true}); +(async()=>{ + const browser=await chromium.launch({headless:true});const receipts=[]; + try { + for(const route of ['report','topicStats','topicReport']){ + const page=await browser.newPage({viewport:{width:1440,height:1100}}); + const errors=[];const responses=[];let narrativeEvidence; + page.on('pageerror',error=>errors.push(error.message)); + page.on('response',async response=>{ + if(!response.url().includes('/api/'))return; + let body;try{body=await response.text()}catch{body='unavailable'} + responses.push({url:response.url(),status:response.status(),body}); + }); + await page.goto(`${base}/${route}/${rid}`,{waitUntil:'networkidle'}); + if(route==='report') { + const response=await page.request.get(`${base}/api/v3/delphi/visualizations?report_id=${encodeURIComponent(rid)}`); + assert(response.ok(),'visualization metadata endpoint succeeded'); + const metadata=await response.json();assert.equal(metadata.status,'success'); + assert(metadata.jobs?.length>0,'actual queue metadata returned with DynamoDB unavailable'); + assert(metadata.jobs.every(job=>job.workLive===false),'published completed graph has no live work'); + responses.push({url:response.url(),status:response.status(),body:JSON.stringify(metadata)}); + } + if(route==='topicStats')await page.getByText('Group Consensus',{exact:false}).first().waitFor({timeout:30000}); + if(route==='topicReport'){ + const selector=page.locator('select'); + await selector.first().waitFor({timeout:30000}); + const options=await selector.first().locator('option').evaluateAll(options=>options.map(o=>({value:o.value,text:o.textContent}))); + const sourceResponse=await page.request.get(`${base}/api/v3/delphi/reports?report_id=${encodeURIComponent(rid)}`); + assert(sourceResponse.ok(),'narrative source endpoint succeeded'); + const source=await sourceResponse.json();assert.equal(source.status,'success'); + const checked=assertFixtureReports(source.reports); + const topicResponse=await page.request.get(`${base}/api/v3/delphi?report_id=${encodeURIComponent(rid)}`); + assert(topicResponse.ok(),'topic source endpoint succeeded'); + const topics=await topicResponse.json();assert.equal(topics.status,'success'); + const checkedTopics=Object.values(topics.runs).flatMap(run=>Object.values(run.topics_by_layer).flatMap(Object.values)).length; + assertReportCoverage(source.reports,topics.runs); + const named=options.find(o=>o.value && source.reports?.[o.value]?.report_data); + assert(named,'at least one available narrative section maps to the rendered selector'); + const stored=source.reports[named.value]; + const {clauses}=checked[named.value]; + await selector.first().selectOption(named.value); + await page.waitForLoadState('networkidle'); + await page.locator('.topic-text-content .paragraph').first().waitFor({timeout:30000}); + const rendered=await page.locator('.topic-text-content').innerText(); + for(const clause of clauses)assert(rendered.includes(clause),'stored narrative clause renders exactly'); + narrativeEvidence={section:named.value,model:stored.model,job_id:stored.job_id,clauses:clauses.length,metadata:stored.metadata,checkedSections:Object.keys(checked),checkedTopics}; + } + const body=await page.locator('body').innerText(); + await page.screenshot({path:path.join(output,route+'.png'),fullPage:true}); + await page.screenshot({path:path.join(output,route+'-viewport.png')}); + const record={route,body,errors,responses,narrativeEvidence};receipts.push(record); + fs.writeFileSync(path.join(output,'browser.json'),JSON.stringify(receipts,null,2)); + assert.equal(errors.length,0,route+': '+errors.join('; ')); + assert(responses.length>0,route+': API requests observed'); + assert(responses.every(r=>r.status<400),route+': '+responses.filter(r=>r.status>=400).map(r=>r.url)); + for(const response of responses){let payload;try{payload=JSON.parse(response.body)}catch{continue}assert.notEqual(payload.status,'error',response.url+': application error')} + // Existing date-display behavior is outside this storage/queue proof. + // Preserve it with the default-backend route and view recordings. + if(route==='report')assert(/people voted/.test(body),route+': participant summary rendered'); + if(route==='topicReport')assert(body.length>300,route+': narrative content rendered'); + console.log('PASS',route,'API responses',responses.length); + await page.close(); + } + } finally {await browser.close()} +})().catch(error=>{console.error(error);process.exitCode=1}); diff --git a/ci/dynamo-removal/report-quality.cjs b/ci/dynamo-removal/report-quality.cjs new file mode 100644 index 0000000000..383196285d --- /dev/null +++ b/ci/dynamo-removal/report-quality.cjs @@ -0,0 +1,38 @@ +const assert = require('node:assert/strict'); + +function assertReportCoverage(reports, runs) { + const topics = Object.values(runs || {}).flatMap(run => Object.values(run.topics_by_layer || {}).flatMap(Object.values)); + assert(topics.length > 0, 'topic coverage requires served topics'); + const expected = topics.map(topic => { + assert(/^.+#\d+#\d+$/.test(topic.topic_key), 'valid topic key required'); + return topic.topic_key.replaceAll('#', '_'); + }); + assert.equal(new Set(expected).size, expected.length, 'duplicate topic section'); + const jobs = new Set(topics.map(topic => topic.topic_key.split('#')[0])); + assert.equal(jobs.size, 1, 'one coherent served narrative job required'); + const [job] = jobs; + expected.push(...['groups', 'group_informed_consensus', 'uncertainty'].map(name => `${job}_global_${name}`)); + const actual = Object.entries(reports || {}).map(([key, report]) => { + assert.equal(report.section, key, 'duplicate or mismatched report section'); + assert.equal(report.job_id, job, 'report belongs to served narrative job'); + return key; + }); + assert.deepEqual(actual.sort(), expected.sort(), 'complete topic and global section coverage'); + return expected.length; +} + +function assertFixtureReports(reports) { + assert(reports && Object.keys(reports).length > 0, 'served reports required'); + const checked = {}; + for (const [section, stored] of Object.entries(reports)) { + const document = typeof stored.report_data === 'string' ? JSON.parse(stored.report_data) : stored.report_data; + assert.equal(stored.model, 'local-narrative-fixture/1'); + assert.equal(stored.metadata?.provider_fixture, true); + assert.equal(document.provider_fixture, true); + const clauses = document.paragraphs.flatMap(p => p.sentences.flatMap(s => s.clauses.map(c => c.text))); + assert.deepEqual(clauses, ['Fixed narrative stand-in for queue, Postgres storage and report rendering proof. No LLM provider was called.']); + checked[section] = { document, clauses }; + } + return checked; +} +module.exports = { assertFixtureReports, assertReportCoverage }; diff --git a/ci/dynamo-removal/report-quality.test.cjs b/ci/dynamo-removal/report-quality.test.cjs new file mode 100644 index 0000000000..9e2cca314d --- /dev/null +++ b/ci/dynamo-removal/report-quality.test.cjs @@ -0,0 +1,33 @@ +const { test } = require('node:test'); +const assert = require('node:assert/strict'); +const { assertReportCoverage } = require('./report-quality.cjs'); +const coverageFixture = () => { + const runs = { current: { topics_by_layer: { 0: { + 0: { topic_key: 'generated#0#0' }, 1: { topic_key: 'generated#0#1' } + } } } }; + const sections = ['generated_0_0', 'generated_0_1', 'generated_global_groups', + 'generated_global_group_informed_consensus', 'generated_global_uncertainty']; + return { runs, reports: Object.fromEntries(sections.map(section => [section, { section, job_id: 'generated' }])) }; +}; +test('requires all topic and global sections from one served job', () => { + const { reports, runs } = coverageFixture(); + assert.equal(assertReportCoverage(reports, runs), 5); +}); +test('rejects a missing topic or global even when remaining reports are valid', () => { + for (const missing of ['generated_0_1', 'generated_global_uncertainty']) { + const { reports, runs } = coverageFixture(); delete reports[missing]; + assert.throws(() => assertReportCoverage(reports, runs), /complete topic and global/); + } +}); +test('rejects duplicate topic keys or duplicate report identities', () => { + const { reports, runs } = coverageFixture(); + runs.current.topics_by_layer[0][1].topic_key = 'generated#0#0'; + assert.throws(() => assertReportCoverage(reports, runs), /duplicate topic/); + const fresh = coverageFixture(); + fresh.reports.generated_0_1.section = 'generated_0_0'; + assert.throws(() => assertReportCoverage(fresh.reports, fresh.runs), /duplicate or mismatched/); +}); +test('rejects reports from a different job', () => { + const { reports, runs } = coverageFixture(); reports.generated_0_0.job_id = 'other-job'; + assert.throws(() => assertReportCoverage(reports, runs), /served narrative job/); +}); diff --git a/delphi/Dockerfile b/delphi/Dockerfile index 2484f7d0ab..caadd9300b 100644 --- a/delphi/Dockerfile +++ b/delphi/Dockerfile @@ -137,6 +137,7 @@ RUN install -d /opt/polis-unflip && \ # logs a failed-send error and traceback for every flush, which buried the # poller's own output; tracing is therefore off unless explicitly enabled. CMD ["bash", "-c", "\ + if [ \"${DELPHI_RESULT_BACKEND}\" = \"postgres\" ]; then export POLIS_JOBS_ENABLED=1; exec polis-jobs; fi; \ echo 'Ensuring DynamoDB tables are set up (runs in all environments)...'; \ python create_dynamodb_tables.py --region ${AWS_REGION} && \ echo 'DynamoDB table setup script finished.'; \ diff --git a/delphi/create_dynamodb_tables.py b/delphi/create_dynamodb_tables.py index 12cf966e96..6450785dea 100644 --- a/delphi/create_dynamodb_tables.py +++ b/delphi/create_dynamodb_tables.py @@ -530,6 +530,8 @@ def _create_tables(dynamodb, tables, existing_tables): def create_tables(endpoint_url=None, region_name='us-east-1', delete_existing=False, evoc_only=False, polismath_only=False, aws_profile=None): + if os.environ.get('DELPHI_RESULT_BACKEND') == 'postgres': + return [] # Use the environment variable if endpoint_url is not provided if endpoint_url is None: endpoint_url = os.environ.get('DYNAMODB_ENDPOINT') @@ -601,6 +603,9 @@ def create_tables(endpoint_url=None, region_name='us-east-1', return created_tables def main(): + if os.environ.get('DELPHI_RESULT_BACKEND') == 'postgres': + print('Postgres result schema is managed by migrations; no Dynamo bootstrap') + return # Parse arguments parser = argparse.ArgumentParser(description='Create DynamoDB tables for Delphi system') parser.add_argument('--endpoint-url', type=str, default=None, diff --git a/delphi/docs/LEGACY_DYNAMO_IMPORT.md b/delphi/docs/LEGACY_DYNAMO_IMPORT.md new file mode 100644 index 0000000000..6e160e7789 --- /dev/null +++ b/delphi/docs/LEGACY_DYNAMO_IMPORT.md @@ -0,0 +1,60 @@ +# Import a local DynamoDB export + +`scripts/import_dynamo_export.py` moves a bounded, checksummed export into a +normal Postgres graph job. The worker verifies source bytes, codec version, +worker code, importer code and runtime; normal queue finalization stores its +immutable artifact and M28 result rows together. No historical job executes. + +The export command requires an explicit loopback endpoint and only uses dummy +local credentials. It never falls back to an AWS endpoint. Specify every family +to export with repeated `--family` arguments. The exporter reads all scan pages +and writes the frozen `delphi-storage-codec/1` files and `SHA256SUMS`. +Keep the source quiescent during export. Consistent scan pages do not provide +a transaction snapshot across a whole table or all families; checksums verify +the completed export but cannot detect changes between scan pages. + +```sh +python scripts/import_dynamo_export.py export-local /tmp/generated-export \ + --endpoint http://127.0.0.1:8000 \ + --family Delphi_CommentEmbeddings --family Delphi_NarrativeReports \ + --family Delphi_JobQueue --family Delphi_JobActiveGuard +python scripts/import_dynamo_export.py preview /tmp/generated-export \ + --zid "$DEMO_ZID" --report-id "$DEMO_REPORT" +python scripts/import_dynamo_export.py enqueue /tmp/generated-export \ + --zid "$DEMO_ZID" --report-id "$DEMO_REPORT" --env local-demo --scope import-demo +``` + +For enqueue, `QUEUE_DATABASE_URL` must use a queue executor login. +`DATABASE_URL` supplies read access to verify that every explicitly named +report belongs to the selected conversation. Preview is offline; it checks +explicit source bindings but cannot verify the live reports mapping. Neither +command remaps conversation IDs or report IDs. Mixed-conversation exports fail. + +The `delphi` worker claims `graph_narrative` and runs the declared +`legacy-dynamo-export/1` model. Repeating the same enqueue returns the same +graph. Source, code or runtime changes produce a new request; an active scope +still prevents overlapping admission. Admission does not publish the result. +After successful execution, publication uses the ordinary explicit +`pd_graph_publish` generation check, so a failed import cannot replace a report. +The approved core contract does not support replacing a failed graph with a +superseding branch; that capability is deferred to #1436. The importer exposes +ordinary admission, status verification and publication only. + +Every source row is counted as one of: + +- Imported result rows in the 18 M28 result families. +- Archived T15/T16 job and guard rows in `legacy_control_files`, with exact + codec bytes on the immutable artifact. They are never converted to current + jobs, active scopes, leases or provider submissions. +- Quarantined result rows containing NUL strings/keys, which JSONB cannot + represent. Their exact codec bytes and `postgres-jsonb-nul` reason remain in + `quarantine` on the immutable artifact. Binary zero bytes are valid base64 + codec data and are imported normally. + +Unknown families, duplicate/noncanonical keys, missing/changed/unlisted files, +wrong conversation/report bindings and oversized artifacts fail before enqueue. +The initial importer is bounded to 450,000 serialized output bytes and the +queue's existing input limit. It refuses larger exports without truncation; +large archives need a separate artifact transport. It processes result values +as tagged codec values, preserving decimal numbers, sets, binary values and +JSON stored as strings. It does not read, write or reinterpret stored votes. diff --git a/delphi/docs/MATH_REBUILD_QUEUE.md b/delphi/docs/MATH_REBUILD_QUEUE.md new file mode 100644 index 0000000000..80a551f971 --- /dev/null +++ b/delphi/docs/MATH_REBUILD_QUEUE.md @@ -0,0 +1,41 @@ +# Rebuild one conversation through the queue + +The capacity poller already admits oversized conversations as `math_rebuild` +jobs. The operator command admits a conversation of any size through the same +Postgres RPC and large worker. It reads conversation counts and timestamps, +estimates memory with the configured poller model, and submits a typed frame. + +Set `DATABASE_URL` to the conversation database, `MATH_CAPACITY_QUEUE_DSN` to +a restricted executor-member login, and `MATH_CAPACITY_QUEUE_ENV` to the worker's +queue namespace. Do not place database credentials in arguments or reports. + +From `delphi/`: + +```sh +PYTHONPATH=. python scripts/enqueue_math_rebuild.py \ + --zid "$DEMO_ZID" --staged-label demo-staged --target-label demo-math \ + --source-commit "$MATH_POLLER_SOURCE_COMMIT" --dry-run +``` + +Remove `--dry-run` to enqueue. Repeating admission while the job is active +returns the existing job. A poisoned scope or an active scope with conflicting +configuration returns exit 1 without creating a job; malformed configuration +or a failed database operation returns exit 2. The JSON outcome and job ID +identify the conflict for the operator. + +The daemon must have `POLIS_JOBS_WORKER_CLASS=large`, +`POLIS_JOBS_STAGES=math_rebuild`, the matching `QUEUE_ENV`, a conversation +`DATABASE_URL`, and a known `MATH_POLLER_MEMORY_LIMIT_MB` (or cgroup limit). +Its `MATH_POLLER_SOURCE_COMMIT` must equal the admission's commit. Do not set +the retired resident worker's `MATH_CAPACITY_CLASS=large`; the daemon starts +`math_poller.py --job` for each claimed job. The child rechecks capacity, +obtains the existing single-writer lock, and rebuilds the one conversation. + +Results land in `math_main`, `math_bidtopid`, and `math_ptptstats` under the +staged label. The child does not publish the target label. The existing +capacity promotion loop requires the queue's successful finalization and a +manifest matching the staged fingerprint before promotion. An operator +admission alone does not register a conversation in a resident capacity router; +use the staged label for local inspection, or let ordinary capacity routing +track and promote its own admission. `prod`, `python`, and identical staged and +target labels are refused for staged writes. diff --git a/delphi/polismath/database/dynamodb.py b/delphi/polismath/database/dynamodb.py index 8ce59e6c4f..21bc3b427c 100644 --- a/delphi/polismath/database/dynamodb.py +++ b/delphi/polismath/database/dynamodb.py @@ -7,6 +7,7 @@ """ import boto3 +from polismath.delphi_storage.resource import result_resource import time import os import logging @@ -49,10 +50,10 @@ def __init__(self, def initialize(self) -> None: """Initialize DynamoDB connection and create tables if needed.""" # Set up environment variables for credentials if not provided and not already set - if not self.aws_access_key_id and not os.environ.get('AWS_ACCESS_KEY_ID'): + if os.environ.get('DELPHI_RESULT_BACKEND') != 'postgres' and not self.aws_access_key_id and not os.environ.get('AWS_ACCESS_KEY_ID'): os.environ['AWS_ACCESS_KEY_ID'] = 'dummy' - if not self.aws_secret_access_key and not os.environ.get('AWS_SECRET_ACCESS_KEY'): + if os.environ.get('DELPHI_RESULT_BACKEND') != 'postgres' and not self.aws_secret_access_key and not os.environ.get('AWS_SECRET_ACCESS_KEY'): os.environ['AWS_SECRET_ACCESS_KEY'] = 'dummy' # Create DynamoDB client @@ -67,7 +68,7 @@ def initialize(self) -> None: kwargs['aws_access_key_id'] = self.aws_access_key_id kwargs['aws_secret_access_key'] = self.aws_secret_access_key - self.dynamodb = boto3.resource('dynamodb', **kwargs) + self.dynamodb = result_resource('dynamodb', **kwargs) # Create tables if they don't exist self._ensure_tables_exist() diff --git a/delphi/polismath/delphi_storage/legacy_import.py b/delphi/polismath/delphi_storage/legacy_import.py new file mode 100644 index 0000000000..151066b125 --- /dev/null +++ b/delphi/polismath/delphi_storage/legacy_import.py @@ -0,0 +1,239 @@ +"""Bounded, lossless codec/1 export import through normal fenced graph execution. + +This never resumes historical jobs or guards. Their exact exports remain an +immutable archive in the import artifact. JSONB-incompatible NUL rows remain +in a separate quarantine with their original canonical bytes and a reason. +""" +from __future__ import annotations + +import hashlib +import json +import os +import stat +from pathlib import Path +import sys +from urllib.parse import urlsplit + +from .codec import FAMILIES, decode_family, encode_family, encode_item, header + +MODEL = "legacy-dynamo-export/1" +CONTROL_FAMILIES = frozenset({"Delphi_JobQueue", "Delphi_JobActiveGuard"}) +MAX_BYTES = 450_000 # Leave room inside the graph's 512 KiB output envelope. +MAX_MANIFEST_BYTES = 16_384 # Twenty known family names and SHA-256 digests. + + +def digest(data): + return hashlib.sha256(data).hexdigest() + + +def code_digest(): + return digest(Path(__file__).read_bytes()) + + +def codec_digest(): + return digest(Path(__file__).with_name("codec.py").read_bytes()) + + +def json_bytes(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), + ensure_ascii=True, allow_nan=False).encode("utf-8") + + +def inventory(files): + return {family: digest(wire.encode("utf-8")) for family, wire in sorted(files.items())} + + +def read_regular_file(path, limit): + """Do not follow symlinks, block on FIFOs, or trust a pre-read size alone.""" + flags = os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK + with os.fdopen(os.open(path, flags), "rb") as source: + metadata = os.fstat(source.fileno()) + if not stat.S_ISREG(metadata.st_mode): + raise ValueError("export input must be a regular file") + if metadata.st_size > limit: + raise ValueError("export file exceeds byte limit") + raw = source.read(limit + 1) + if len(raw) > limit: + raise ValueError("export file exceeds byte limit") + return raw + + +def read_export(directory): + """Read an explicit manifest; reject missing, changed, extra or foreign files.""" + directory = Path(directory) + expected = {} + for line in read_regular_file(directory / "SHA256SUMS", MAX_MANIFEST_BYTES).decode("utf-8").splitlines(): + sha, filename = line.split(" ", 1) + family = filename.removesuffix(".jsonl") + if (filename != family + ".jsonl" or family not in FAMILIES + or family in expected or len(sha) != 64 + or any(c not in "0123456789abcdef" for c in sha)): + raise ValueError("invalid export manifest") + expected[family] = sha + actual = {p.name for p in directory.glob("*.jsonl")} + if not expected or actual != {f + ".jsonl" for f in expected}: + raise ValueError("export inventory mismatch") + files = {} + remaining = MAX_BYTES + for family, sha in sorted(expected.items()): + raw = read_regular_file(directory / (family + ".jsonl"), remaining) + remaining -= len(raw) + if digest(raw) != sha or decode_family(raw)[0] != family: + raise ValueError("export digest or family mismatch") + files[family] = raw.decode("utf-8") + return files + + +def export_local(client, directory, families): + """Canonicalize every page from an explicitly local DynamoDB client.""" + directory = Path(directory) + if directory.exists() and any(directory.iterdir()): + raise ValueError("export destination must be empty") + directory.mkdir(parents=True, exist_ok=True) + files = {} + total_bytes = 0 + for family in sorted(set(families)): + if family not in FAMILIES: + raise ValueError("unknown export family") + total_bytes += len(header(family).encode("utf-8")) + 1 + if total_bytes > MAX_BYTES: + raise ValueError("export exceeds bounded inline import") + items, start = [], None + while True: + arguments = dict(TableName=family, ConsistentRead=True) + if start is not None: + arguments["ExclusiveStartKey"] = start + reply = client.scan(**arguments) + for item in reply.get("Items", []): + # Canonical sorting cannot change row byte sizes. Check each + # row before retaining it or fetching another scan page. + total_bytes += len(encode_item(family, item).encode("utf-8")) + 1 + if total_bytes > MAX_BYTES: + raise ValueError("export exceeds bounded inline import") + items.append(item) + start = reply.get("LastEvaluatedKey") + if not start: + break + raw = encode_family(family, items) + files[family] = raw.decode("utf-8") + (directory / (family + ".jsonl")).write_bytes(raw) + if not files: + raise ValueError("no families selected") + (directory / "SHA256SUMS").write_text("".join( + f"{sha} {family}.jsonl\n" for family, sha in inventory(files).items())) + return inventory(files) + + +def local_client(endpoint): + parsed = urlsplit(endpoint) + if (parsed.scheme != "http" or parsed.hostname not in {"127.0.0.1", "::1", "localhost"} + or parsed.username or parsed.password or parsed.path not in {"", "/"} + or parsed.query or parsed.fragment): + raise ValueError("export requires an explicit loopback DynamoDB endpoint") + import boto3 + from botocore.config import Config + return boto3.client("dynamodb", endpoint_url=endpoint, region_name="us-east-1", + aws_access_key_id="local", aws_secret_access_key="local", + config=Config(connect_timeout=5, read_timeout=30, retries={"max_attempts": 1})) + + +def has_nul(value): + if isinstance(value, str): + return "\0" in value + if isinstance(value, dict): + return any(has_nul(k) or has_nul(v) for k, v in value.items()) + if isinstance(value, (list, set, tuple)): + return any(has_nul(v) for v in value) + return False + + +def check_binding(item, zid, report_ids): + def scalar(name): + value = item[name] + return value.get("S", value.get("N")) + for name in ("conversation_id", "zid"): + if name in item and scalar(name) != str(zid): + raise ValueError("source conversation mismatch") + for name, separator in (("zid_tick", ":"), ("zid_tick_gid", ":"), ("zid_topic_jobid", "#")): + if name in item and (scalar(name) or "").split(separator, 1)[0] != str(zid): + raise ValueError("source composite conversation mismatch") + for name in ("report_id", "rid_section_model"): + if name in item and (scalar(name) or "").split("#", 1)[0] not in report_ids: + raise ValueError("source report requires explicit binding") + + +def validate_report_mapping(connection, zid, report_ids): + """Validate explicit report bindings against the local application database.""" + with connection.cursor() as cursor: + cursor.execute("SELECT report_id FROM reports WHERE zid=%s AND report_id=ANY(%s)", + (zid, list(report_ids))) + actual = {row[0] for row in cursor.fetchall()} + if actual != set(report_ids): + raise ValueError("source report is not attached to the requested conversation") + + +def import_output(files, zid, report_ids): + if type(zid) is not int or zid <= 0 or not files or set(files) - set(FAMILIES): + raise ValueError("invalid import identity or family set") + output = dict(model=MODEL, source_sha256=digest(json_bytes(inventory(files))), + source_inventory=inventory(files), legacy_control_files={}, quarantine={}, counts={}) + results = {} + for family, wire in sorted(files.items()): + actual, rows = decode_family(wire.encode("utf-8")) + if actual != family: + raise ValueError("source family mismatch") + good, bad = [], [] + for row in rows: + check_binding(row, zid, report_ids) + (bad if has_nul(row) else good).append(row) + if family in CONTROL_FAMILIES: + output["legacy_control_files"][family] = wire + output["counts"][family] = dict(source=len(rows), archived=len(rows), imported=0, quarantined=0) + else: + results[family] = encode_family(family, good).decode("utf-8") + if bad: + output["quarantine"][family] = dict(reason="postgres-jsonb-nul", + codec_wire=encode_family(family, bad).decode("utf-8"), row_count=len(bad)) + output["counts"][family] = dict(source=len(rows), archived=0, + imported=len(good), quarantined=len(bad)) + if results: + output["family_files"] = results + if len(json_bytes(output)) > MAX_BYTES: + raise ValueError("import artifact exceeds bounded inline import; no rows imported") + return output + + +def build_spec(files, zid, report_ids, worker_path=None): + """Import one checksummed export as an ordinary one-node graph.""" + output = import_output(files, zid, report_ids) + worker = Path(worker_path) if worker_path else Path(__file__).resolve().parents[2] / "scripts/job_graph_stage.py" + # This single-key ASCII object has exactly PostgreSQL jsonb's text spelling. + snapshot = {"texts": ["legacy import"]} + snapshot_sha = digest(json.dumps(snapshot).encode("utf-8")) + config = dict(family_files=files, source_sha256=output["source_sha256"], + report_ids=sorted(set(report_ids)), importer_sha256=code_digest(), + codec_sha256=codec_digest()) + spec = dict(schema="polis-job-graph/1", nodes=[dict( + key="legacy_import", stage="graph_narrative", **{"class": "delphi"}, + declared=dict(snapshot=dict(data=snapshot, sha256=snapshot_sha), + code=digest(worker.read_bytes()), model=MODEL, + runtime="python-" + sys.version.split()[0], seed=0, config=config, + mode="full", memory_bytes=64 * 1024 * 1024, work_units=1), + inputs=[], max_attempts=3)]) + if len(json_bytes(spec)) > 900_000: + raise ValueError("import admission exceeds graph input bound") + return spec + + +def execute_import(frame): + declared = frame["input"]["declared"] + config = declared["config"] + if (frame["stage"] != "graph_narrative" or declared["model"] != MODEL + or frame["input"]["artifacts"] + or config.get("importer_sha256") != code_digest() + or config.get("codec_sha256") != codec_digest()): + raise ValueError("legacy import provenance mismatch") + output = import_output(config["family_files"], frame["zid"], config["report_ids"]) + if config["source_sha256"] != output["source_sha256"]: + raise ValueError("legacy import source digest mismatch") + return output diff --git a/delphi/polismath/delphi_storage/postgres.py b/delphi/polismath/delphi_storage/postgres.py new file mode 100644 index 0000000000..d11d18b43f --- /dev/null +++ b/delphi/polismath/delphi_storage/postgres.py @@ -0,0 +1,83 @@ +"""Immutable run-bound PostgreSQL results using the frozen storage codec. + +The caller owns the connection and transaction. Methods never commit, allowing +an importer to stage, seal and finalize atomically. Normal graph children emit +``family_files`` with :func:`family_files`; the fenced artifact insert stores +those bytes in the same transaction as successful queue finalization. +""" +from __future__ import annotations + +from typing import Any, Iterable, Mapping + +from .codec import ( + CODEC_VERSION, FAMILIES, CodecError, decode_family, encode_family, + header, item_from_python, to_python, +) +import json + +RESULT_FAMILIES = frozenset(FAMILIES) - {"Delphi_JobQueue", "Delphi_JobActiveGuard"} + + +def family_files(families: Mapping[str, Iterable[dict[str, Any]]]) -> dict[str, str]: + """Encode native Python rows; floats must first follow the existing Decimal writer conversion.""" + result = {} + for family, items in families.items(): + if family not in RESULT_FAMILIES: + raise CodecError(f"not a result family: {family}") + result[family] = encode_family(family, (item_from_python(x) for x in items)).decode("utf-8") + return result + + +def decode_rows(family: str, rows: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Validate tagged JSONB values with codec/1 before returning Python types.""" + from .codec import dumps + _, decoded = decode_family((header(family) + "\n" + "".join(dumps(row) + "\n" for row in rows)).encode("utf-8")) + return [{name: to_python(value) for name, value in row.items()} for row in decoded] + + +class PostgresResultReader: + def __init__(self, connection: Any, env: str): + self.connection, self.env = connection, env + + def _call(self, name: str, casts: str, args: tuple[Any, ...]) -> Any: + # Names and casts are internal constants; all external values bind. + with self.connection.cursor() as cursor: + cursor.execute(f"SELECT public.{name}({casts})", (self.env, *args)) + value = cursor.fetchone()[0] + return json.loads(value) if isinstance(value, str) else value + + def read_artifact_family(self, artifact_id: str, family: str) -> list[dict[str, Any]]: + rows = self._call("pd_result_artifact_family", "%s::text,%s::uuid,%s::text", (artifact_id, family)) + return decode_rows(family, rows) + + def read_served_family(self, zid: int, scope: str, family: str) -> list[dict[str, Any]] | None: + rows = self._call("pd_result_served", "%s::text,%s::integer,%s::text,%s::text", (zid, scope, family)) + return None if rows is None else decode_rows(family, rows) + + def read_served_bundle(self, zid: int, scope: str) -> dict[str, Any] | None: + reply = self._call("pd_result_served_bundle", "%s::text,%s::integer,%s::text", (zid, scope)) + if reply is None: + return None + return {"generation": reply["generation"], "families": { + family: decode_rows(family, rows) for family, rows in reply["families"].items() + }} + + +class PostgresResultStore(PostgresResultReader): + def __init__(self, connection: Any, env: str, job_id: str, owner_id: str, + attempt_id: str, lease_epoch: int): + super().__init__(connection, env) + self.token = (job_id, owner_id, attempt_id, lease_epoch) + + def write_family(self, family: str, items: Iterable[dict[str, Any]]) -> dict[str, Any]: + wire = family_files({family: items})[family] + return self.write_family_wire(family, wire) + + def write_family_wire(self, family: str, wire: str) -> dict[str, Any]: + actual_family, _ = decode_family(wire.encode("utf-8")) + if family != actual_family or family not in RESULT_FAMILIES: + raise CodecError("result family mismatch") + return self._call("pd_result_put_family", "%s::text,%s::uuid,%s::uuid,%s::uuid,%s::bigint,%s::text,%s::text", (*self.token, family, wire)) + + def finish(self) -> dict[str, Any]: + return self._call("pd_result_seal", "%s::text,%s::uuid,%s::uuid,%s::uuid,%s::bigint", self.token) diff --git a/delphi/polismath/delphi_storage/resource.py b/delphi/polismath/delphi_storage/resource.py new file mode 100644 index 0000000000..b38e671479 --- /dev/null +++ b/delphi/polismath/delphi_storage/resource.py @@ -0,0 +1,202 @@ +"""Read-only compatibility surface for published PostgreSQL result families. + +Selection is explicit; a PostgreSQL error never opens a DynamoDB connection. +Legacy writers must move through graph finalization instead of mutating results. +""" +from __future__ import annotations +import os +import json +import re +from decimal import Decimal +from types import SimpleNamespace +from typing import Any +from .codec import FAMILIES, canonical_number, decode_family, dumps, item_from_python +from .postgres import RESULT_FAMILIES, decode_rows + + +def result_resource(service_name='dynamodb', **kwargs): + backend = os.environ.get('DELPHI_RESULT_BACKEND', 'dynamodb') + if backend not in ('dynamodb', 'postgres'): + raise ValueError('invalid DELPHI_RESULT_BACKEND') + if service_name != 'dynamodb' or backend == 'dynamodb': + import boto3 + return boto3.resource(service_name, **kwargs) + if os.environ.get('DELPHI_OUTPUT_MANIFEST'): + from .writer import WriterResource + return WriterResource() + return PostgresResource() + + +def _clauses(expression, names, values): + if expression is None: + return [] + if not isinstance(expression, str): + node=expression.get_expression() + op=node['operator']; args=node['values'] + if op=='AND': + return _clauses(args[0],names,values)+_clauses(args[1],names,values) + if op in ('=', 'begins_with'): + return [(op,args[0].name,args[1])] + raise ValueError(f'unsupported result condition {op}') + result=[] + for part in re.split(r'\s+AND\s+',expression,flags=re.I): + equal=re.fullmatch(r'\s*([#\w]+)\s*=\s*(:\w+)\s*',part) + prefix=re.fullmatch(r'\s*begins_with\(\s*([#\w]+)\s*,\s*(:\w+)\s*\)\s*',part) + match=equal or prefix + if match is None: + raise ValueError(f'unsupported result expression {part}') + result.append(('=' if equal else 'begins_with',names.get(match[1],match[1]),values[match[2]])) + return result + + +def _tag(value): + if isinstance(value,str):return 'S',value + if isinstance(value,bool):return 'BOOL',value + if isinstance(value,(int,Decimal)):return 'N',canonical_number(str(value)) + raise ValueError('result filter requires string, boolean or exact numeric value') + + +class PostgresResource: + def __init__(self, connection=None): + self.env=os.environ.get('DELPHI_RESULT_ENV') + if not self.env: + raise ValueError('DELPHI_RESULT_ENV is required for Postgres results') + self.connection=connection + self.meta=SimpleNamespace(client=self) + self.tables=SimpleNamespace(all=lambda: [self.Table(f) for f in sorted(RESULT_FAMILIES)]) + + def _connect(self): + if self.connection is None: + import psycopg2 + dsn=os.environ.get('DELPHI_RESULT_DATABASE_URL') or os.environ.get('DATABASE_URL') + if not dsn: + raise ValueError('DELPHI_RESULT_DATABASE_URL or DATABASE_URL is required') + self.connection=psycopg2.connect(dsn,application_name='delphi-pg-results/1') + self.connection.autocommit=True + return self.connection + + def Table(self,name): + if name not in RESULT_FAMILIES and name != 'Delphi_JobQueue': + raise ValueError(f'{name} is not a published result family; use the queue graph API') + return PostgresTable(self,name) + + def list_tables(self,**kwargs):return {'TableNames':sorted(RESULT_FAMILIES)} + def describe_table(self,TableName,**kwargs): + self.Table(TableName) + with self._connect().cursor() as cursor: + cursor.execute('SELECT 1 FROM public.delphi_result_current_rows LIMIT 0') + return {'Table':{'TableName':TableName,'TableStatus':'ACTIVE'}} + def batch_get_item(self,RequestItems,**kwargs): + response={} + for family,request in RequestItems.items(): + table=self.Table(family) + response[family]=[item for key in request['Keys'] if (item:=table.get_item(Key=key).get('Item')) is not None] + return {'Responses':response,'UnprocessedKeys':{}} + + +class PostgresTable: + def __init__(self,resource,name): + self.resource,self.name=resource,name + self.table_name=name + self.meta=resource.meta + + def load(self):return self.resource.describe_table(TableName=self.name) + def get_item(self,**kwargs): + rows=self._read(kwargs,True) + if not rows['Items']:return {} + item=rows['Items'][0] + if self.name=='Delphi_JobQueue' and not item.get('archived'): + with self.resource._connect().cursor() as cursor: + cursor.execute('SELECT public.pq_job_status(%s::text,%s::uuid)',[self.resource.env,item['job_id']]) + status=cursor.fetchall()[0][0] + attempt=(status or {}).get('attempt_id') + entries=[] + if attempt: + cursor.execute("""SELECT to_char(ts AT TIME ZONE 'UTC','YYYY-MM-DD"T"HH24:MI:SS.US"Z"'), + CASE stream WHEN 'stderr' THEN 'ERROR' ELSE 'INFO' END,line + FROM public.pq_attempt_logs(%s::text,%s::uuid,NULL,1000) + WHERE stream IN ('stdout','stderr')""",[self.resource.env,attempt]) + entries=[dict(timestamp=timestamp,level=level,message=line) for timestamp,level,line in cursor.fetchall()] + item={**item,'logs':json.dumps({'entries':entries}),'log_attempt_id':attempt} + return {'Item':item} + def query(self,**kwargs):return self._read(kwargs) + def scan(self,**kwargs):return self._read(kwargs) + def _read(self,params,get=False): + conditions=_clauses(params.get('KeyConditionExpression'),params.get('ExpressionAttributeNames',{}),params.get('ExpressionAttributeValues',{})) + conditions += [('=',k,v) for k,v in params.get('Key',{}).items()] + binds=[self.resource.env,self.name] + where=['env=%s','family=%s'] + scope=os.environ.get('DELPHI_RESULT_SCOPE') + if scope:where.append('scope_key=%s');binds.append(scope) + for op,key,value in conditions: + tag,wire=_tag(value) + if op=='begins_with': + if tag!='S':raise ValueError('begins_with requires a string') + where.append('starts_with(item->%s->>%s,%s)');binds.extend([key,tag,wire]) + else: + where.append('item->%s = %s::jsonb');binds.extend([key,dumps({tag:wire})]) + with self.resource._connect().cursor() as cursor: + if self.name == 'Delphi_JobQueue': + cursor.execute('SELECT row_to_json(j) FROM public.delphi_result_jobs j WHERE env=%s'+(' AND scope_key=%s' if scope else ''), [self.resource.env,scope] if scope else [self.resource.env]) + jobs=[row[0] for row in cursor.fetchall()] + active_ids={job['job_id'] for job in jobs} + archive_where='env=%s AND family=%s' + archive_binds=[self.resource.env,self.name] + if scope:archive_where+=' AND scope_key=%s';archive_binds.append(scope) + cursor.execute('SELECT zid,scope_key,generation,codec_wire FROM public.delphi_result_legacy_controls WHERE '+archive_where+' ORDER BY zid,scope_key',archive_binds) + records=[] + for zid,archive_scope,generation,wire in cursor.fetchall(): + family,_=decode_family(wire.encode('utf-8')) + archived=[json.loads(line) for line in wire.splitlines()[1:]] + if family!=self.name:raise ValueError('legacy control family mismatch') + for tagged in archived: + if tagged['job_id']['S'] not in active_ids: + records.append((zid,archive_scope,generation,{**tagged,'archived':{'BOOL':True}})) + records += [(job.get('conversation_id','queue'),'queue',0,item_from_python(job)) for job in jobs] + else: + cursor.execute('SELECT zid,scope_key,generation,item FROM public.delphi_result_current_rows WHERE '+' AND '.join(where)+' ORDER BY zid,scope_key,item_key::text',binds) + records=cursor.fetchall() + generations={dumps([str(zid),scope]):str(generation) for zid,scope,generation,_ in records} + start=params.get('ExclusiveStartKey') + if start and start.get('_polis_pg_generations')!=generations: + raise ValueError('result generation changed; restart pagination') + items=[];seen={} + for zid,scope,generation,tagged in records: + # decode_family validates sort order; one-row decoding retains every AV type. + item=decode_rows(self.name,[tagged])[0] + if not all(item.get(k)==v if op=='=' else isinstance(item.get(k),str) and item[k].startswith(v) for op,k,v in conditions):continue + key=dumps([item_from_python(item)[name] for name,_ in FAMILIES[self.name]['key']]) + if key in seen: + if seen[key]!=item:raise ValueError('ambiguous result scopes; set DELPHI_RESULT_SCOPE') + continue + seen[key]=item;items.append(item) + keys=[key for key,_ in FAMILIES[self.name]['key']] + index_order={'ConversationIndex':'created_at','StatusCreatedIndex':'created_at','ReportIdTimestampIndex':'timestamp','zid-created_at-index':'created_at'} + index=params.get('IndexName') + if index and index not in index_order:raise ValueError('unsupported result index') + ordering=([index_order[index]] if index else [])+keys + if index:items=[item for item in items if item.get(index_order[index]) is not None] + items.sort(key=lambda item:tuple(item[key] for key in ordering)) + if params.get('ScanIndexForward') is False:items.reverse() + if start: + startkey={k:v for k,v in start.items() if k!='_polis_pg_generations'} + position=next((i for i,item in enumerate(items) if all(item.get(k)==v for k,v in startkey.items())),None) + if position is None:raise ValueError('stale result cursor') + items=items[position+1:] + limit=params.get('Limit',1000) + if not isinstance(limit,int) or limit<1:raise ValueError('invalid result limit') + page=items[:limit] + filters=_clauses(params.get('FilterExpression'),params.get('ExpressionAttributeNames',{}),params.get('ExpressionAttributeValues',{})) + filtered=[item for item in page if all(item.get(k)==v if op=='=' else isinstance(item.get(k),str) and item[k].startswith(v) for op,k,v in filters)] + result={'Items':filtered,'Count':len(filtered),'ScannedCount':len(page)} + if len(items)>limit: + result['LastEvaluatedKey']={**{key:page[-1][key] for key in keys},'_polis_pg_generations':generations} + if params.get('ProjectionExpression'): + projected=[params.get('ExpressionAttributeNames',{}).get(k.strip(),k.strip()) for k in params['ProjectionExpression'].split(',')] + result['Items']=[{k:v for k,v in item.items() if k in projected} for item in filtered] + return result + + def __getattr__(self,name): + if name in ('put_item','update_item','delete_item','batch_writer','delete'): + raise RuntimeError('Published Delphi results are immutable; submit a new graph run') + raise AttributeError(name) diff --git a/delphi/polismath/delphi_storage/writer.py b/delphi/polismath/delphi_storage/writer.py new file mode 100644 index 0000000000..12afc5f720 --- /dev/null +++ b/delphi/polismath/delphi_storage/writer.py @@ -0,0 +1,200 @@ +"""Attempt-local compatibility writes, collected by the fenced daemon after exit. + +SQLite is a private scratchpad shared by pipeline subprocesses, never a serving +store. Its base is the immutable publication captured at admission. No database +credentials or Dynamo connection are needed here. Failed attempts cannot publish. +""" +from __future__ import annotations +from contextlib import contextmanager +import hashlib +import json +import os +from pathlib import Path +import re +import sqlite3 +from types import SimpleNamespace +from .codec import FAMILIES, dumps, encode_family, item_from_python, to_python, encode_item +from .postgres import RESULT_FAMILIES, decode_rows +from .resource import _clauses + + +class WriterResource: + def __init__(self): + manifest = Path(os.environ['DELPHI_OUTPUT_MANIFEST']) + frame = json.loads(Path(os.environ['DELPHI_FRAME']).read_text()) + base = frame.get('writer_base', {}) + if (not manifest.is_absolute() or base.get('schema') != 'delphi-writer-base/1' + or frame['job_id'] != os.environ['DELPHI_JOB_ID'] + or frame['attempt_id'] != os.environ['DELPHI_ATTEMPT_ID'] + or frame['run_id'] != os.environ['DELPHI_RUN_ID'] + or frame['config'].get('result_backend') != 'postgres' + or base['zid'] != frame['zid']): + raise ValueError('Postgres writes require a bound queue attempt') + self.directory = manifest.parent + self.frame = frame + self.connection = sqlite3.connect(self.directory/'writer.sqlite', timeout=30) + self.connection.execute('PRAGMA busy_timeout=30000') + self.connection.executescript('''CREATE TABLE IF NOT EXISTS binding(value TEXT PRIMARY KEY); + CREATE TABLE IF NOT EXISTS items(family TEXT,key TEXT,item TEXT,PRIMARY KEY(family,key)); + CREATE TABLE IF NOT EXISTS dirty(family TEXT PRIMARY KEY);''') + binding = dumps({k:frame[k] for k in ('env','job_id','run_id','attempt_id','lease_epoch')}) + with self.connection: + self.connection.execute('BEGIN IMMEDIATE') + row = self.connection.execute('SELECT value FROM binding').fetchone() + if row is None: + self.connection.execute('INSERT INTO binding VALUES(?)',(binding,)) + for family, rows in base['families'].items(): + if family not in RESULT_FAMILIES: + raise ValueError('unsupported base family') + for item in rows: + self.connection.execute('INSERT INTO items VALUES(?,?,?)', + (family,self.key(family,item),dumps(item))) + elif row[0] != binding: + raise ValueError('attempt scratch binding mismatch') + self.meta = SimpleNamespace(client=self) + self.tables = SimpleNamespace(all=lambda:[self.Table(f) for f in sorted(RESULT_FAMILIES)]) + + @staticmethod + def key(family, item): + return dumps([item[name] for name,_ in FAMILIES[family]['key']]) + + def Table(self, name): + if name not in RESULT_FAMILIES and name != 'Delphi_JobQueue': + raise ValueError('unsupported writer family') + return WriterTable(self,name) + + def list_tables(self, **kwargs): + return {'TableNames':sorted(RESULT_FAMILIES)} + + def describe_table(self, TableName, **kwargs): + self.Table(TableName) + return {'Table':{'TableName':TableName,'TableStatus':'ACTIVE'}} + + def batch_get_item(self, RequestItems, **kwargs): + return {'Responses':{f:[x for key in request['Keys'] if (x:=self.Table(f).get_item(Key=key).get('Item'))] + for f,request in RequestItems.items()},'UnprocessedKeys':{}} + + def reset(self): + with self.connection: + families = RESULT_FAMILIES - {'Delphi_NarrativeReports','report_narrative_store','Delphi_CollectiveStatement','Delphi_TopicAgendaSelections'} + self.connection.executemany('DELETE FROM items WHERE family=?',[(f,) for f in families]) + self.connection.executemany('INSERT OR IGNORE INTO dirty VALUES(?)',[(f,) for f in families]) + + def spool(self): + result = {} + total = 0 + # Emit the full snapshot, including empty families. No deleted item can + # reappear through the previous-artifact edge. Preserve untouched rows. + for family in sorted(RESULT_FAMILIES): + rows = [json.loads(row[0]) for row in self.connection.execute('SELECT item FROM items WHERE family=?',(family,))] + wire = encode_family(family,[item_from_python(row) for row in [row for tagged in rows for row in decode_rows(family,[tagged])]]) + total += len(wire) + if len(wire)>67108864 or total>268435456: + raise ValueError('result spool byte limit') + filename = family+'.jsonl' + path = self.directory/filename + with path.open('xb') as f: + f.write(wire) + result[family] = dict(file=filename,sha256=hashlib.sha256(wire).hexdigest()) + return result + + +class WriterTable: + def __init__(self, resource, name): + self.resource = resource + self.name = self.table_name = name + self.meta = resource.meta + self.table_status = 'ACTIVE' + + def load(self): + return self.resource.describe_table(TableName=self.name) + + def _writable(self, kwargs): + if self.name not in RESULT_FAMILIES: + raise RuntimeError('queue state belongs to polis-jobs') + if kwargs.get('ConditionExpression'): + raise ValueError('conditional result writes are not supported') + + def put_item(self, Item, **kwargs): + self._writable(kwargs) + tagged = item_from_python(Item) + # Validate all values and primary key before modifying the scratchpad. + encode_family(self.name,[tagged]) + zid = str(self.resource.frame['zid']) + for field in ('conversation_id','zid'): + if field in Item and str(Item[field]) != zid: + raise ValueError('writer conversation mismatch') + with self.resource.connection: + self.resource.connection.execute('INSERT OR REPLACE INTO items VALUES(?,?,?)', + (self.name,self.resource.key(self.name,tagged),encode_item(self.name,tagged))) + self.resource.connection.execute('INSERT OR IGNORE INTO dirty VALUES(?)',(self.name,)) + return {} + + def delete_item(self, Key, **kwargs): + self._writable(kwargs) + with self.resource.connection: + self.resource.connection.execute('DELETE FROM items WHERE family=? AND key=?', + (self.name,self.resource.key(self.name,item_from_python(Key)))) + self.resource.connection.execute('INSERT OR IGNORE INTO dirty VALUES(?)',(self.name,)) + return {} + + def update_item(self, Key, UpdateExpression, **kwargs): + self._writable(kwargs) + # Audited result producers use SET of whole attributes. Refuse every + # other grammar instead of silently approximating Dynamo semantics. + if not UpdateExpression.startswith('SET '): + raise ValueError('unsupported result update') + names = kwargs.get('ExpressionAttributeNames',{}) + values = kwargs.get('ExpressionAttributeValues',{}) + item = self.get_item(Key=Key).get('Item',dict(Key)) + for clause in UpdateExpression[4:].split(','): + match = re.fullmatch(r'\s*([#\w]+)\s*=\s*(:\w+)\s*',clause) + if match is None: + raise ValueError('unsupported result update') + item[names.get(match[1],match[1])] = values[match[2]] + self.put_item(Item=item) + return {'Attributes':item} if kwargs.get('ReturnValues')=='ALL_NEW' else {} + + @contextmanager + def batch_writer(self, **kwargs): + self._writable(kwargs) + yield self + + def get_item(self, Key, **kwargs): + items = self._read(dict(kwargs,Key=Key))['Items'] + return {'Item':items[0]} if items else {} + + def query(self, **kwargs):return self._read(kwargs) + def scan(self, **kwargs):return self._read(kwargs) + + def _read(self, params): + names,values = params.get('ExpressionAttributeNames',{}),params.get('ExpressionAttributeValues',{}) + conditions = _clauses(params.get('KeyConditionExpression'),names,values) + conditions += [('=',k,v) for k,v in params.get('Key',{}).items()] + matches = lambda item,clauses: all(item.get(k)==v if op=='=' else isinstance(item.get(k),str) and item[k].startswith(v) for op,k,v in clauses) + tagged = [json.loads(row[0]) for row in self.resource.connection.execute('SELECT item FROM items WHERE family=?',(self.name,))] + items = [row for item in tagged for row in decode_rows(self.name,[item])] + items = [item for item in items if matches(item,conditions)] + keys = [k for k,_ in FAMILIES[self.name]['key']] + indexes={'ConversationIndex':'created_at','ReportIdTimestampIndex':'timestamp','zid-created_at-index':'created_at'} + index=params.get('IndexName') + if index and index not in indexes:raise ValueError('unsupported result index') + ordering=([indexes[index]] if index else [])+keys + if index:items=[item for item in items if indexes[index] in item] + items.sort(key=lambda item:tuple(item[k] for k in ordering),reverse=params.get('ScanIndexForward') is False) + start=params.get('ExclusiveStartKey') + if start: + position=next((i for i,item in enumerate(items) if all(item.get(k)==v for k,v in start.items())),None) + if position is None:raise ValueError('stale writer cursor') + items=items[position+1:] + limit=params.get('Limit',1000) + if type(limit) is not int or limit<1:raise ValueError('invalid result limit') + page=items[:limit] + filtered=[item for item in page if matches(item,_clauses(params.get('FilterExpression'),names,values))] + result={'Items':filtered,'Count':len(filtered),'ScannedCount':len(page)} + if len(items)>limit:result['LastEvaluatedKey']={k:page[-1][k] for k in keys} + if params.get('ProjectionExpression'): + fields=[names.get(k.strip(),k.strip()) for k in params['ProjectionExpression'].split(',')] + result['Items']=[{k:v for k,v in item.items() if k in fields} for item in filtered] + if params.get('Select')=='COUNT':result.pop('Items') + return result diff --git a/delphi/polismath/job_child/__init__.py b/delphi/polismath/job_child/__init__.py index ecfc1f5a50..e8a0756d60 100644 --- a/delphi/polismath/job_child/__init__.py +++ b/delphi/polismath/job_child/__init__.py @@ -371,9 +371,11 @@ def _nullable(value, kind, what): def validate_manifest(m: Any) -> None: """The output-manifest/1 shape: closed key sets, typed values.""" - if not isinstance(m, dict) or set(m) != set(MANIFEST_KEYS): + writer = isinstance(m, dict) and m.get("schema") == "polis-jobs.output-manifest/2" + expected_keys = set(MANIFEST_KEYS) | ({"family_spool"} if "family_spool" in m else {"results"}) if writer else set(MANIFEST_KEYS) + if not isinstance(m, dict) or set(m) != expected_keys: raise ManifestError(f"manifest keys must be exactly {sorted(MANIFEST_KEYS)}") - if m["schema"] != MANIFEST_SCHEMA: + if m["schema"] != MANIFEST_SCHEMA and not writer: raise ManifestError("wrong manifest schema") for key in ("job_id", "attempt_id"): if not _is_uuid(m[key]): @@ -406,7 +408,7 @@ def validate_manifest(m: Any) -> None: expected = {"store", "family", "table", "rows", "key_prefix" if has_prefix else "keys"} if set(out) != expected: raise ManifestError(f"{where} keys must be exactly {sorted(expected)}") - if out["store"] != "dynamodb": + if out["store"] != ("postgres" if writer else "dynamodb"): raise ManifestError(f"{where}.store must be 'dynamodb' in output-manifest/1") if out["family"] not in DYNAMODB_FAMILIES: raise ManifestError(f"{where}.family {out['family']!r} is not a codec family") @@ -461,6 +463,12 @@ def encode_manifest(manifest: Dict[str, Any]) -> bytes: def write_manifest(ctx: JobContext, manifest: Dict[str, Any]) -> str: """Write the manifest atomically at DELPHI_OUTPUT_MANIFEST; return the sha256 of its bytes.""" + if os.environ.get("DELPHI_RESULT_BACKEND") == "postgres": + from polismath.delphi_storage.writer import WriterResource + resource = WriterResource() + manifest = dict(manifest, schema="polis-jobs.output-manifest/2", + outputs=[dict(o, store="postgres") for o in manifest["outputs"]], + family_spool=resource.spool() if manifest["outcome"] == "succeeded" else {}) data = encode_manifest(manifest) if os.path.lexists(ctx.manifest_path): raise ManifestError("the manifest already exists") diff --git a/delphi/polismath/job_child/census.py b/delphi/polismath/job_child/census.py index 6e9285bad8..c0324c3196 100644 --- a/delphi/polismath/job_child/census.py +++ b/delphi/polismath/job_child/census.py @@ -172,11 +172,11 @@ def default_pg_query() -> PgQuery: def default_dynamodb(region: Optional[str] = None): """The DynamoDB resource configured the way run_delphi.py's layer discovery does it.""" - import boto3 + from polismath.delphi_storage.resource import result_resource raw = os.environ.get("DYNAMODB_ENDPOINT") endpoint = raw if raw and raw.strip() else None if endpoint: - return boto3.resource("dynamodb", endpoint_url=endpoint, region_name="us-east-1", + return result_resource("dynamodb", endpoint_url=endpoint, region_name="us-east-1", aws_access_key_id="dummy", aws_secret_access_key="dummy") - return boto3.resource("dynamodb", region_name=region or os.environ.get("AWS_REGION", "us-east-1")) + return result_resource("dynamodb", region_name=region or os.environ.get("AWS_REGION", "us-east-1")) diff --git a/delphi/run_delphi.py b/delphi/run_delphi.py index e882f1c258..1981fb39e1 100644 --- a/delphi/run_delphi.py +++ b/delphi/run_delphi.py @@ -116,24 +116,31 @@ def main(): # validate_arg is not used in the python script execution steps, but kept for parity with bash # validate_arg = "--validate" if args.validate else "" - # --- Reset all data before processing --- - print(f"{YELLOW}Resetting all existing data for conversation {zid} before processing...{NC}") - reset_command = [ - "python", - "umap_narrative/reset_conversation.py", - f"--zid={zid}", - ] - # If a report ID is provided, pass it to the reset script for full cleanup - if rid: - reset_command.append(f"--rid={rid}") - print(f"{YELLOW}Using report ID {rid} for full narrative report cleanup.{NC}") + if os.environ.get("DELPHI_RESULT_BACKEND") == "postgres": + if job is None: + raise RuntimeError("Postgres pipelines must be admitted through polis-jobs") + from polismath.delphi_storage.writer import WriterResource + # Clear only this attempt's private working copy; previous publications survive. + WriterResource().reset() + else: + # --- Reset all data before processing --- + print(f"{YELLOW}Resetting all existing data for conversation {zid} before processing...{NC}") + reset_command = [ + "python", + "umap_narrative/reset_conversation.py", + f"--zid={zid}", + ] + # If a report ID is provided, pass it to the reset script for full cleanup + if rid: + reset_command.append(f"--rid={rid}") + print(f"{YELLOW}Using report ID {rid} for full narrative report cleanup.{NC}") - reset_process = subprocess.run(reset_command) - if reset_process.returncode != 0: - print(f"{RED}Data reset failed with exit code {reset_process.returncode}. Aborting pipeline.{NC}") - # Under the daemon the exit-code set is closed (0/1/2/4/5/6): any other failure is 1. - sys.exit(1 if job is not None else reset_process.returncode) - print(f"{GREEN}Data reset complete.{NC}") + reset_process = subprocess.run(reset_command) + if reset_process.returncode != 0: + print(f"{RED}Data reset failed with exit code {reset_process.returncode}. Aborting pipeline.{NC}") + # Under the daemon the exit-code set is closed (0/1/2/4/5/6): any other failure is 1. + sys.exit(1 if job is not None else reset_process.returncode) + print(f"{GREEN}Data reset complete.{NC}") print(f"{GREEN}Processing conversation {zid}...{NC}") @@ -263,6 +270,7 @@ def main(): # First, determine available layers from DynamoDB try: import boto3 + from polismath.delphi_storage.resource import result_resource from boto3.dynamodb.conditions import Key raw_endpoint = os.environ.get('DYNAMODB_ENDPOINT') @@ -270,13 +278,13 @@ def main(): # Using dummy credentials for local, IAM role for AWS if endpoint_url: - dynamodb = boto3.resource('dynamodb', + dynamodb = result_resource('dynamodb', endpoint_url=endpoint_url, region_name='us-east-1', aws_access_key_id='dummy', aws_secret_access_key='dummy') else: - dynamodb = boto3.resource('dynamodb', region_name=args.region) + dynamodb = result_resource('dynamodb', region_name=args.region) table = dynamodb.Table('Delphi_CommentHierarchicalClusterAssignments') diff --git a/delphi/scripts/delphi_cli.py b/delphi/scripts/delphi_cli.py index 2b100789b9..e703420142 100755 --- a/delphi/scripts/delphi_cli.py +++ b/delphi/scripts/delphi_cli.py @@ -9,6 +9,7 @@ import argparse import sys import boto3 +from polismath.delphi_storage.resource import result_resource import json import uuid import os @@ -61,7 +62,7 @@ def setup_dynamodb(endpoint_url=None, region='us-east-1'): os.environ.setdefault('AWS_ACCESS_KEY_ID', 'fakeMyKeyId') os.environ.setdefault('AWS_SECRET_ACCESS_KEY', 'fakeSecretAccessKey') - return boto3.resource('dynamodb', endpoint_url=endpoint_url, region_name=region) + return result_resource('dynamodb', endpoint_url=endpoint_url, region_name=region) def submit_job(dynamodb, zid, job_type='FULL_PIPELINE', priority=50, max_votes=None, batch_size=None, # For FULL_PIPELINE/PCA diff --git a/delphi/scripts/delphi_graph_stages.py b/delphi/scripts/delphi_graph_stages.py new file mode 100644 index 0000000000..012e3f4f3f --- /dev/null +++ b/delphi/scripts/delphi_graph_stages.py @@ -0,0 +1,238 @@ +"""Bounded numerical Delphi graph adapters; all durable writes are fenced by SQL. + +Narratives and topic names use explicitly labelled fixed stand-ins for plumbing proof. +Demo snapshots with 5–2000 texts are admitted by this initial adapter. +Raw votes never enter these adapters; narrative inputs contain signed aggregate +counts from the math snapshot. Comment ids are supplied explicitly. +""" +from decimal import Decimal +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path +import sys + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) +# Load the frozen stdlib-only codec without importing unrelated math poller modules. +import importlib.util +_spec = importlib.util.spec_from_file_location('delphi_graph_codec', ROOT / 'polismath/delphi_storage/codec.py') +_codec = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(_codec) +encode_family, decode_family = _codec.encode_family, _codec.decode_family +item_from_python, to_python = _codec.item_from_python, _codec.to_python + +MODELS = {'graph_embed': 'sentence-transformers/all-MiniLM-L6-v2', + 'graph_cluster': 'delphi-umap-evoc/1', 'graph_topics': 'delphi-tfidf-keywords/1', + 'graph_narrative': 'local-narrative-fixture/1'} + + +def code_digest(): + paths = [Path(__file__), ROOT / 'umap_narrative/numerical_stages.py', + ROOT / 'umap_narrative/polismath_commentgraph/utils/converter.py', + ROOT / 'umap_narrative/polismath_commentgraph/schemas/dynamo_models.py', + ROOT / 'umap_narrative/narrative_data.py'] + paths.extend(sorted((ROOT / 'umap_narrative/report_experimental').rglob('*.xml'))) + digest = hashlib.sha256() + for path in paths: + digest.update(path.read_bytes()) + return digest.hexdigest() + + +def model_digest(path): + """Pin actual local model bytes, independent of machine-specific cache paths.""" + path = Path(path) + files = sorted(p for p in path.rglob('*') if p.is_file() and '.cache' not in p.parts) + if not files or not any(p.name.endswith(('.safetensors', '.bin')) for p in files): + raise ValueError('local embedding model weights required') + h = hashlib.sha256() + for p in files: + h.update(p.relative_to(path).as_posix().encode() + b'\0') + with p.open('rb') as f: + for chunk in iter(lambda: f.read(1024 * 1024), b''): + h.update(chunk) + return h.hexdigest() + + +def _decimal(value): + if isinstance(value, float): + return Decimal(str(value)) + if isinstance(value, dict): + return {k: _decimal(v) for k, v in value.items()} + if isinstance(value, list): + return [_decimal(v) for v in value] + return value + + +def family_files(families): + return {name: encode_family(name, [item_from_python(_decimal(row)) for row in rows]).decode() + for name, rows in families.items()} + + +def read_family(output, name): + family, rows = decode_family(output['family_files'][name].encode()) + if family != name: + raise ValueError('family identity mismatch') + return [{k: to_python(v) for k, v in row.items()} for row in rows] + + +def upstream(frame, role): + artifact = frame['input']['artifacts'][role] + if hashlib.sha256(artifact['payload'].encode()).hexdigest() != artifact['sha256']: + raise ValueError('upstream artifact digest mismatch') + value = json.loads(artifact['payload']) + hydrated = frame.get('result_families', {}).get(role) + if hydrated is not None: + value['family_files'] = hydrated + return value + + + +def narrative_sections(topics, context, ids, texts): + """Use existing topic/global filters, selection limits, XML and report prompts.""" + import asyncio + import xmltodict + import xml.etree.ElementTree as ET + from umap_narrative.narrative_data import NarrativeSelection + if context.get('schema') != 'delphi-narrative-context/1': + raise ValueError('invalid narrative context') + text_by_id = dict(zip(ids,texts)) + records = [{**row,'comment':text_by_id[row['comment_id']]} for row in context['comments'] if row['comment_id'] in text_by_id] + for topic in topics: + members=set(topic['comment_ids']) + for row in records: + if row['comment_id'] in members: + row[f"layer{topic['layer_id']}_cluster_id"]=topic['cluster_id'] + selector = NarrativeSelection() + entries=[(topic,dict(topic_cluster_id=topic['cluster_id'],topic_layer_id=topic['layer_id'], + topic_citations=topic['comment_ids']), 'topics') for topic in topics] + for name,filter_type,threshold in [('groups','comment_extremity',1.0), + ('group_informed_consensus','group_aware_consensus','dynamic'),('uncertainty','uncertainty_ratio',0.2)]: + entries.append((dict(topic_label=name,size=len(records),_global=name), + dict(filter_type=filter_type,filter_threshold=threshold),name)) + directory=ROOT/'umap_narrative/report_experimental' + system=(directory/'system.xml').read_text() + result=[] + for topic,args,template in entries: + selected=[row for row in records if selector.filter_topics(row,**args)] + if not selected: + continue + structured=asyncio.run(selector.get_comments_as_xml(dict(processed_comments=records),selector.filter_topics,args)) + document=xmltodict.parse((directory/'subtaskPrompts'/f'{template}.xml').read_text()) + document['polisAnalysisPrompt']['data']={'content':{'structured_comments':structured}} + topic=dict(topic,_prompt=xmltodict.unparse(document,pretty=True),_system=system, + _allowed_ids=[int(row.attrib['id']) for row in ET.fromstring(structured).findall('comment')]) + result.append(topic) + return result + +def execute(frame): + d = frame['input']['declared'] + stage = frame['stage'] + if MODELS.get(stage) != d['model']: + raise ValueError('stage model mismatch') + texts = d['snapshot']['data']['texts'] + config = d['config'] + ids = config['comment_ids'] + if not 5 <= len(texts) <= 2000 or any(not isinstance(t, str) or not t.strip() or len(t) > 4096 for t in texts): + raise ValueError('expected 5–2000 nonempty bounded texts') + if len(ids) != len(texts) or len(set(ids)) != len(ids) or any(type(i) is not int or i < 0 for i in ids): + raise ValueError('unique nonnegative comment ids required') + zid = str(frame['zid']) + families = {} + output = {'model': d['model'], 'statement_count': len(texts)} + if stage == 'graph_embed': + import numpy as np + from sentence_transformers import SentenceTransformer + model_path = os.environ['DELPHI_EMBED_MODEL_PATH'] + if model_digest(model_path) != config['model_sha256']: + raise ValueError('embedding model bytes differ from admission') + model = SentenceTransformer(model_path, device='cpu', local_files_only=True) + vectors = model.encode(texts, convert_to_numpy=True, batch_size=32, show_progress_bar=False) + if vectors.shape != (len(texts), 384) or not np.isfinite(vectors).all() or (np.linalg.norm(vectors, axis=1) == 0).any(): + raise ValueError('invalid MiniLM embedding output') + families['Delphi_CommentEmbeddings'] = [dict(conversation_id=zid, comment_id=i, + embedding=dict(vector=v.tolist(), dimensions=384, model=d['model'])) for i,v in zip(ids,vectors)] + elif stage == 'graph_cluster': + import numpy as np + source = read_family(upstream(frame, 'embeddings'), 'Delphi_CommentEmbeddings') + by_id = {int(row['comment_id']): row for row in source} + vectors = np.asarray([by_id[i]['embedding']['vector'] for i in ids], dtype=np.float32) + if str(ROOT / 'umap_narrative') not in sys.path: + sys.path.insert(0, str(ROOT / 'umap_narrative')) + from umap_narrative.numerical_stages import project_and_cluster, characterize_comment_clusters + from polismath_commentgraph.utils.converter import DataConverter + points, layers = project_and_cluster(vectors) + if not layers or any(len(layer) != len(texts) for layer in layers) or not np.isfinite(points).all(): + raise ValueError('invalid Delphi clustering output') + output.update(layers=[layer.tolist() for layer in layers], points=points.tolist()) + dump = lambda models: [model.model_dump(exclude_none=True) for model in models] + families['Delphi_CommentHierarchicalClusterAssignments'] = dump( + DataConverter.batch_convert_clusters(zid,layers,points,ids)) + families['Delphi_UMAPGraph'] = dump(DataConverter.batch_convert_umap_edges(zid,points,layers,comment_ids=ids)) + families['Delphi_UMAPConversationConfig'] = [DataConverter.create_conversation_meta( + zid,vectors,layers).model_dump(exclude_none=True)] + elif stage == 'graph_topics': + import numpy as np + if str(ROOT / 'umap_narrative') not in sys.path: + sys.path.insert(0, str(ROOT / 'umap_narrative')) + from umap_narrative.numerical_stages import project_and_cluster, characterize_comment_clusters + from polismath_commentgraph.utils.converter import DataConverter + clusters = upstream(frame, 'clusters') + layers, points = [np.asarray(layer) for layer in clusters['layers']], np.asarray(clusters['points']) + characteristics, names, features = {}, {}, [] + for layer_id, labels in enumerate(layers): + chars = {str(key):value for key,value in characterize_comment_clusters(labels,texts).items()} + characteristics[f'layer{layer_id}'] = chars + names[f'layer{layer_id}'] = {key:'Keywords: '+', '.join(value.get('top_words',[])[:3]) + for key,value in chars.items()} + features.extend(model.model_dump(exclude_none=True) for model in + DataConverter.batch_convert_cluster_characteristics(zid,chars,layer_id)) + topics = [model.model_dump(exclude_none=True) for model in + DataConverter.batch_convert_topics(zid,layers,points,texts,names,characteristics)] + for topic in topics: + indexes = [i for i,label in enumerate(layers[topic['layer_id']]) if label == topic['cluster_id']] + topic['sample_statements'] = [dict(id=ids[i],text=texts[i]) for i in indexes[:3]] + topic['comment_ids'] = [ids[i] for i in indexes] + families['Delphi_CommentClustersStructureKeywords'] = topics + families['Delphi_CommentClustersFeatures'] = features + output['topics'] = topics + else: + topics = upstream(frame, 'topics')['topics'] + if config.get('narrative_context'): + topics = narrative_sections(topics, config['narrative_context'], ids, texts) + report_id = config['report_id'] + job = frame['job_id'] + # A visible fixture exercises the complete queue/read/render contract only. + names, reports = [], [] + completed_at = datetime.now(timezone.utc).isoformat() + for topic in topics: + label = topic.get('cluster_id',0) + layer = topic.get('layer_id',0) + title = 'Fixed proof topic' + if not topic.get('_global'): + names.append(dict(conversation_id=zid,topic_key=f'{job}#{layer}#{label}',layer_id=layer,cluster_id=label, + topic_name=title,model_name=d['model'],job_id=job,created_at=completed_at)) + section = f"{job}_global_{topic['_global']}" if topic.get('_global') else f'{job}_{layer}_{label}' + report = {'title':title,'paragraphs':[{'title':'Local provider fixture', + 'sentences':[{'clauses':[{'text':"Fixed narrative stand-in for queue, Postgres storage and report rendering proof. No LLM provider was called.",'citations':[]}]}]}], + 'provider_fixture':True} + reports.append(dict(report_id=report_id,section=section,model=d['model'], + rid_section_model=f'{report_id}#{section}#{d["model"]}',timestamp=completed_at, + job_id=job,report_data=json.dumps(report,sort_keys=True),metadata={'provider_fixture':True})) + completed_at = datetime.now(timezone.utc).isoformat() + for row in names: + row['created_at'] = completed_at + for row in reports: + row['timestamp'] = completed_at + families['Delphi_CommentClustersLLMTopicNames'] = names + families['Delphi_NarrativeReports'] = reports + if config.get('narrative_context'): + families['Delphi_CommentExtremity'] = [dict(conversation_id=zid,comment_id=str(row['comment_id']), + extremity_value=row['comment_extremity'],calculation_method='pca_based', + calculation_timestamp=completed_at,component_values={}) for row in config['narrative_context']['comments']] + output.update(provider_fixture=True,topics=len(topics), + text='Fixed stand-in; no LLM provider was called.') + output['family_files'] = family_files(families) + return output diff --git a/delphi/scripts/delphi_narrative_snapshot.py b/delphi/scripts/delphi_narrative_snapshot.py new file mode 100644 index 0000000000..2c5ebb1a77 --- /dev/null +++ b/delphi/scripts/delphi_narrative_snapshot.py @@ -0,0 +1,134 @@ +"""Read one consistent, sign-aware narrative summary from local Postgres. + +Uses the existing Delphi GroupDataProcessor and semantic vote loader. No votes +are written; no Dynamo client is initialized. Raw votes never enter job frames. +""" +import hashlib +import json +from pathlib import Path +import re +import sys + +ROOT = Path(__file__).resolve().parents[1] +for path in [ROOT, ROOT/'umap_narrative']: + if str(path) not in sys.path: + sys.path.insert(0,str(path)) + + + +def unfold_group_assignments(math): + """Expand the math contract's base-cluster IDs, never mistake them for PIDs.""" + base = math.get('base-clusters') + groups = math.get('group-clusters') + if (not isinstance(base, dict) or not isinstance(base.get('id'), list) + or not isinstance(base.get('members'), list) + or len(base['id']) != len(base['members']) or not isinstance(groups, list)): + raise ValueError('invalid folded group mapping shape') + + def identity(value): + if type(value) is not int or value < 0: + raise ValueError('invalid folded group mapping identity') + return value + + mapping = {} + participants = set() + for bid, members in zip(base['id'], base['members']): + identity(bid) + if bid in mapping or not isinstance(members, list): + raise ValueError('duplicate or invalid base cluster') + mapping[bid] = members + for pid in members: + identity(pid) + if pid in participants: + raise ValueError('duplicate participant in base clusters') + participants.add(pid) + assignments, unfolded, used_bids, seen_groups = {}, [], set(), set() + for group in groups: + if not isinstance(group, dict) or not isinstance(group.get('members'), list): + raise ValueError('invalid folded group') + gid = identity(group.get('id')) + if gid in seen_groups: + raise ValueError('duplicate group identity') + seen_groups.add(gid) + members = [] + for bid in group['members']: + identity(bid) + if bid not in mapping or bid in used_bids: + raise ValueError('missing or multiply assigned base cluster') + used_bids.add(bid) + members.extend(mapping[bid]) + for pid in members: + assignments[str(pid)] = gid + unfolded.append((gid, members)) + if not assignments: + raise ValueError('empty folded group assignments') + if 'group_clusters' in math: + alias = math['group_clusters'] + if not isinstance(alias, list) or len(alias) != len(unfolded): + raise ValueError('unfolded group alias mismatch') + for group, (gid, members) in zip(alias, unfolded): + if not isinstance(group, dict) or not isinstance(group.get('members'), list): + raise ValueError('invalid unfolded group alias') + identity(group.get('id')) + for pid in group['members']: + identity(pid) + if group['id'] != gid or group['members'] != members: + raise ValueError('unfolded group alias mismatch') + return assignments + + +def build_narrative_context(connection, zid, math_env): + from psycopg2.extras import RealDictCursor + from polismath_commentgraph.utils.storage import PostgresClient + from polismath_commentgraph.utils.group_data import GroupDataProcessor + from polismath.utils.vote_convention import RowConventionSource, database_row_fetcher, using_convention_source + + class BoundClient(PostgresClient): + def __init__(self): + pass + + def query(self, query, params=None): + query = re.sub(r'(? 524288: + raise ValueError('stage artifact exceeds 512 KiB') + return dict(schema='polis-job-artifact-manifest/1',job_id=frame['job_id'],run_id=frame['run_id'],attempt_id=frame['attempt_id'],stage=stage, + input_sha256=frame['input_sha256'],outcome='succeeded',output=dict(role='result',schema=stage+'/1',payload=payload,sha256=hashlib.sha256(payload.encode()).hexdigest())) + if declared['model'] in {'sentence-transformers/all-MiniLM-L6-v2', 'delphi-umap-evoc/1', 'delphi-tfidf-keywords/1', 'local-narrative-fixture/1'}: + from delphi_graph_stages import execute, code_digest + if declared['config'].get('adapter_sha256') != code_digest(): + raise ValueError('numerical adapter provenance mismatch') + output = execute(frame) + if os.environ.get('DELPHI_OUTPUT_MANIFEST'): + files = output.pop('family_files') + directory = Path(os.environ['DELPHI_OUTPUT_MANIFEST']).parent + output['family_spool'] = {} + for family, wire in files.items(): + filename = family + '.jsonl' + (directory / filename).write_text(wire, encoding='utf-8') + output['family_spool'][family] = dict(file=filename, sha256=hashlib.sha256(wire.encode()).hexdigest()) + payload = json.dumps(output, sort_keys=True, separators=(',', ':'), allow_nan=False) + if len(payload.encode()) > 524288: + raise ValueError('stage artifact exceeds 512 KiB') + return dict(schema='polis-job-artifact-manifest/1',job_id=frame['job_id'],run_id=frame['run_id'],attempt_id=frame['attempt_id'],stage=stage, + input_sha256=frame['input_sha256'],outcome='succeeded',output=dict(role='result',schema=stage+'/1',payload=payload,sha256=hashlib.sha256(payload.encode()).hexdigest())) for artifact in inp['artifacts'].values(): if hashlib.sha256(artifact['payload'].encode()).hexdigest() != artifact['sha256']: raise ValueError('artifact content mismatch') diff --git a/delphi/scripts/job_poller.py b/delphi/scripts/job_poller.py index 9f811db6f7..37c26cf4ec 100755 --- a/delphi/scripts/job_poller.py +++ b/delphi/scripts/job_poller.py @@ -526,6 +526,8 @@ class JobProcessor: def __init__(self, endpoint_url=None, region='us-east-1'): """Initialize the job processor.""" + if os.environ.get('DELPHI_RESULT_BACKEND') == 'postgres': + raise RuntimeError('Postgres jobs run through polis-jobs, never the Dynamo poller') self.worker_id = str(uuid.uuid4()) raw_endpoint = endpoint_url or os.environ.get('DYNAMODB_ENDPOINT') self.endpoint_url = raw_endpoint if raw_endpoint and raw_endpoint.strip() else None @@ -1435,6 +1437,9 @@ def poll_and_process(processor: JobProcessor, interval: int = 10): def main(): + if os.environ.get('DELPHI_RESULT_BACKEND') == 'postgres': + os.environ['POLIS_JOBS_ENABLED'] = '1' + os.execvp('polis-jobs', ['polis-jobs']) # This function is correct. parser = argparse.ArgumentParser(description='Delphi Job Poller Service') parser.add_argument('--endpoint-url', type=str, default=None) diff --git a/delphi/tests/dynamo_removal/Dockerfile b/delphi/tests/dynamo_removal/Dockerfile new file mode 100644 index 0000000000..7465206146 --- /dev/null +++ b/delphi/tests/dynamo_removal/Dockerfile @@ -0,0 +1,15 @@ +# CI builds DELPHI_BASE from delphi/Dockerfile. Build boxes may reuse an existing +# dependency image; all candidate Python sources and the freshly built daemon +# are copied below, so old installed application sources cannot win imports. +ARG DELPHI_BASE=polis-dynamo-deps:local +FROM ${DELPHI_BASE} +COPY delphi/ /app/ +COPY --from=queue-binary polis-jobs /usr/local/bin/polis-jobs +RUN python -m pip install --no-cache-dir pytest==8.3.5 pytest-cov==6.0.0 +ENV PYTHONPATH=/app:/app/scripts PYTHONDONTWRITEBYTECODE=1 +ENV DELPHI_EMBED_MODEL_PATH=/opt/polis/embedding/all-MiniLM-L6-v2 +RUN if [ ! -f "$DELPHI_EMBED_MODEL_PATH/config.json" ]; then \ + python -c 'from sentence_transformers import SentenceTransformer; import os; SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2").save(os.environ["DELPHI_EMBED_MODEL_PATH"])'; \ + fi +ENTRYPOINT [] +CMD ["polis-jobs"] diff --git a/delphi/tests/dynamo_removal/Dockerfile.dockerignore b/delphi/tests/dynamo_removal/Dockerfile.dockerignore new file mode 100644 index 0000000000..3f51585a06 --- /dev/null +++ b/delphi/tests/dynamo_removal/Dockerfile.dockerignore @@ -0,0 +1,6 @@ +.git +**/node_modules +**/target +**/.venv +**/.env +**/__pycache__ diff --git a/delphi/tests/dynamo_removal/admit_demo.py b/delphi/tests/dynamo_removal/admit_demo.py new file mode 100644 index 0000000000..e5998f4fcd --- /dev/null +++ b/delphi/tests/dynamo_removal/admit_demo.py @@ -0,0 +1,26 @@ +from contextlib import closing +"""Read the complete public demo statement snapshot and admit actual Delphi jobs.""" +import hashlib,json,os +from pathlib import Path +import psycopg2 +from job_graph_client import GraphClient,numerical_spec +from delphi_graph_stages import model_digest +from delphi_narrative_snapshot import build_narrative_context + +with psycopg2.connect(os.environ['DATABASE_URL']) as conn: + conn.set_session(isolation_level='REPEATABLE READ', readonly=True) + narrative_context=build_narrative_context(conn,1424,os.environ['MATH_ENV']) + with conn.cursor() as cur: + cur.execute('SELECT tid,txt FROM comments WHERE zid=%s ORDER BY tid',(1424,)) + rows=cur.fetchall() + data={'texts':[text for _,text in rows]} + cur.execute('SELECT public.pd_graph_hash(%s::jsonb)',(json.dumps(data),)) + sha=cur.fetchone()[0] +with closing(psycopg2.connect(os.environ['QUEUE_DATABASE_URL'])) as conn: + plan=numerical_spec(data['texts'],[tid for tid,_ in rows],'rlocaldynamo1424',sha, + model_digest(os.environ['DELPHI_EMBED_MODEL_PATH']), + narrative_context=narrative_context) + result=GraphClient(conn,'demo1424').admit(1424,'delphi','full-public-fixture-v1',plan) + result['statements']=len(rows) + Path('/proof/demo-admitted.json').write_text(json.dumps(result,indent=2)) + print(json.dumps(result)) diff --git a/delphi/tests/dynamo_removal/audit_narrative.py b/delphi/tests/dynamo_removal/audit_narrative.py new file mode 100644 index 0000000000..1ff871d9b9 --- /dev/null +++ b/delphi/tests/dynamo_removal/audit_narrative.py @@ -0,0 +1,128 @@ +"""Read-only independent recount of the admitted aggregate vote context.""" +import argparse +from collections import Counter +import hashlib +import json +import os +from pathlib import Path +import re +import sys + +ROOT = Path(__file__).resolve().parents[2] +sys.path[:0] = [str(ROOT), str(ROOT / "scripts")] +from polismath.utils.vote_convention import SEMANTIC_AGREE, SEMANTIC_DISAGREE, SEMANTIC_PASS + +SEMANTIC_SIGNS = (SEMANTIC_DISAGREE, SEMANTIC_PASS, SEMANTIC_AGREE) +COUNT_SIGNS = (("agrees", SEMANTIC_AGREE), ("disagrees", SEMANTIC_DISAGREE), ("passes", SEMANTIC_PASS)) + + +def digest(value, **kwargs): + return hashlib.sha256(json.dumps(value, **kwargs).encode()).hexdigest() + + +def canonical_groups(math): + bases = dict(zip(math["base-clusters"]["id"], math["base-clusters"]["members"])) + assert len(bases) == len(math["base-clusters"]["id"]) + groups = {g["id"]: [pid for bid in g["members"] for pid in bases[bid]] + for g in math["group-clusters"]} + assert len(groups) == len(math["group-clusters"]) + assert groups == {g["id"]: g["members"] for g in math["group_clusters"]} + assignments = {str(pid): gid for gid, members in groups.items() for pid in members} + assert len(assignments) == sum(map(len, groups.values())) + return groups, assignments + + +def audit_context(context, math, votes, statement_ids): + groups, assignments = canonical_groups(math) + assert context["group_mapping"] == "base-clusters-unfold/1" + assert context["source_convention"] == "semantic:+1=agree" + assert context["group_assignments_sha256"] == digest(assignments, sort_keys=True, separators=(",", ":")) + assert context["math_sha256"] == digest(math, sort_keys=True, separators=(",", ":"), allow_nan=False) + counts = Counter((v["tid"], assignments.get(str(v["pid"])), v["vote"]) + for v in votes if v["vote"] is not None) + overall = Counter() + for (tid, gid, sign), count in counts.items(): + assert sign in SEMANTIC_SIGNS + overall[tid, sign] += count + rows = context["comments"] + assert len(rows) == len(statement_ids) == len({row["comment_id"] for row in rows}) + assert {row["comment_id"] for row in rows} == set(statement_ids) + scopes = 0 + for row in rows: + tid = row["comment_id"] + assert row["comment-id"] == tid + for name, sign in COUNT_SIGNS: + assert row["total-" + name] == row[name] == overall[tid, sign], (tid, name) + assert row["total-votes"] == row["votes"] == sum(overall[tid, s] for s in SEMANTIC_SIGNS) + scopes += 1 + present = 0 + for gid in groups: + total = sum(counts[tid, gid, s] for s in SEMANTIC_SIGNS) + present += total > 0 + for name, sign in COUNT_SIGNS: + assert row.get(f"group-{gid}-{name}", 0) == counts[tid, gid, sign], (tid, gid, name) + assert row.get(f"group-{gid}-votes", 0) == total + scopes += 1 + assert row["num_groups"] == present + return dict(comments=len(rows), scoped_counts=scopes, count_fields=scopes * 4, + group_sizes={str(g): len(m) for g, m in groups.items()}, + latest_votes=len(votes), unassigned_votes=sum(n for (t, g, s), n in counts.items() if g is None)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--proof", type=Path, default=Path("/proof")) + parser.add_argument("--env", default="demo1424") + parser.add_argument("--zid", type=int, default=1424) + parser.add_argument("--scope", default="delphi") + parser.add_argument("--context-only", type=Path, help="diagnostic recount; does not certify published reports") + args = parser.parse_args() + import psycopg2 + from psycopg2.extras import RealDictCursor + from polismath.utils.vote_convention import RowConventionSource, database_row_fetcher, using_convention_source, load_semantic_votes + from polismath.delphi_storage.postgres import PostgresResultReader + from delphi_graph_stages import code_digest + with psycopg2.connect(os.environ["DATABASE_URL"]) as conn: + conn.set_session(isolation_level="REPEATABLE READ", readonly=True) + def query(sql, params=None): + with conn.cursor(cursor_factory=RealDictCursor) as cur: + cur.execute(re.sub(r"(? actual queued graph child. + +prepare seeds generated codec data and admits work. verify waits for the real +daemon and publishes via the public generation-checked RPC. This script never +claims or finalizes a queue job and never fabricates process exit proof. +""" +import argparse +from contextlib import closing +import json +import os +from pathlib import Path +import subprocess +import sys +import time +import uuid + +import psycopg2 + +DELPHI = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(DELPHI)) +sys.path.insert(0, str(DELPHI / "scripts")) +from job_graph_client import GraphClient +from polismath.delphi_storage import legacy_import as legacy +from polismath.delphi_storage.codec import decode_family, encode_family, item_from_python +from polismath.delphi_storage.golden_corpus import corpus +from polismath.delphi_storage.postgres import PostgresResultReader + +ZID, REPORT, ENV, SCOPE = 9001, "r9001generated", "proof-import", "generated-import" + + +def cli(*arguments): + completed = subprocess.run([sys.executable, str(DELPHI / "scripts/import_dynamo_export.py"), + *map(str, arguments)], capture_output=True, text=True) + if completed.returncode: + raise RuntimeError(completed.stderr.strip()) + return json.loads(completed.stdout) + + +def seed_database(): + with psycopg2.connect(os.environ["DATABASE_URL"]) as connection: + with connection.cursor() as cursor: + cursor.execute("INSERT INTO users(uid,hname,email) VALUES (%s,'Generated importer owner'," + "'importer@example.invalid') ON CONFLICT(uid) DO NOTHING", (ZID,)) + cursor.execute("INSERT INTO conversations(zid,owner,topic) VALUES (%s,%s,'Generated importer') " + "ON CONFLICT(zid) DO NOTHING", (ZID, ZID)) + cursor.execute("INSERT INTO reports(report_id,zid) SELECT %s,%s WHERE NOT EXISTS" + "(SELECT 1 FROM reports WHERE report_id=%s)", (REPORT, ZID, REPORT)) + legacy.validate_report_mapping(connection, ZID, [REPORT]) + + +def prepare(root, endpoint): + seed_database() + client = legacy.local_client(endpoint) + data = corpus() + # Deliberately incompatible result data proves durable quarantine, no votes. + data["Delphi_CommentEmbeddings"].append(item_from_python(dict( + conversation_id=str(ZID), comment_id=999, text="generated\0quarantine"))) + for family, rows in sorted(data.items()): + key = legacy.FAMILIES[family]["key"] + client.create_table(TableName=family, BillingMode="PAY_PER_REQUEST", + AttributeDefinitions=[dict(AttributeName=n, AttributeType=t) for n, t in key], + KeySchema=[dict(AttributeName=n, KeyType="HASH" if i == 0 else "RANGE") + for i, (n, _) in enumerate(key)]) + client.get_waiter("table_exists").wait(TableName=family) + for row in rows: + client.put_item(TableName=family, Item=row) + directory = root / ("export-" + uuid.uuid4().hex) + args = ["export-local", directory, "--endpoint", endpoint] + for family in sorted(data): + args.extend(["--family", family]) + exported = cli(*args) + assert set(exported) == set(legacy.FAMILIES) + source = legacy.read_export(directory) + for family in data: + assert source[family].encode() == encode_family(family, data[family]) + preview = cli("preview", directory, "--zid", ZID, "--report-id", REPORT) + arguments = ["enqueue", directory, "--zid", ZID, "--report-id", REPORT, + "--env", ENV, "--scope", SCOPE] + first, again = cli(*arguments), cli(*arguments) + assert first["outcome"] == "enqueued", first + assert again["outcome"] == "existing" and first["graph_id"] == again["graph_id"] + assert first["counts"] == preview["counts"] + receipt = dict(graph_id=first["graph_id"], root_job_id=first["root_job_id"], + export_dir=str(directory), source_sha256=first["source_sha256"], + counts=first["counts"], families=len(data), duplicate_outcome=again["outcome"]) + (root / "import-receipt.json").write_text(json.dumps(receipt, sort_keys=True, indent=2) + "\n") + print(json.dumps(dict(phase="prepared", **receipt), sort_keys=True)) + + +def verify(root, wait_seconds): + receipt = json.loads((root / "import-receipt.json").read_text()) + with closing(psycopg2.connect(os.environ["QUEUE_DATABASE_URL"])) as connection: + graph = GraphClient(connection, ENV) + deadline = time.monotonic() + wait_seconds + while True: + status = graph.status(receipt["graph_id"]) + assert len(status["nodes"]) == 1 + node = status["nodes"][0] + state = node["readiness"]["state"] + if state == "succeeded": + break + if state in {"dead", "cancelled"}: + raise AssertionError("import graph failed: " + state) + if time.monotonic() >= deadline: + raise AssertionError("real importer worker has not completed: " + state) + time.sleep(1) + artifact = node["artifact"] + payload = artifact["payload"] + assert artifact["content_sha"] == legacy.digest(payload.encode()) + output = json.loads(payload) + assert output["counts"] == receipt["counts"] + source = legacy.read_export(receipt["export_dir"]) + assert output == legacy.import_output(source, ZID, [REPORT]) + for family in legacy.CONTROL_FAMILIES: + assert output["legacy_control_files"][family] == source[family] + assert output["counts"]["Delphi_CommentEmbeddings"]["quarantined"] == 1 + assert decode_family(output["quarantine"]["Delphi_CommentEmbeddings"]["codec_wire"].encode())[1][0]["text"]["S"] == "generated\0quarantine" + served = graph.served(ZID, SCOPE) + if served is None: + publication = graph.publish(receipt["graph_id"], node["job_id"], 0) + assert publication["outcome"] == "published", publication + else: + assert served["bundle"]["root"] == artifact["artifact_id"] + reader = PostgresResultReader(connection, ENV) + bundle = reader.read_served_bundle(ZID, SCOPE) + assert set(bundle["families"]) == set(legacy.FAMILIES) - legacy.CONTROL_FAMILIES + for family, rows in bundle["families"].items(): + assert len(rows) == receipt["counts"][family]["imported"] + # Reencode native PG readback to prove tags, decimals, sets, and bytes. + assert encode_family(family, [item_from_python(row) for row in rows]).decode() == output["family_files"][family] + repeated = ["enqueue", receipt["export_dir"], "--zid", ZID, "--report-id", REPORT, + "--env", ENV, "--scope", SCOPE] + again = cli(*repeated) + assert again["outcome"] == "existing" and again["graph_id"] == receipt["graph_id"] + result = dict(phase="verified", graph_id=receipt["graph_id"], artifact_id=artifact["artifact_id"], + result_families=len(bundle["families"]), archived_control_families=len(output["legacy_control_files"]), + quarantined_rows=1, source_families=len(receipt["counts"]), + source_rows=sum(c["source"] for c in receipt["counts"].values()), + imported_rows=sum(c["imported"] for c in receipt["counts"].values()), + archived_rows=sum(c["archived"] for c in receipt["counts"].values()), + attempts=node["attempts"], generation=bundle["generation"], + duplicate_outcome=again["outcome"]) + (root / "import-verification.json").write_text(json.dumps(result, sort_keys=True, indent=2) + "\n") + print(json.dumps(result, sort_keys=True)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("phase", choices=("prepare", "verify")) + parser.add_argument("--directory", type=Path, required=True) + parser.add_argument("--endpoint", default="http://127.0.0.1:8000") + parser.add_argument("--wait-seconds", type=int, default=0) + args = parser.parse_args() + args.directory.mkdir(parents=True, exist_ok=True) + if args.phase == "prepare": + prepare(args.directory, args.endpoint) + else: + verify(args.directory, args.wait_seconds) + + +if __name__ == "__main__": + main() diff --git a/delphi/tests/dynamo_removal/math_proof.py b/delphi/tests/dynamo_removal/math_proof.py new file mode 100644 index 0000000000..04cc9bb420 --- /dev/null +++ b/delphi/tests/dynamo_removal/math_proof.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""Verify a real math job's receipt and promote its staged local demo bundle.""" +import argparse +from contextlib import closing +import json +import os +import time + +import psycopg2 + +from polismath.database.postgres import PostgresClient, PostgresConfig, staged_newer +from polismath.poller.capacity_queue import QueueClient, QueueSettings, decode_frame_uri +from polismath.poller.rebuild_child import check_child_label + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--job-id", required=True) + parser.add_argument("--zid", type=int, required=True) + parser.add_argument("--staged-label", required=True) + parser.add_argument("--target-label", required=True) + parser.add_argument("--wait-seconds", type=int, default=0) + args = parser.parse_args() + # This proof never promotes production labels. + check_child_label(args.target_label, served_env="prod", target_label=args.staged_label, env={}) + queue = QueueClient(QueueSettings(os.environ["MATH_CAPACITY_QUEUE_DSN"], + os.environ["MATH_CAPACITY_QUEUE_ENV"])) + deadline = time.monotonic() + args.wait_seconds + while True: + status = queue.job_status(args.job_id) + if status["state"] == "succeeded": + break + if status["state"] in {"dead", "cancelled"} or time.monotonic() >= deadline: + raise AssertionError("real math job not successful: " + status["state"]) + time.sleep(1) + frame = json.loads(decode_frame_uri(status["input"]["uri"])) + assert frame["zid"] == args.zid + assert frame["config"]["staged_label"] == args.staged_label + assert frame["config"]["target_label"] == args.target_label + receipt = queue.receipt(args.job_id) + assert receipt.finalized, "successful job has no valid stored manifest receipt" + database_url = os.environ["DATABASE_URL"] + pg = PostgresClient(PostgresConfig(url=database_url, math_env=args.target_label, ssl_mode="disable")) + pg.initialize() + try: + with closing(psycopg2.connect(database_url)) as lock: + lock.autocommit = True + with lock.cursor() as cursor: + cursor.execute("SELECT pg_try_advisory_lock(hashtext(%s))", + ("polis-math-python:" + args.target_label,)) + assert cursor.fetchone()[0], "another writer owns target label" + fps = pg.math_fingerprints([args.zid], [args.staged_label, args.target_label]) + staged = fps.get((args.zid, args.staged_label)) + target = fps.get((args.zid, args.target_label)) + assert staged and staged.complete and receipt.binds(staged, args.staged_label) + if staged_newer(staged, target): + pg.promote_bundle(args.zid, from_env=args.staged_label, to_env=args.target_label, + expected_target=target, expected_staged=staged) + promoted = pg.math_fingerprints([args.zid], [args.target_label])[(args.zid, args.target_label)] + assert promoted.complete and promoted.lvt == staged.lvt + print(json.dumps(dict(outcome="verified-and-promoted", job_id=args.job_id, + attempts=status["attempt_count"], receipt_sha256=receipt.output_sha256, + staged_math_tick=staged.math_tick, target_math_tick=promoted.math_tick, + vote_hwm=promoted.lvt, target_label=args.target_label), sort_keys=True)) + finally: + pg.shutdown() + + +if __name__ == "__main__": + main() diff --git a/delphi/tests/dynamo_removal/seed.py b/delphi/tests/dynamo_removal/seed.py new file mode 100644 index 0000000000..8cdf0251e3 --- /dev/null +++ b/delphi/tests/dynamo_removal/seed.py @@ -0,0 +1,51 @@ +"""Local public-fixture importer; explicit source convention, all votes via module.""" +import argparse +import csv +import json +import os +from pathlib import Path +import psycopg2 +from psycopg2.extras import execute_values, RealDictCursor +from polismath.utils.vote_convention import ( + semantic_vote, storage_vote, database_row_fetcher, RowConventionSource, +) + + +def main(): + parser=argparse.ArgumentParser() + parser.add_argument('--source-agree',required=True,type=int,choices=(-1,1)) + args=parser.parse_args() + root=Path(__file__).resolve().parents[2]/'real_data' + data=next(root.glob('*-biodiversity')) + comments=list(csv.DictReader(next(data.glob('*comments.csv')).open())) + votes=list(csv.DictReader(next(data.glob('*votes.csv')).open())) + pids=sorted({int(v['voter-id']) for v in votes}|{int(c['author-id']) for c in comments}) + zid=1424 + with psycopg2.connect(os.environ['DATABASE_URL']) as conn: + def query(sql): + with conn.cursor(cursor_factory=RealDictCursor) as cursor: + cursor.execute(sql);return cursor.fetchall() + convention=RowConventionSource(database_row_fetcher(query)).current() + with conn.cursor() as cur: + cur.execute("INSERT INTO users(uid,hname,email) VALUES (%s,'Public fixture owner','fixture@example.invalid')",(zid,)) + cur.execute("INSERT INTO conversations(zid,owner,topic,description,is_active,topics_enabled) VALUES (%s,%s,'Biodiversity public fixture','Local queue and Postgres proof',true,true)",(zid,zid)) + cur.execute("INSERT INTO reports(rid,zid,report_id) VALUES (%s,%s,'rlocaldynamo1424')",(zid,zid)) + cur.execute("INSERT INTO zinvites(zid,zinvite) VALUES(%s,'local-dynamo-1424')",(zid,)) + # Each participant receives a unique test account. + execute_values(cur,'INSERT INTO users(uid,hname,email) VALUES %s',[(100000+p,'Fixture participant',f'fixture{p}@example.invalid') for p in pids]) + execute_values(cur,'INSERT INTO participants(zid,pid,uid,created,mod) VALUES %s',[(zid,p,100000+p,1700000000000,0) for p in pids]) + execute_values(cur,'INSERT INTO comments(zid,tid,pid,uid,txt,mod,is_meta,created,modified,active) VALUES %s',[ + (zid,int(x['comment-id']),int(x['author-id']),100000+int(x['author-id']),x['comment-body'],int(x['moderated']),False,int(x['timestamp'])*1000,int(x['timestamp'])*1000,True) for x in comments]) + converted=[(zid,int(x['voter-id']),int(x['comment-id']),storage_vote(semantic_vote(int(x['vote']),args.source_agree),convention.agree_value),int(x['timestamp'])*1000) for x in votes] + execute_values(cur,'INSERT INTO votes(zid,pid,tid,vote,created) VALUES %s',converted,page_size=1) + cur.execute('UPDATE participants p SET vote_count=v.n FROM (SELECT pid,count(*) n FROM votes WHERE zid=%s GROUP BY pid) v WHERE p.zid=%s AND p.pid=v.pid',(zid,zid)) + cur.execute('UPDATE conversations SET participant_count=%s WHERE zid=%s',(len(pids),zid)) + cur.execute('SELECT tid,vote,count(*) FROM votes WHERE zid=%s GROUP BY tid,vote',(zid,)) + actual={(tid,semantic_vote(v,convention.agree_value)):n for tid,v,n in cur.fetchall()} + expected={} + for x in votes: + k=(int(x['comment-id']),semantic_vote(int(x['vote']),args.source_agree));expected[k]=expected.get(k,0)+1 + assert actual==expected,'vote polarity/count self-check failed' + print(json.dumps(dict(comments=len(comments),participants=len(pids),votes=len(votes),vote_count_groups_checked=len(expected),source_agree=args.source_agree,storage_agree=convention.agree_value))) + +if __name__=='__main__':main() diff --git a/delphi/tests/dynamo_removal/wait_dynamo.py b/delphi/tests/dynamo_removal/wait_dynamo.py new file mode 100644 index 0000000000..12b7d475b0 --- /dev/null +++ b/delphi/tests/dynamo_removal/wait_dynamo.py @@ -0,0 +1,15 @@ +"""Wait for the owned local DynamoDB service, never a default AWS endpoint.""" +import time +from polismath.delphi_storage.legacy_import import local_client +client=local_client('http://127.0.0.1:8000') +last=None +for _ in range(60): + try: + client.list_tables() + print('Local DynamoDB ready') + break + except Exception as exc: + last=exc + time.sleep(1) +else: + raise RuntimeError('Owned DynamoDB did not become ready') from last diff --git a/delphi/tests/dynamo_removal/wait_publish.py b/delphi/tests/dynamo_removal/wait_publish.py new file mode 100644 index 0000000000..15767e4681 --- /dev/null +++ b/delphi/tests/dynamo_removal/wait_publish.py @@ -0,0 +1,54 @@ +from contextlib import closing +"""Observe actual jobs, assert independent retry, then publish one coherent bundle.""" +import json,os,time +from pathlib import Path +import psycopg2 +from job_graph_client import GraphClient +from polismath.delphi_storage.postgres import PostgresResultReader +root=Path('/proof'); admitted=json.loads((root/'demo-admitted.json').read_text()) +with closing(psycopg2.connect(os.environ['QUEUE_DATABASE_URL'])) as conn: + client=GraphClient(conn,'demo1424') + # Bound the proof without changing worker leases. + deadline=time.monotonic()+1800 + prior=None + while time.monotonic()0 + assert counts['Delphi_NarrativeReports']>0 + # Bind coverage to the computed topic structure, so omitting a name and its + # report together cannot make a partial narrative publication pass. + job=nodes['n']['job_id'] + structure=bundle['families']['Delphi_CommentClustersStructureKeywords'] + expected_topics={f"{job}#{int(t['layer_id'])}#{int(t['cluster_id'])}" for t in structure} + assert len(expected_topics)==len(structure)>0, 'duplicate computed topic' + names=bundle['families']['Delphi_CommentClustersLLMTopicNames'] + assert len(names)==len(expected_topics) and {t['topic_key'] for t in names}==expected_topics, 'complete topic names required' + expected_sections={key.replace('#','_') for key in expected_topics} + expected_sections.update(f'{job}_global_{name}' for name in ('groups','group_informed_consensus','uncertainty')) + reports=bundle['families']['Delphi_NarrativeReports'] + assert len(reports)==len(expected_sections), 'duplicate or missing report section' + assert {row['section'] for row in reports}==expected_sections, 'complete topic and global reports required' + assert all(row['job_id']==job for row in reports), 'reports must belong to published narrative job' + assert all(row['metadata']['provider_fixture'] is True and json.loads(row['report_data'])['provider_fixture'] is True for row in reports) + assert all(row['model']=='local-narrative-fixture/1' for row in reports) +with psycopg2.connect(os.environ['DATABASE_URL']) as conn: + with conn.cursor() as cur: + cur.execute('SELECT stage,attempt_count FROM polis_queue_jobs WHERE env=%s AND job_id=ANY(%s::uuid[])',('demo1424',[n['job_id'] for n in nodes.values()])) + attempts=dict(cur.fetchall()) + assert attempts['graph_embed']==1 and attempts['graph_cluster']==2,attempts + assert attempts['graph_topics']==1 and attempts['graph_narrative']==1,attempts +receipt={'statement_count':316,'attempts':attempts,'published_generation':bundle['generation'],'family_rows':counts,'provider':'fixed stand-in; no LLM calls'} +(root/'demo-result.json').write_text(json.dumps(receipt,indent=2));print(json.dumps(receipt),flush=True) diff --git a/delphi/tests/dynamo_writers/child.py b/delphi/tests/dynamo_writers/child.py new file mode 100755 index 0000000000..f4c1ce8351 --- /dev/null +++ b/delphi/tests/dynamo_writers/child.py @@ -0,0 +1,42 @@ +#!/usr/local/bin/python +"""CI-only result producer: fixed data, actual resource and daemon protocol. + +Does not replace production scripts or call providers. Its executable path is +selected explicitly by test-dynamo-writers.sh inside an isolated worker. +""" +import os +from datetime import datetime,timedelta,timezone +from polismath import job_child +from polismath.delphi_storage.resource import result_resource +from polismath.delphi_storage.postgres import RESULT_FAMILIES +from polismath.delphi_storage.codec import FAMILIES + +job=job_child.JobContext.from_env(expected_stage=os.environ['DELPHI_STAGE'],zid=None, + allowed_phases={'run','submit','recheck'},default_phase='run') +assert 'QUEUE_DATABASE_URL' not in os.environ +resource=result_resource() +phase=job.phase +cost={'llm_tokens_in':None,'llm_tokens_out':None,'provider_batches':None} +outcome='succeeded';after=None +if phase=='run': + # Exercise every computational family with exact tagged data. Actual math + # and language generation remain covered by their independent suites. + for family in sorted(RESULT_FAMILIES-{'Delphi_NarrativeReports','report_narrative_store','Delphi_CollectiveStatement','Delphi_TopicAgendaSelections'}): + special={'zid':'1','conversation_id':'1','zid_tick':'1:1','zid_tick_gid':'1:1:0'} + for index in range(3 if family=='Delphi_CommentEmbeddings' else 1): + row={k:(index if tag=='N' else special.get(k,str(index))) for k,tag in FAMILIES[family]['key']} + row['generated_result_fixture']=True + resource.Table(family).put_item(Item=row) +elif phase=='submit': + job_child.request_provider_intent(job,provider='anthropic',model='fixed-provider-proof',batch={'request_count':1}) + cost['provider_batches']=[{'provider':'anthropic','batch_id':'generated-provider-batch','submitted_at':'2026-01-01T00:00:00Z'}] + outcome='parked';after=(datetime.now(timezone.utc)+timedelta(seconds=1)).isoformat() +else: + assert job.provider_batch_id=='generated-provider-batch' + resource.Table('Delphi_NarrativeReports').put_item(Item={ + 'rid_section_model':'rlocalwriter#section#fixed-provider-proof','timestamp':'2026-01-01T00:00:00Z', + 'report_id':'rlocalwriter','job_id':job.job_id,'report_data':'{"provider_fixture":true}'}) + cost['provider_batches']=[{'provider':'anthropic','batch_id':job.provider_batch_id,'submitted_at':'2026-01-01T00:00:00Z'}] +manifest=job_child.build_manifest(job,outcome=outcome,inputs=job_child.empty_inputs(),outputs=[], + models={'embed':None,'topic':None,'narrative':'fixed-provider-proof' if phase!='run' else None},cost=cost,recheck_after=after) +job_child.write_manifest(job,manifest) diff --git a/delphi/tests/dynamo_writers/prove.py b/delphi/tests/dynamo_writers/prove.py new file mode 100644 index 0000000000..c2713839fd --- /dev/null +++ b/delphi/tests/dynamo_writers/prove.py @@ -0,0 +1,137 @@ +"""Real PostgreSQL + daemon proof. Generated results; Dynamo is stopped by CI.""" +import hashlib +import json +import os +from pathlib import Path +import subprocess +import time +import uuid +import psycopg2 +from psycopg2.extras import Json +from polismath.delphi_storage.codec import item_from_python,decode_family,encode_item +from polismath.delphi_storage.postgres import family_files + +admin=psycopg2.connect(os.environ['DATABASE_URL']);admin.autocommit=True +executor=psycopg2.connect(os.environ['QUEUE_DATABASE_URL']);executor.autocommit=True +checks=[] +def sql(query,args=(),connection=admin): + with connection.cursor() as c: + c.execute(query,args) + return c.fetchall() if c.description else [] +def call(name,casts,args,connection=executor): + return sql('SELECT public.'+name+'('+','.join('%s::'+cast for cast in casts.split(','))+')',args,connection)[0][0] +def denied(query,args=(),connection=executor): + try:sql(query,args,connection) + except psycopg2.Error as e:return str(e) + raise AssertionError('unexpected SQL success') +def record(name):checks.append(name);print('PASS '+name,flush=True) +def rows(family):return sql("SELECT item FROM delphi_result_current_rows WHERE env='writer-proof' AND zid=1 AND scope_key='delphi' AND family=%s ORDER BY item_key::text",[family]) +def mutate(family,item,operation='put',request=None): + return call('pd_result_mutate','text,text,text,uuid,text,text,jsonb', + ['writer-proof','delphi','generated-http-owner',request or str(uuid.uuid4()),family,operation,Json(json.loads(encode_item(family,item_from_python(item))))]) +def admit(key=None,config=None,sha=None,stage='delphi_full_pipeline',zid=1): + config=config or {'include_moderation':True,'model':'fixed-provider-proof'} + return call('pd_writer_admit','text,integer,text,text,text,text,uuid,uuid,text,text,jsonb,text', + ['writer-proof',zid,'delphi','generated-api',key or str(uuid.uuid4()),sha or hashlib.sha256(json.dumps(config,sort_keys=True).encode()).hexdigest(), + str(uuid.uuid4()),str(uuid.uuid4()),stage,'rlocalwriter' if stage=='delphi_narrative' else None,Json(config),'d6f9ed6093e46b3c07db98f3c644bbbd9c4e87cd']) +def status(job):return call('pq_job_status','text,uuid',['writer-proof',job]) +def wait(job,state): + until=time.monotonic()+90 + while time.monotonic()=2 + assert sql("SELECT pd_result_artifact_family('writer-proof',%s,'Delphi_CommentEmbeddings')",[full_artifact])[0][0]==[r[0] for r in rows('Delphi_CommentEmbeddings')] + record('real daemon full-result spool and narrative submit/park/recheck, run-bound publication and prior artifacts preserved') + finally: + daemon.terminate() + try:daemon.wait(timeout=15) + except subprocess.TimeoutExpired:daemon.kill();daemon.wait() +assert sql('SELECT count(*) FROM votes')[0][0]==0 +assert sql('SELECT count(*) FROM delphi_artifacts a JOIN polis_queue_jobs q USING(env,job_id) WHERE q.state<>\'succeeded\'')[0][0]==0 +assert 'permission denied' in denied('UPDATE delphi_graph_served SET generation=99') +Path('/proof/completed.json').write_text(json.dumps({'checks':checks,'votes':0},indent=2)+'\n') +print('PASS all writer boundary scenarios; no Dynamo or paid provider requests',flush=True) diff --git a/delphi/tests/job_graph/results.py b/delphi/tests/job_graph/results.py new file mode 100644 index 0000000000..dd0c44e003 --- /dev/null +++ b/delphi/tests/job_graph/results.py @@ -0,0 +1,111 @@ +#!/usr/bin/env python3 +"""M28 real-Postgres negative controls, on the graph proof's local database.""" +import hashlib +import json +import sys +import uuid +from pathlib import Path +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) +from polismath.delphi_storage.postgres import family_files +from polismath.delphi_storage.codec import encode_family,item_from_python +from prove import graph, spec, nodes, rpc, sql, lit, js, record +from adversarial import claim, ident, finish + +FAMILY = 'Delphi_CommentEmbeddings' + + +def row(value='first'): + return dict(conversation_id='1', comment_id=0, text=value, embedding=[]) + + +def make_manifest(c, payload): + wire=json.dumps(payload, sort_keys=True, separators=(',', ':')) + return dict(schema='polis-job-artifact-manifest/1', job_id=c['job_id'],run_id=c['run_id'], + attempt_id=c['attempt_id'],stage=c['stage'],input_sha256=c['graph_input_sha'], + outcome='succeeded',output=dict(role='result',schema=c['stage']+'/1',payload=wire, + sha256=hashlib.sha256(wire.encode()).hexdigest())) + + +def main(): + sql("INSERT INTO users(uid,hname) VALUES(1,'generated result owner') ON CONFLICT DO NOTHING; INSERT INTO conversations(zid,owner,topic) VALUES(1,1,'generated result proof') ON CONFLICT DO NOTHING") + scope='result-proof-'+uuid.uuid4().hex[:8] + publish_scope='result-publish-'+uuid.uuid4().hex[:8] + single=spec();single['nodes']=single['nodes'][:1] + g=graph(scope,single) + c=claim();assert c['job_id']==nodes(g)['e']['job_id'] + wire=family_files({FAMILY:[row()]})[FAMILY] + put=lambda token, data: rpc('pd_result_put_family',*token,lit(FAMILY),lit(data)) + invalid = wire.replace('"text":{"S":"first"}', '"text":{"N":"01"}') + assert 'invalid result row' in sql('SELECT pd_result_put_family('+','.join([*ident(c),lit(FAMILY),lit(invalid)])+')',True,False) + wrong_zid = family_files({FAMILY:[{**row(),'conversation_id':'2'}]})[FAMILY] + assert 'conversation mismatch' in sql('SELECT pd_result_put_family('+','.join([*ident(c),lit(FAMILY),lit(wrong_zid)])+')',True,False) + written=put(ident(c),wire) + assert written['row_count']==1 and put(ident(c),wire)==written + bad=ident(c);bad[2]=lit(uuid.uuid4()) + denied=sql('SELECT pd_result_put_family('+','.join([*bad,lit(FAMILY),lit(wire)])+')',True,False) + assert 'fenced' in denied + changed=family_files({FAMILY:[row('different')]})[FAMILY] + assert 'conflict' in sql('SELECT pd_result_put_family('+','.join([*ident(c),lit(FAMILY),lit(changed)])+')',True,False) + record('R01_result_fencing_idempotence_conflict') + ref=rpc('pd_result_seal',*ident(c)) + assert ref==rpc('pd_result_seal',*ident(c)) + assert sql(f"SELECT count(*) FROM delphi_result_current_rows WHERE env='proof' AND scope_key={lit(scope)}",True)=='0' + rpc('pq_end_attempt',*ident(c),"'confirm_exit'",'NULL','true') + badref={**ref,'sha256':'0'*64} + invalid=make_manifest(c,{'results':badref}) + raw=json.dumps(invalid,sort_keys=True,separators=(',',':')); sha=hashlib.sha256(raw.encode()).hexdigest() + sql(f"INSERT INTO polis_queue_logs(env,attempt_id,seq,stream,line) VALUES('proof',{lit(c['attempt_id'])},0,'manifest',{lit(raw)})",True) + assert 'binding' in sql('SELECT pd_graph_finalize('+','.join([*ident(c),"'unused'",lit(sha)])+')',True,False) + assert sql(f"SELECT count(*) FROM delphi_artifacts WHERE job_id={lit(c['job_id'])}")=='0' + reply,_=finish(c,make_manifest(c,{'results':ref}),1);assert reply['outcome']=='succeeded' + aid=nodes(g)['e']['artifact']['artifact_id'] + exact=rpc('pd_result_artifact_wire',"'proof'",lit(aid),lit(FAMILY)) + assert exact['wire']==wire and exact['sha256']==written['sha256'] and exact['batch_sha256']==ref['sha256'] + read=rpc('pd_result_artifact_family',"'proof'",lit(aid),lit(FAMILY)) + assert read[0]['text']=={'S':'first'} and len(read)==1 + assert rpc('pd_result_artifact_family',"'other-environment'",lit(aid),lit(FAMILY))==[] + record('R02_manifest_binding_atomic_rollback_and_environment_isolation') + sql(f"UPDATE delphi_result_rows SET item='{{}}' WHERE batch_id={lit(c['attempt_id'])}",ok=False) + sql(f"DELETE FROM delphi_result_families WHERE batch_id={lit(c['attempt_id'])}",ok=False) + sql('SELECT * FROM delphi_result_rows',True,False) + record('R03_results_immutable_and_base_tables_denied') + # Build a real three-node graph with inline rows, then publish via the existing CAS. + graph2=graph(publish_scope,spec()) + for stage in ['graph_embed','graph_cluster','graph_narrative']: + token=claim(stage=stage) + assert token['job_id'] in {x['job_id'] for x in nodes(graph2).values()} + rpc('pq_end_attempt',*ident(token),"'confirm_exit'",'NULL','true') + payload={'family_files':family_files({FAMILY:[] if stage=='graph_narrative' else [row(stage)]})} + if stage=='graph_narrative': + archived_wire=encode_family('Delphi_JobQueue',[item_from_python(dict(job_id='generated-legacy',conversation_id='1',logs='before\0after',status='PROCESSING'))]).decode() + payload['legacy_control_files']={'Delphi_JobQueue':archived_wire} + assert finish(token,make_manifest(token,payload),0)[0]['outcome']=='succeeded' + assert sql(f"SELECT pd_result_served_bundle('proof',1,{lit(publish_scope)}) IS NULL",True)=='t' + n=nodes(graph2)['n'] + assert rpc('pd_graph_publish',"'proof'",lit(graph2['graph_id']),lit(n['job_id']),'0')['outcome']=='published' + bundle=rpc('pd_result_served_bundle',"'proof'",'1',lit(publish_scope)) + assert bundle=={'generation':1,'families':{}} or bundle=={'generation':1,'families':{FAMILY:[]}} + # Explicit empty root family shadows ancestor rows; no stale rows resurface. + assert rpc('pd_result_served',"'proof'",'1',lit(publish_scope),lit(FAMILY))==[] + assert rpc('pd_graph_publish',"'proof'",lit(graph2['graph_id']),lit(n['job_id']),'0')['outcome']=='conflict' + record('R04_coherent_publication_CAS_and_explicit_empty_shadows_old_rows') + archived=json.loads(sql(f"SELECT row_to_json(c) FROM delphi_result_legacy_controls c WHERE env='proof' AND scope_key={lit(publish_scope)}",True)) + assert archived['codec_wire']==archived_wire and archived['family']=='Delphi_JobQueue' + assert sql(f"SELECT job_id FROM delphi_result_publications WHERE env='proof' AND scope_key={lit(publish_scope)}",True)==n['job_id'] + assert sql("SELECT count(*) FROM delphi_result_jobs WHERE job_id='generated-legacy'",True)=='0' + record('R05_legacy_control_wire_NUL_and_published_root_metadata_without_activation') + # M29 extends actual stage models, but must not restore deferred graph features. + rejected_scope='result-deferred-'+uuid.uuid4().hex[:8] + assert 'superseding dead branches is not supported' in graph( + rejected_scope,spec(),sup=graph2['graph_id'],ok=False) + assert sql(f"SELECT count(*) FROM delphi_graphs WHERE env='proof' AND scope_key={lit(rejected_scope)}")=='0' + assert sql("""SELECT to_regclass('public.delphi_graph_breakers') IS NULL + AND to_regclass('public.delphi_graph_provider_resolutions') IS NULL + AND to_regprocedure('public.pd_graph_breaker_event()') IS NULL + AND to_regprocedure('public.pd_graph_resolve_provider(text,uuid,text,text,jsonb)') IS NULL""")=='t' + assert rpc('pd_result_served_bundle',"'proof'",'1',lit(publish_scope))==bundle + record('R06_deferred_features_absent_and_superseding_admission_refused_after_M29') + + +if __name__=='__main__': + main() diff --git a/delphi/tests/job_graph/test_numerical_stages.py b/delphi/tests/job_graph/test_numerical_stages.py new file mode 100644 index 0000000000..d284897248 --- /dev/null +++ b/delphi/tests/job_graph/test_numerical_stages.py @@ -0,0 +1,155 @@ +"""Fast adapter boundary checks; full numeric execution belongs to mm5 proof.""" +import hashlib +import importlib.util +import json +from pathlib import Path +import sys +import tempfile +import unittest + +SCRIPTS = Path(__file__).resolve().parents[2] / 'scripts' +sys.path.insert(0,str(SCRIPTS)) +import delphi_graph_stages as stages +import job_graph_stage +from delphi_narrative_snapshot import unfold_group_assignments + + +class NumericalBoundary(unittest.TestCase): + def test_folded_groups_expand_base_ids_to_actual_participants(self): + math = {'base-clusters': {'id': [7, 42], 'members': [[90, 0], [351]]}, + 'group-clusters': [{'id': 8, 'members': [42]}, {'id': 3, 'members': [7]}], + 'group_clusters': [{'id': 8, 'members': [351]}, {'id': 3, 'members': [90, 0]}]} + original = json.dumps(math) + self.assertEqual(unfold_group_assignments(math), {'351': 8, '90': 3, '0': 3}) + self.assertEqual(json.dumps(math), original) + self.assertNotIn('42', unfold_group_assignments(math)) + + def test_malformed_folded_groups_refuse_instead_of_using_ids_as_participants(self): + valid = {'base-clusters': {'id': [7, 42], 'members': [[90, 0], [351]]}, + 'group-clusters': [{'id': 8, 'members': [42]}, {'id': 3, 'members': [7]}], + 'group_clusters': [{'id': 8, 'members': [351]}, {'id': 3, 'members': [90, 0]}]} + mutations = [ + lambda m: m.pop('base-clusters'), + lambda m: m['base-clusters']['id'].__setitem__(1, 7), + lambda m: m['base-clusters']['members'][1].append(90), + lambda m: m['group-clusters'][0]['members'].__setitem__(0, 999), + lambda m: m['group-clusters'][1]['members'].append(42), + lambda m: m['group-clusters'][1].__setitem__('id', 8), + lambda m: m['group_clusters'][0]['members'].__setitem__(0, 42), + lambda m: m['group_clusters'][0]['members'].__setitem__(0, True), + lambda m: m['base-clusters']['id'].__setitem__(0, 7.0), + ] + for mutate in mutations: + with self.subTest(mutation=mutations.index(mutate)): + math = json.loads(json.dumps(valid)); mutate(math) + with self.assertRaises(ValueError): + unfold_group_assignments(math) + + def frame(self): + topics = {'topics':[dict(cluster_id=0,layer_id=0,topic_label='Trees',size=5)]} + payload = json.dumps(topics) + declared = dict(model=stages.MODELS['graph_narrative'],mode='full', + code=hashlib.sha256((SCRIPTS/'job_graph_stage.py').read_bytes()).hexdigest(), + runtime='python-'+sys.version.split()[0],seed=42, + snapshot=dict(data=dict(texts=['constructed text']*5)), + config=dict(comment_ids=list(range(5)),report_id='constructed-report',adapter_sha256=stages.code_digest())) + inp = dict(schema='polis-job-input/1',declared=declared, + artifacts=dict(topics=dict(payload=payload,sha256=hashlib.sha256(payload.encode()).hexdigest()))) + frame = dict(schema='polis-job-stage-frame/1',stage='graph_narrative',zid=9001, + job_id='00000000-0000-4000-8000-000000009001',run_id='run',attempt_id='attempt',input=inp) + return self.digest(frame) + + def digest(self, frame): + wire=json.dumps(frame['input']) + frame.update(input_json=wire,input_sha256=hashlib.sha256(wire.encode()).hexdigest()) + return frame + + def test_fixture_is_visible_and_codec_roundtrips(self): + result=job_graph_stage.run(self.frame()) + output=json.loads(result['output']['payload']) + self.assertTrue(output['provider_fixture']) + rows=stages.read_family(output,'Delphi_NarrativeReports') + self.assertTrue(json.loads(rows[0]['report_data'])['provider_fixture']) + self.assertEqual(rows[0]['model'],'local-narrative-fixture/1') + self.assertEqual(result['output']['sha256'],hashlib.sha256(result['output']['payload'].encode()).hexdigest()) + + def test_signed_wire_is_authoritative_over_reserialized_duplicate(self): + frame=self.frame() + frame['input']['declared']['config']['summary_metric']=0.12345678901234568 + self.digest(frame) + # Model serde_json's redundant f64 reserialization without changing wire. + frame['input']=json.loads(frame['input_json']) + frame['input']['declared']['config']['summary_metric']=0.12345678901234567 + output=json.loads(job_graph_stage.run(frame)['output']['payload']) + self.assertTrue(output['provider_fixture']) + + def test_changed_signed_wire_refused(self): + frame=self.frame(); frame['input_json'] += ' ' + with self.assertRaisesRegex(ValueError,'resolved input digest'): + job_graph_stage.run(frame) + + def test_upstream_digest_refused(self): + frame=self.frame(); frame['input']['artifacts']['topics']['payload']='{}' + with self.assertRaisesRegex(ValueError,'upstream artifact digest'): + job_graph_stage.run(self.digest(frame)) + + def test_code_change_refused(self): + frame=self.frame(); frame['input']['declared']['config']['adapter_sha256']='0'*64 + with self.assertRaisesRegex(ValueError,'provenance'): + job_graph_stage.run(self.digest(frame)) + + def test_duplicate_comment_ids_refused(self): + frame=self.frame(); frame['input']['declared']['config']['comment_ids']=[0]*5 + with self.assertRaisesRegex(ValueError,'comment ids'): + job_graph_stage.run(self.digest(frame)) + + def test_too_many_texts_refused_without_truncation(self): + frame=self.frame(); frame['input']['declared']['snapshot']['data']['texts']=['text']*2001 + with self.assertRaisesRegex(ValueError,'bounded texts'): + job_graph_stage.run(self.digest(frame)) + + def test_real_summary_selects_topic_and_all_eligible_global_sections(self): + # Aggregate semantic counts, never raw storage votes or DB inserts. + ids=list(range(1,6)); texts=['generated statement '+str(i) for i in ids] + records=[dict(comment_id=i,**{'comment-id':i,'total-votes':10,'total-agrees':6, + 'total-disagrees':1,'total-passes':3},votes=10,agrees=6,disagrees=1,passes=3, + comment_extremity=1.5,group_aware_consensus=0.9,num_groups=2) for i in ids] + context=dict(schema='delphi-narrative-context/1',comments=records) + topic=dict(layer_id=0,cluster_id=1,comment_ids=ids,topic_label='Generated',size=5) + selected=stages.narrative_sections([topic],context,ids,texts) + self.assertEqual(len(selected),4) + self.assertEqual({p.get('_global') for p in selected}, + {None,'groups','group_informed_consensus','uncertainty'}) + self.assertTrue(all(p['_allowed_ids']==ids for p in selected)) + self.assertTrue(all('generated statement 1' in p['_prompt'] for p in selected)) + + def test_citation_ids_match_actual_xml_after_legacy_comment_limit(self): + import xml.etree.ElementTree as ET + import xmltodict + ids=list(range(1,121));texts=['generated statement '+str(i) for i in ids] + records=[dict(comment_id=i,**{'comment-id':i,'total-votes':10,'total-agrees':6, + 'total-disagrees':1,'total-passes':3},votes=10,agrees=6,disagrees=1,passes=3, + comment_extremity=1.5,group_aware_consensus=0.9,num_groups=2) for i in ids] + sections=stages.narrative_sections([],dict(schema='delphi-narrative-context/1',comments=records),ids,texts) + self.assertEqual(len(sections),3) + for section in sections: + structured=xmltodict.parse(section['_prompt'])['polisAnalysisPrompt']['data']['content']['structured_comments'] + actual=[int(row.attrib['id']) for row in ET.fromstring(structured).findall('comment')] + self.assertEqual(section['_allowed_ids'],actual) + self.assertLess(len(actual),len(ids)) + + def test_model_digest_changes_with_weights(self): + with tempfile.TemporaryDirectory() as directory: + path=Path(directory)/'model.safetensors'; path.write_bytes(b'generated model bytes') + old=stages.model_digest(directory) + path.write_bytes(b'changed generated bytes') + self.assertNotEqual(old,stages.model_digest(directory)) + + def test_float_codec_conversion_is_explicit(self): + output=dict(family_files=stages.family_files({'Delphi_UMAPGraph':[ + dict(conversation_id='9001',edge_id='0_0',position={'x':0.25,'y':0.5})]})) + from decimal import Decimal + self.assertEqual(stages.read_family(output,'Delphi_UMAPGraph')[0]['position']['x'],Decimal('0.25')) + +if __name__=='__main__': + unittest.main() diff --git a/delphi/tests/poller/test_capacity_queue_postgres.py b/delphi/tests/poller/test_capacity_queue_postgres.py index d47475c555..5f25c832ec 100644 --- a/delphi/tests/poller/test_capacity_queue_postgres.py +++ b/delphi/tests/poller/test_capacity_queue_postgres.py @@ -235,6 +235,32 @@ def release(self, scope): class TestTheContract: + def test_operator_cli_sizes_and_admits_once(self, queue_db, db, env, labels, + monkeypatch, capsys): + from scripts import enqueue_math_rebuild as cli + + small, large = labels + (zid,) = fresh_zids(1) + seed_conversation(db, zid, participants=3, comments=3) + monkeypatch.setenv("DATABASE_URL", queue_db[0]) + monkeypatch.setenv("MATH_CAPACITY_QUEUE_DSN", queue_db[1]) + monkeypatch.setenv("MATH_CAPACITY_QUEUE_ENV", env) + args = ["--zid", str(zid), "--staged-label", large, + "--target-label", small, "--source-commit", COMMIT] + assert cli.main(args + ["--dry-run"]) == 0 + dry = json.loads(capsys.readouterr().out) + assert dry["outcome"] == "dry_run" and jobs(db, env) == 0 + assert dry["sizes"] == {"votes": 6, "voters": 3, "comments": 3} + assert cli.main(args) == 0 + first = json.loads(capsys.readouterr().out) + assert first["outcome"] == "enqueued" + assert first["config"] == dry["config"] + assert cli.main(args) == 0 + second = json.loads(capsys.readouterr().out) + assert second["outcome"] == "existing" + assert second["job_id"] == first["job_id"] and jobs(db, env) == 1 + assert job_row(db, env, first["job_id"])[1:3] == ("math_rebuild", "large") + def test_enqueue_as_the_executor_login_and_the_row_it_makes(self, queue_db, db, env, labels): small, large = labels (zid,) = fresh_zids(1) diff --git a/delphi/tests/poller/test_enqueue_math_rebuild.py b/delphi/tests/poller/test_enqueue_math_rebuild.py new file mode 100644 index 0000000000..94fbbf0aed --- /dev/null +++ b/delphi/tests/poller/test_enqueue_math_rebuild.py @@ -0,0 +1,72 @@ +"""Operator admissions share the real queue contract, even for a small conversation.""" +from unittest.mock import Mock +import json + +import pytest + +from scripts import enqueue_math_rebuild as cli +from polismath.poller.admission import MemoryModel + + +COMMIT = "a" * 40 + + +@pytest.mark.parametrize("zid,staged,target,commit", [ + (0, "stage", "python", COMMIT), + (7, "python", "target", COMMIT), + (7, "prod", "target", COMMIT), + (7, "stage", "stage", COMMIT), + (7, "stage", "python", "unknown"), + (7, "stage/invalid", "python", COMMIT), +]) +def test_refuses_invalid_admission_before_database(monkeypatch, zid, staged, target, commit): + connection = Mock(side_effect=AssertionError("must not connect")) + monkeypatch.setattr(cli.psycopg2, "connect", connection) + assert cli.main(["--zid", str(zid), "--staged-label", staged, + "--target-label", target, "--source-commit", commit]) == 2 + connection.assert_not_called() + + +def test_small_conversation_uses_existing_admission_without_threshold_override(): + queue = Mock() + queue.enqueue_math_rebuild.return_value = ("enqueued", "generated-job") + model = MemoryModel() + result = cli.admit(7, staged_label="staged", target_label="python", + source_commit=COMMIT, model=model, snapshot=((36, 6, 6), 123), + queue=queue) + assert result["outcome"] == "enqueued" + assert result["config"]["need_bytes"] == model.above_base_bytes(36, 6, 6) + assert result["config"]["input_through_ms"] == 123 + queue.enqueue_math_rebuild.assert_called_once_with( + 7, config=result["config"], staged_label="staged", target_label="python") + + +def test_dry_run_is_read_only(): + queue = Mock() + result = cli.admit(7, staged_label="staged", target_label="python", + source_commit=COMMIT, model=MemoryModel(), + snapshot=((0, 0, 0), None), queue=queue, dry_run=True) + assert result["outcome"] == "dry_run" and result["job_id"] is None + queue.enqueue_math_rebuild.assert_not_called() + + +def test_database_errors_do_not_disclose_connection_details(monkeypatch, capsys): + monkeypatch.setenv("DATABASE_URL", "generated-db-url") + monkeypatch.setattr(cli, "read_snapshot", Mock( + side_effect=cli.psycopg2.OperationalError("sensitive-connection-detail"))) + assert cli.main(["--zid", "7", "--staged-label", "stage", "--target-label", "python", + "--source-commit", COMMIT, "--dry-run"]) == 2 + assert "sensitive-connection-detail" not in capsys.readouterr().err + + +def test_active_scope_conflict_reports_failure_and_preserves_existing_job(monkeypatch, capsys): + monkeypatch.setenv("DATABASE_URL", "generated-db-url") + monkeypatch.setattr(cli, "read_snapshot", Mock(return_value=((36, 6, 6), 123))) + queue = Mock() + queue.enqueue_math_rebuild.return_value = ("conflict", "existing-job") + monkeypatch.setattr(cli, "QueueClient", Mock(return_value=queue)) + assert cli.main(["--zid", "7", "--staged-label", "stage", "--target-label", "python", + "--source-commit", COMMIT]) == 1 + result = json.loads(capsys.readouterr().out) + assert result["outcome"] == "conflict" and result["job_id"] == "existing-job" + queue.enqueue_math_rebuild.assert_called_once() diff --git a/delphi/tests/test_delphi_legacy_import.py b/delphi/tests/test_delphi_legacy_import.py new file mode 100644 index 0000000000..71ad5bbcad --- /dev/null +++ b/delphi/tests/test_delphi_legacy_import.py @@ -0,0 +1,248 @@ +"""All twenty generated codec families cross the local export/import boundary.""" +import copy +import json +import os +import stat +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock, Mock + +import pytest + +from polismath.delphi_storage import legacy_import as legacy +from polismath.delphi_storage.codec import decode_family, encode_family, item_from_python +from polismath.delphi_storage.golden_corpus import GOLDEN, corpus + + +ZID = 9001 +REPORT = "r9001generated" + + +def files(): + return {family: encode_family(family, rows).decode() for family, rows in corpus().items()} + + +def test_every_family_accounted_for_and_control_rows_never_activated(): + source = files() + result = legacy.import_output(source, ZID, [REPORT]) + assert set(result["counts"]) == set(legacy.FAMILIES) + for family, wire in source.items(): + counts = result["counts"][family] + assert counts["source"] == sum(counts[k] for k in ("imported", "archived", "quarantined")) + if family in legacy.CONTROL_FAMILIES: + assert result["legacy_control_files"][family] == wire + assert family not in result["family_files"] and counts["imported"] == 0 + else: + assert result["family_files"][family] == wire + + +def test_nul_quarantine_preserves_exact_row_and_other_row_imports(): + family = "Delphi_CommentEmbeddings" + rows = [item_from_python(dict(conversation_id=str(ZID), comment_id=1, text="valid")), + item_from_python(dict(conversation_id=str(ZID), comment_id=2, text="bad\0value"))] + result = legacy.import_output({family: encode_family(family, rows).decode()}, ZID, []) + assert result["counts"][family] == dict(source=2, imported=1, archived=0, quarantined=1) + assert result["quarantine"][family]["reason"] == "postgres-jsonb-nul" + assert decode_family(result["family_files"][family].encode())[1] == rows[:1] + assert decode_family(result["quarantine"][family]["codec_wire"].encode())[1] == rows[1:] + # Canonical wire in a JSON string has escaped backslashes, never literal NUL. + assert "\0" not in json.dumps(result) + + +def test_binary_nul_is_not_quarantined(): + assert not legacy.has_nul({"B": b"\0"}) + assert legacy.has_nul({"M": {"bad\0key": {"S": "value"}}}) + + +@pytest.mark.parametrize("zid,reports", [(9002, [REPORT]), (ZID, []), (ZID, ["wrong-report"])]) +def test_cross_conversation_or_unbound_report_refused(zid, reports): + with pytest.raises(ValueError, match="mismatch|binding"): + legacy.import_output(files(), zid, reports) + + +def test_manifest_reader_accepts_frozen_golden_bytes(): + assert legacy.read_export(GOLDEN) == files() + + +def test_local_export_consumes_all_pages_and_roundtrips(tmp_path): + family = "Delphi_CommentEmbeddings" + rows = corpus()[family] + client = Mock() + token = {"conversation_id": {"S": str(ZID)}, "comment_id": {"N": "1"}} + client.scan.side_effect = [{"Items": rows[:1], "LastEvaluatedKey": token}, {"Items": rows[1:]}] + legacy.export_local(client, tmp_path, [family]) + assert client.scan.call_args_list[1].kwargs["ExclusiveStartKey"] == token + assert legacy.read_export(tmp_path)[family] == encode_family(family, rows).decode() + + +@pytest.mark.parametrize("endpoint", ["https://dynamodb.us-east-1.amazonaws.com", "http://example.com", + "http://127.0.0.1@evil.invalid", "http://127.0.0.1/?target=prod"]) +def test_export_refuses_nonlocal_endpoints_before_sdk(endpoint): + with pytest.raises(ValueError, match="loopback"): + legacy.local_client(endpoint) + + +def test_changed_export_refused(tmp_path): + family = "Delphi_CommentEmbeddings" + client = Mock() + client.scan.return_value = {"Items": corpus()[family]} + legacy.export_local(client, tmp_path, [family]) + path = tmp_path / (family + ".jsonl") + path.write_bytes(path.read_bytes().replace(b"generated-embedding-model", b"changed-embedding-model")) + with pytest.raises(ValueError, match="digest"): + legacy.read_export(tmp_path) + + +def frame(): + spec = legacy.build_spec(files(), ZID, [REPORT]) + return dict(stage="graph_narrative", zid=ZID, + input=dict(declared=spec["nodes"][0]["declared"], artifacts={})) + + +def test_import_worker_rechecks_source_and_code_provenance(): + actual = frame() + assert legacy.execute_import(actual) == legacy.import_output(files(), ZID, [REPORT]) + for name in ("source_sha256", "importer_sha256", "codec_sha256"): + changed = copy.deepcopy(actual) + changed["input"]["declared"]["config"][name] = "0" * 64 + with pytest.raises(ValueError, match="digest|provenance"): + legacy.execute_import(changed) + + +def test_normal_graph_stage_manifest_binds_import_payload(): + from scripts import job_graph_stage + actual = frame() + actual.update(schema="polis-job-stage-frame/1", job_id="generated-job", + run_id="generated-run", attempt_id="generated-attempt") + actual["input"]["schema"] = "polis-job-input/1" + wire = json.dumps(actual["input"]) + actual.update(input_json=wire, input_sha256=legacy.digest(wire.encode())) + manifest = job_graph_stage.run(actual) + payload = manifest["output"]["payload"] + assert manifest["output"]["sha256"] == legacy.digest(payload.encode()) + assert json.loads(payload)["counts"] == legacy.import_output(files(), ZID, [REPORT])["counts"] + + +def test_identical_export_produces_identical_request_and_different_export_does_not(): + first = legacy.build_spec(files(), ZID, [REPORT]) + assert first == legacy.build_spec(files(), ZID, [REPORT]) + changed = files() + changed["Delphi_CommentEmbeddings"] = changed["Delphi_CommentEmbeddings"].replace( + "generated-embedding-model", "changed-embedding-model") + assert first != legacy.build_spec(changed, ZID, [REPORT]) + + +def test_report_mapping_uses_actual_database_relationship(): + connection = MagicMock() + cursor = connection.cursor.return_value.__enter__.return_value + cursor.fetchall.return_value = [(REPORT,)] + legacy.validate_report_mapping(connection, ZID, [REPORT]) + cursor.fetchall.return_value = [] + with pytest.raises(ValueError, match="attached"): + legacy.validate_report_mapping(connection, ZID, [REPORT]) + + +def test_oversized_artifact_refused_without_truncation(): + family = "Delphi_CommentEmbeddings" + source = {family: encode_family(family, [item_from_python(dict( + conversation_id=str(ZID), comment_id=1, text="x" * legacy.MAX_BYTES))]).decode()} + with pytest.raises(ValueError, match="bounded inline"): + legacy.import_output(source, ZID, []) + + +@pytest.mark.parametrize("target", ["SHA256SUMS", "Delphi_CommentEmbeddings.jsonl"]) +@pytest.mark.parametrize("kind", ["symlink", "fifo"]) +def test_export_reader_refuses_nonregular_inputs_without_blocking(tmp_path, target, kind): + family = "Delphi_CommentEmbeddings" + raw = encode_family(family, [item_from_python(dict(conversation_id=str(ZID), comment_id=1))]) + (tmp_path / (family + ".jsonl")).write_bytes(raw) + (tmp_path / "SHA256SUMS").write_text(f"{legacy.digest(raw)} {family}.jsonl\n") + path = tmp_path / target + original = path.read_bytes() + path.unlink() + if kind == "symlink": + outside = tmp_path / "original" + outside.write_bytes(original) + path.symlink_to(outside) + else: + os.mkfifo(path) + # A subprocess timeout is the regression assertion: an ordinary FIFO open + # would block forever before any regular-file validation could run. + program = """from polismath.delphi_storage.legacy_import import read_export +import sys +try: + read_export(sys.argv[1]) +except (OSError, ValueError): + sys.exit(0) +sys.exit(1) +""" + result = subprocess.run([sys.executable, "-c", program, str(tmp_path)], + cwd=Path(legacy.__file__).resolve().parents[2], capture_output=True, timeout=5) + assert result.returncode == 0, result.stderr.decode() + + +@pytest.mark.parametrize("target", ["SHA256SUMS", "Delphi_CommentEmbeddings.jsonl"]) +def test_export_reader_refuses_oversize_regular_files(tmp_path, target): + family = "Delphi_CommentEmbeddings" + raw = encode_family(family, []) + (tmp_path / (family + ".jsonl")).write_bytes(raw) + (tmp_path / "SHA256SUMS").write_text(f"{legacy.digest(raw)} {family}.jsonl\n") + limit = legacy.MAX_MANIFEST_BYTES if target == "SHA256SUMS" else legacy.MAX_BYTES + with (tmp_path / target).open("wb") as source: + source.truncate(limit + 1) + with pytest.raises(ValueError, match="byte limit"): + legacy.read_export(tmp_path) + + +def test_export_reader_enforces_actual_bytes_if_file_grows_after_stat(tmp_path, monkeypatch): + source = tmp_path / "growing" + source.write_bytes(b"x" * 33) + monkeypatch.setattr(legacy.os, "fstat", lambda fd: SimpleNamespace(st_mode=stat.S_IFREG, st_size=0)) + with pytest.raises(ValueError, match="byte limit"): + legacy.read_regular_file(source, 32) + + +@pytest.mark.parametrize("paged", [False, True]) +def test_export_budget_stops_before_fetching_more_or_writing_oversize_family(tmp_path, paged): + family = "Delphi_CommentEmbeddings" + rows = [item_from_python(dict(conversation_id=str(ZID), comment_id=i, text="x" * 240_000)) for i in range(2)] + token = {"conversation_id": {"S": str(ZID)}, "comment_id": {"N": "1"}} + pages = ([{"Items": rows[:1], "LastEvaluatedKey": token}, {"Items": rows[1:], "LastEvaluatedKey": token}] + if paged else [{"Items": rows, "LastEvaluatedKey": token}]) + client = Mock() + client.scan.side_effect = pages + with pytest.raises(ValueError, match="bounded inline"): + legacy.export_local(client, tmp_path, [family]) + assert client.scan.call_count == len(pages) + assert not (tmp_path / (family + ".jsonl")).exists() + assert not (tmp_path / "SHA256SUMS").exists() + + +def test_export_budget_is_shared_across_families(tmp_path): + families = ["Delphi_CommentEmbeddings", "Delphi_CommentExtremity"] + client = Mock() + client.scan.side_effect = [ + {"Items": [item_from_python(dict(conversation_id=str(ZID), comment_id=1, text="x" * 240_000))]}, + {"Items": [item_from_python(dict(conversation_id=str(ZID), comment_id="1", text="x" * 240_000))]}, + ] + with pytest.raises(ValueError, match="bounded inline"): + legacy.export_local(client, tmp_path, families) + assert client.scan.call_count == 2 + assert not (tmp_path / "SHA256SUMS").exists() + + +def test_export_reader_shares_byte_limit_across_families(tmp_path): + source = { + "Delphi_CommentEmbeddings": [item_from_python(dict(conversation_id=str(ZID), comment_id=1, text="x" * 240_000))], + "Delphi_CommentExtremity": [item_from_python(dict(conversation_id=str(ZID), comment_id="1", text="x" * 240_000))], + } + manifest = [] + for family, rows in source.items(): + raw = encode_family(family, rows) + (tmp_path / (family + ".jsonl")).write_bytes(raw) + manifest.append(f"{legacy.digest(raw)} {family}.jsonl\n") + (tmp_path / "SHA256SUMS").write_text("".join(manifest)) + with pytest.raises(ValueError, match="byte limit"): + legacy.read_export(tmp_path) diff --git a/delphi/tests/test_delphi_postgres_results.py b/delphi/tests/test_delphi_postgres_results.py new file mode 100644 index 0000000000..745ffd8ba9 --- /dev/null +++ b/delphi/tests/test_delphi_postgres_results.py @@ -0,0 +1,46 @@ +"""Codec boundary tests; no database mocks stand in for the separate SQL proof.""" +import json +import unittest +from decimal import Decimal + +from polismath.delphi_storage.codec import CodecError, decode_family +from polismath.delphi_storage.postgres import family_files, decode_rows + + +class PostgresResultCodecTest(unittest.TestCase): + def test_exact_numeric_binary_and_set_roundtrip(self): + family = "Delphi_CommentEmbeddings" + rows = [{"conversation_id": "1", "comment_id": Decimal("7"), + "embedding": [Decimal("0.1234567890123456789012345678")], + "binary": b"\x00\xff", "labels": {"tree", "water"}, + "document": '{"preserved":"as string"}'}] + wire = family_files({family: rows})[family] + self.assertEqual(decode_rows(family, [json.loads(x) for x in wire.splitlines()[1:]]), rows) + self.assertEqual(decode_family(wire.encode())[0], family) + + def test_empty_family_is_explicit(self): + family = "Delphi_CommentEmbeddings" + self.assertEqual(decode_family(family_files({family: []})[family].encode())[1], []) + self.assertEqual(decode_rows(family, []), []) + + def test_queue_tables_are_not_results(self): + for family in ["Delphi_JobQueue", "Delphi_JobActiveGuard", "unknown"]: + with self.subTest(family=family), self.assertRaises(CodecError): + family_files({family: []}) + + def test_duplicate_key_refused(self): + row = {"conversation_id": "1", "comment_id": 1} + with self.assertRaises(CodecError): + family_files({"Delphi_CommentEmbeddings": [row, row]}) + + def test_float_refused_instead_of_losing_precision(self): + with self.assertRaises(CodecError): + family_files({"Delphi_CommentEmbeddings": [{"conversation_id": "1", "comment_id": 1, "value": 0.25}]}) + + def test_read_invalid_tag_refused(self): + with self.assertRaises(CodecError): + decode_rows("Delphi_CommentEmbeddings", [{"conversation_id": {"S": "1"}, "comment_id": {"N": "01"}}]) + + +if __name__ == "__main__": + unittest.main() diff --git a/delphi/tests/test_delphi_result_resource.py b/delphi/tests/test_delphi_result_resource.py new file mode 100644 index 0000000000..86a781b605 --- /dev/null +++ b/delphi/tests/test_delphi_result_resource.py @@ -0,0 +1,125 @@ +import os +import unittest +from unittest.mock import patch, Mock +from decimal import Decimal +from polismath.delphi_storage.resource import PostgresResource, result_resource +from polismath.delphi_storage.codec import item_from_python, encode_family + + +class Cursor: + def __init__(self,rows):self.rows=rows;self.archives=[];self.calls=[] + def __enter__(self):return self + def __exit__(self,*args):pass + def execute(self,sql,args=()):self.calls.append((sql,args)) + def fetchall(self):return self.archives if 'legacy_controls' in self.calls[-1][0] else self.rows + + +class Connection: + def __init__(self,rows):self.cur=Cursor(rows) + def cursor(self):return self.cur + + +class ResultResourceTest(unittest.TestCase): + def setUp(self): + self.env=patch.dict(os.environ,DELPHI_RESULT_BACKEND='postgres',DELPHI_RESULT_ENV='test-results') + self.env.start();self.addCleanup(self.env.stop) + self.rows=[(1,'scope',1,item_from_python(dict(conversation_id='1',comment_id=i,value=Decimal('0.25')))) for i in range(3)] + self.connection=Connection(self.rows) + self.table=PostgresResource(self.connection).Table('Delphi_CommentEmbeddings') + + def test_query_pushes_bound_key_and_preserves_decimal(self): + reply=self.table.query(KeyConditionExpression='conversation_id = :id',ExpressionAttributeValues={':id':'1'}) + self.assertEqual(reply['Count'],3) + self.assertEqual(reply['Items'][0]['value'],Decimal('0.25')) + sql,binds=self.connection.cur.calls[0] + self.assertIn('item->%s = %s::jsonb',sql) + self.assertIn('conversation_id',binds) + + def test_pagination_and_generation_change(self): + first=self.table.query(Limit=1) + self.assertEqual(self.table.query(Limit=1,ExclusiveStartKey=first['LastEvaluatedKey'])['Items'][0]['comment_id'],1) + self.connection.cur.rows=[(zid,'scope',2,row) for zid,_,_,row in self.rows] + with self.assertRaisesRegex(ValueError,'generation changed'): + self.table.query(ExclusiveStartKey=first['LastEvaluatedKey']) + + def test_ambiguous_scopes_refused(self): + self.connection.cur.rows.append((1,'other',1,item_from_python(dict(conversation_id='1',comment_id=0,value=Decimal('1'))))) + with self.assertRaisesRegex(ValueError,'ambiguous'): + self.table.query() + + def test_cross_conversation_cursor_tracks_each_generation(self): + self.connection.cur.rows += [(2,'scope',1,item_from_python(dict(conversation_id='2',comment_id=0)))] + first=self.table.scan(Limit=1) + self.connection.cur.rows=[(zid,scope,2 if zid==1 else generation,item) for zid,scope,generation,item in self.connection.cur.rows] + with self.assertRaisesRegex(ValueError,'generation changed'): + self.table.scan(ExclusiveStartKey=first['LastEvaluatedKey']) + + def test_legacy_metadata_retains_nul_and_exact_numbers_without_activation(self): + connection=Connection([({'job_id':'current','status':'COMPLETED'},)]) + archived=[dict(job_id='legacy',conversation_id='1',status='PROCESSING',logs='a\0b',job_config={'large':Decimal('9007199254740993')},binary=b'bytes'),dict(job_id='current',status='PROCESSING')] + wire=encode_family('Delphi_JobQueue',[item_from_python(item) for item in archived]).decode() + connection.cur.archives=[(1,'scope',4,wire)] + reply=PostgresResource(connection).Table('Delphi_JobQueue').scan() + legacy=next(item for item in reply['Items'] if item['job_id']=='legacy') + self.assertTrue(legacy['archived']);self.assertEqual(legacy['logs'],'a\0b') + self.assertEqual(legacy['job_config']['large'],Decimal('9007199254740993')) + self.assertEqual(legacy['binary'],b'bytes') + self.assertEqual(next(item for item in reply['Items'] if item['job_id']=='current')['status'],'COMPLETED') + + def test_mutation_refused(self): + with self.assertRaisesRegex(RuntimeError,'immutable'): + self.table.put_item(Item={}) + + def test_postgres_selection_never_calls_boto(self): + # Import is lazy: this test works with no boto3 package installed. + with patch.dict('sys.modules',{'boto3':None}): + self.assertIsInstance(result_resource(),PostgresResource) + + def test_index_sort_uses_created_time(self): + rows=[({'job_id':'z-older','created_at':'2025-01-01T00:00:00Z'},), + ({'job_id':'a-newer','created_at':'2026-01-01T00:00:00Z'},)] + table=PostgresResource(Connection(rows)).Table('Delphi_JobQueue') + self.assertEqual(table.query(IndexName='ConversationIndex',ScanIndexForward=False,Limit=1)['Items'][0]['job_id'],'a-newer') + + def test_job_get_reads_bound_attempt_logs(self): + connection=Connection([({'job_id':'generated-job'},)]) + fetch=connection.cur.fetchall + def rows(): + sql=connection.cur.calls[-1][0] + if 'pq_job_status' in sql:return [({'attempt_id':'generated-attempt'},)] + if 'pq_attempt_logs' in sql:return [('2026-01-01T00:00:00Z','INFO','actual child output')] + return fetch() + connection.cur.fetchall=rows + item=PostgresResource(connection).Table('Delphi_JobQueue').get_item(Key={'job_id':'generated-job'})['Item'] + self.assertEqual(item['log_attempt_id'],'generated-attempt') + self.assertIn('actual child output',item['logs']) + self.assertEqual(connection.cur.calls[-1][1],['test-results','generated-attempt']) + self.assertIn('NULL,1000',connection.cur.calls[-1][0]) + + def test_job_metadata_uses_same_graph_scope_as_archives(self): + connection=Connection([]) + with patch.dict(os.environ,DELPHI_RESULT_SCOPE='delphi'): + PostgresResource(connection).Table('Delphi_JobQueue').scan() + self.assertEqual(len(connection.cur.calls),2) + for sql,binds in connection.cur.calls: + self.assertIn('scope_key=%s',sql) + self.assertEqual(binds[-1],'delphi') + + def test_default_and_explicit_dynamo_forward_unchanged(self): + for backend in (None, 'dynamodb'): + with self.subTest(backend=backend), patch.dict(os.environ, {}, clear=True): + if backend is not None: + os.environ['DELPHI_RESULT_BACKEND'] = backend + boto = Mock() + with patch.dict('sys.modules', {'boto3': boto}): + result = result_resource(endpoint_url='http://localhost:8000') + self.assertIs(result, boto.resource.return_value) + boto.resource.assert_called_once_with('dynamodb', endpoint_url='http://localhost:8000') + + def test_unknown_backend_refused(self): + with patch.dict(os.environ,DELPHI_RESULT_BACKEND='postgress'): + with self.assertRaisesRegex(ValueError,'invalid'): + result_resource() + + +if __name__=='__main__':unittest.main() diff --git a/delphi/tests/test_delphi_writer.py b/delphi/tests/test_delphi_writer.py new file mode 100644 index 0000000000..642cd40656 --- /dev/null +++ b/delphi/tests/test_delphi_writer.py @@ -0,0 +1,115 @@ +"""Only generated result rows; no votes or paid provider calls.""" +import json +import os +import uuid +from decimal import Decimal +import pytest +from polismath.delphi_storage.codec import item_from_python,decode_family +from polismath.delphi_storage.resource import result_resource +from polismath.delphi_storage.writer import WriterResource +from polismath.delphi_storage.postgres import RESULT_FAMILIES + +@pytest.fixture +def writer(tmp_path,monkeypatch): + jid,run,attempt=[str(uuid.uuid4()) for _ in range(3)] + frame=dict(schema='polis-jobs.frame/1',env='proof',zid=1,job_id=jid,run_id=run,attempt_id=attempt,lease_epoch='1', + config={'result_backend':'postgres'},writer_base={'schema':'delphi-writer-base/1','zid':1,'families':{}}) + path=tmp_path/'frame.json';path.write_text(json.dumps(frame)) + for key,value in dict(DELPHI_RESULT_BACKEND='postgres',DELPHI_OUTPUT_MANIFEST=str(tmp_path/'manifest.json'), + DELPHI_FRAME=str(path),DELPHI_JOB_ID=jid,DELPHI_RUN_ID=run,DELPHI_ATTEMPT_ID=attempt).items():monkeypatch.setenv(key,value) + import boto3 + monkeypatch.setattr(boto3,'resource',lambda *a,**k:pytest.fail('Dynamo resource created')) + return result_resource() + +@pytest.mark.parametrize('family',sorted(RESULT_FAMILIES)) +def test_roundtrip_every_family(writer,family): + from polismath.delphi_storage.codec import FAMILIES + special={'zid':'1','conversation_id':'1','zid_tick':'1:1','zid_tick_gid':'1:1:0','zid_topic_jobid':'1#topic#job'} + row={k:(Decimal(0) if tag=='N' else special.get(k,'generated')) for k,tag in FAMILIES[family]['key']} + row.update(value=Decimal('1.234567890123456789'),document='{"keep":"string"}',binary=b'\x00\x01',names={'x','y'}) + table=writer.Table(family) + with table.batch_writer() as batch:batch.put_item(Item=row) + key={k:row[k] for k,_ in FAMILIES[family]['key']} + assert WriterResource().Table(family).get_item(Key=key)['Item']==row + table.update_item(Key=key,UpdateExpression='SET #v = :v',ExpressionAttributeNames={'#v':'new'},ExpressionAttributeValues={':v':'changed'}) + assert table.get_item(Key=key)['Item']['new']=='changed' + table.delete_item(Key=key) + assert table.get_item(Key=key)=={} + +def test_spool_keeps_exact_decimal_and_deleted_family(writer): + table=writer.Table('Delphi_CommentEmbeddings') + table.put_item(Item=dict(conversation_id='1',comment_id=Decimal(2),value=Decimal('0.00000000000000000001'))) + files=writer.spool() + assert len(files)==18 + for family,description in files.items(): + actual,rows=decode_family((writer.directory/description['file']).read_bytes()) + assert family==actual + assert len(rows)==(1 if family==table.name else 0) + +def test_queue_and_cross_conversation_writes_refuse(writer): + with pytest.raises(RuntimeError,match='queue state'):writer.Table('Delphi_JobQueue').put_item(Item={'job_id':'x'}) + with pytest.raises(ValueError,match='conversation'):writer.Table('Delphi_CommentEmbeddings').put_item(Item={'conversation_id':'2','comment_id':1}) + +def test_reset_is_private_and_preserves_other_products(writer): + writer.Table('Delphi_CommentEmbeddings').put_item(Item={'conversation_id':'1','comment_id':1}) + writer.Table('Delphi_CollectiveStatement').put_item(Item={'zid_topic_jobid':'1#t#j','text':'keep'}) + writer.reset() + assert not writer.Table('Delphi_CommentEmbeddings').scan()['Items'] + assert writer.Table('Delphi_CollectiveStatement').scan()['Items'][0]['text']=='keep' + +def test_changed_attempt_binding_refuses(writer,monkeypatch): + monkeypatch.setenv('DELPHI_ATTEMPT_ID',str(uuid.uuid4())) + with pytest.raises(ValueError,match='bound queue attempt'):WriterResource() + +def test_pagination_projection_and_filters(writer): + table=writer.Table('Delphi_CommentEmbeddings') + for cid in range(5):table.put_item(Item={'conversation_id':'1','comment_id':cid,'text':str(cid)}) + first=table.query(KeyConditionExpression='conversation_id = :z',ExpressionAttributeValues={':z':'1'},Limit=2) + second=table.query(KeyConditionExpression='conversation_id = :z',ExpressionAttributeValues={':z':'1'},Limit=2,ExclusiveStartKey=first['LastEvaluatedKey']) + assert [i['comment_id'] for i in second['Items']]==[2,3] + assert table.scan(Select='COUNT')['Count']==5 + +def test_actual_pipeline_orchestration_uses_private_reset_and_v2_manifest(writer,monkeypatch): + import run_delphi + from polismath.job_child import census + import sys + from types import SimpleNamespace + frame=writer.frame + frame.update(stage='delphi_full_pipeline',phase='run',report_id=None,inputs={},provider={'batch_id':None}) + frame['config'].update(include_moderation=True,exclude_comment_selections=True,model=None,batch_size=None) + Path=__import__('pathlib').Path + Path(os.environ['DELPHI_FRAME']).write_text(json.dumps(frame)) + for key,value in {'DELPHI_LEASE_EPOCH':'1','DELPHI_STAGE':'delphi_full_pipeline','DELPHI_PHASE':'run'}.items():monkeypatch.setenv(key,value) + writer.Table('Delphi_CollectiveStatement').put_item(Item={'zid_topic_jobid':'1#old#report','text':'preserved'}) + writer.Table('Delphi_CommentEmbeddings').put_item(Item={'conversation_id':'1','comment_id':999}) + monkeypatch.setattr(census,'default_pg_query',lambda:None) + monkeypatch.setattr(census,'observe_inputs',lambda *a:{'math_env':'proof','math_tick':None,'math_caching_tick':None,'comment_set_sha256':None,'vote_hwm':None}) + scripts=[] + def stage(command,**kwargs): + scripts.append(command[1]) + assert 'reset_conversation' not in command[1] + resource=result_resource() + if command[1].endswith('run_math_pipeline.py'): + resource.Table('Delphi_PCAConversationConfig').put_item(Item={'zid':'1','math_tick':1}) + if command[1].endswith('run_pipeline.py'): + resource.Table('Delphi_CommentEmbeddings').put_item(Item={'conversation_id':'1','comment_id':0}) + return SimpleNamespace(returncode=0) + monkeypatch.setattr(run_delphi.subprocess,'run',stage) + monkeypatch.setattr(sys,'argv',['run_delphi.py','--zid=1']) + with pytest.raises(SystemExit) as done:run_delphi.main() + assert done.value.code==0 and len(scripts)>=4 + manifest=json.loads(Path(os.environ['DELPHI_OUTPUT_MANIFEST']).read_text()) + assert manifest['schema']=='polis-jobs.output-manifest/2' + assert all(o['store']=='postgres' for o in manifest['outputs']) + resource=result_resource() + assert [i['comment_id'] for i in resource.Table('Delphi_CommentEmbeddings').scan()['Items']]==[0] + assert resource.Table('Delphi_CollectiveStatement').scan()['Items'][0]['text']=='preserved' + +def test_postgres_bootstrap_and_legacy_poller_never_construct_dynamo(writer,monkeypatch): + import boto3 + import create_dynamodb_tables + from scripts.job_poller import JobProcessor + monkeypatch.setattr(boto3,'Session',lambda *a,**k:pytest.fail('Dynamo session created')) + assert create_dynamodb_tables.create_tables()==[] + assert create_dynamodb_tables.main() is None + with pytest.raises(RuntimeError,match='polis-jobs'):JobProcessor() diff --git a/delphi/tests/test_narrative_audit.py b/delphi/tests/test_narrative_audit.py new file mode 100644 index 0000000000..4f11ef4405 --- /dev/null +++ b/delphi/tests/test_narrative_audit.py @@ -0,0 +1,47 @@ +"""Independent proof auditor catches group-ID confusion and factual drift.""" +import importlib.util +from pathlib import Path +from polismath.utils.vote_convention import SEMANTIC_AGREE, SEMANTIC_DISAGREE + +import pytest + +spec = importlib.util.spec_from_file_location("audit_narrative", Path(__file__).parent / "dynamo_removal/audit_narrative.py") +audit = importlib.util.module_from_spec(spec) +spec.loader.exec_module(audit) + + +def fixture(): + math = {"base-clusters": {"id": [500, 600], "members": [[0], [77]]}, + "group-clusters": [{"id": 5, "members": [500]}, {"id": 9, "members": [600]}], + "group_clusters": [{"id": 5, "members": [0]}, {"id": 9, "members": [77]}]} + context = dict(group_mapping="base-clusters-unfold/1", source_convention="semantic:+1=agree", + group_assignments_sha256=audit.digest({"0": 5, "77": 9}, sort_keys=True, separators=(",", ":")), + math_sha256=audit.digest(math, sort_keys=True, separators=(",", ":"), allow_nan=False), + comments=[{"comment_id": 0, "comment-id": 0, "num_groups": 2, + "votes": 2, "agrees": 1, "disagrees": 1, "passes": 0, + "total-votes": 2, "total-agrees": 1, "total-disagrees": 1, "total-passes": 0, + "group-5-votes": 1, "group-5-agrees": 1, + "group-9-votes": 1, "group-9-disagrees": 1}]) + votes = [dict(tid=0, pid=0, vote=SEMANTIC_AGREE), dict(tid=0, pid=77, vote=SEMANTIC_DISAGREE)] + return context, math, votes + + +def test_canonical_membership_uses_ids_not_array_positions(): + context, math, votes = fixture() + result = audit.audit_context(context, math, votes, [0]) + assert result["group_sizes"] == {"5": 1, "9": 1} + assert result["scoped_counts"] == 3 + + +def test_recount_rejects_reversed_group_direction(): + context, math, votes = fixture() + context["comments"][0]["group-5-agrees"] = 0 + with pytest.raises(AssertionError): + audit.audit_context(context, math, votes, [0]) + + +def test_recount_rejects_different_unfolded_alias(): + context, math, votes = fixture() + math["group_clusters"][0]["members"] = [77] + with pytest.raises(AssertionError): + audit.audit_context(context, math, votes, [0]) diff --git a/delphi/umap_narrative/502_calculate_priorities.py b/delphi/umap_narrative/502_calculate_priorities.py index eb50dbf0d7..3bbad098bd 100755 --- a/delphi/umap_narrative/502_calculate_priorities.py +++ b/delphi/umap_narrative/502_calculate_priorities.py @@ -10,6 +10,7 @@ import argparse import boto3 +from polismath.delphi_storage.resource import result_resource import json import logging import os @@ -51,7 +52,7 @@ def __init__(self, conversation_id: int, endpoint_url: str = None): boto3_kwargs['endpoint_url'] = endpoint_url # Initialize DynamoDB connection using the prepared arguments - self.dynamodb = boto3.resource('dynamodb', **boto3_kwargs) + self.dynamodb = result_resource('dynamodb', **boto3_kwargs) # Get table references self.comment_routing_table = self.dynamodb.Table('Delphi_CommentRouting') diff --git a/delphi/umap_narrative/701_static_datamapplot_for_layer.py b/delphi/umap_narrative/701_static_datamapplot_for_layer.py index 0ffb353cd6..67c2768154 100755 --- a/delphi/umap_narrative/701_static_datamapplot_for_layer.py +++ b/delphi/umap_narrative/701_static_datamapplot_for_layer.py @@ -14,6 +14,7 @@ import numpy as np import json import boto3 +from polismath.delphi_storage.resource import result_resource from boto3.dynamodb.conditions import Key import logging import sys @@ -37,7 +38,7 @@ class DynamoDBStorage: def __init__(self, endpoint_url=None): self.endpoint_url = endpoint_url or os.environ.get("DYNAMODB_ENDPOINT", "http://dynamodb-local:8000") self.region = os.environ.get("AWS_REGION", "us-east-1") - self.dynamodb = boto3.resource('dynamodb', endpoint_url=self.endpoint_url, region_name=self.region) + self.dynamodb = result_resource('dynamodb', endpoint_url=self.endpoint_url, region_name=self.region) # Define table names using the new Delphi_ naming scheme self.table_names = { diff --git a/delphi/umap_narrative/702_consensus_divisive_datamapplot.py b/delphi/umap_narrative/702_consensus_divisive_datamapplot.py index fefba91348..49a4768132 100755 --- a/delphi/umap_narrative/702_consensus_divisive_datamapplot.py +++ b/delphi/umap_narrative/702_consensus_divisive_datamapplot.py @@ -15,6 +15,7 @@ import matplotlib.pyplot as plt import json import boto3 +from polismath.delphi_storage.resource import result_resource import logging import traceback from decimal import Decimal @@ -68,7 +69,7 @@ def __init__(self, endpoint_url=None): else: self.endpoint_url = None self.region = DYNAMODB_CONFIG['region'] - self.dynamodb = boto3.resource('dynamodb', + self.dynamodb = result_resource('dynamodb', endpoint_url=self.endpoint_url, region_name=self.region, aws_access_key_id=DYNAMODB_CONFIG['access_key'], @@ -108,7 +109,7 @@ def load_data_from_dynamodb(zid, layer_num=0): # Set up DynamoDB client endpoint_url = os.environ.get('DYNAMODB_ENDPOINT') - dynamodb = boto3.resource('dynamodb', + dynamodb = result_resource('dynamodb', endpoint_url=endpoint_url, region_name=os.environ.get('AWS_REGION', 'us-east-1'), aws_access_key_id=os.environ.get('AWS_ACCESS_KEY_ID', 'fakeMyKeyId'), diff --git a/delphi/umap_narrative/801_narrative_report_batch.py b/delphi/umap_narrative/801_narrative_report_batch.py index 8adca5e5db..312bcb4cdb 100755 --- a/delphi/umap_narrative/801_narrative_report_batch.py +++ b/delphi/umap_narrative/801_narrative_report_batch.py @@ -27,6 +27,7 @@ import logging import argparse import boto3 +from polismath.delphi_storage.resource import result_resource import asyncio import numpy as np import pandas as pd @@ -66,7 +67,7 @@ def __init__(self, table_name="Delphi_NarrativeReports", dynamodb_resource=None) self.dynamodb = dynamodb_resource else: endpoint_url = os.environ.get('DYNAMODB_ENDPOINT') or None - self.dynamodb = boto3.resource( + self.dynamodb = result_resource( 'dynamodb', endpoint_url=endpoint_url, region_name=os.environ.get('AWS_REGION', 'us-east-1') @@ -145,61 +146,10 @@ def get_report(self, report_id, section, model): logger.error(f"Error getting report: {str(e)}") return None -class PolisConverter: - """Convert between CSV and XML formats for Polis data.""" - - @staticmethod - def convert_to_xml(comment_data): - """ - Convert comment data to XML format. - - Args: - comment_data: List of dictionaries with comment data - - Returns: - String with XML representation of the comment data - """ - # Create root element - root = ET.Element("polis-comments") - - # Process each comment - for record in comment_data: - # Extract base comment data - comment = ET.SubElement(root, "comment", { - "id": str(record.get("comment-id", "")), - "votes": str(record.get("total-votes", 0)), - "agrees": str(record.get("total-agrees", 0)), - "disagrees": str(record.get("total-disagrees", 0)), - "passes": str(record.get("total-passes", 0)), - }) - - # Add comment text - text = ET.SubElement(comment, "text") - text.text = record.get("comment", "") - - # Process group data - group_keys = [] - for key in record.keys(): - if key.startswith("group-") and key.count("-") >= 2: - group_id = key.split("-")[1] - if group_id not in group_keys: - group_keys.append(group_id) - - # Add data for each group - for group_id in group_keys: - group = ET.SubElement(comment, f"group-{group_id}", { - "votes": str(record.get(f"group-{group_id}-votes", 0)), - "agrees": str(record.get(f"group-{group_id}-agrees", 0)), - "disagrees": str(record.get(f"group-{group_id}-disagrees", 0)), - "passes": str(record.get(f"group-{group_id}-passes", 0)), - }) - - # Convert to string with pretty formatting - rough_string = ET.tostring(root, 'utf-8') - reparsed = parseString(rough_string) - return reparsed.toprettyxml(indent=" ") +from umap_narrative.narrative_data import PolisConverter, NarrativeSelection + -class BatchReportGenerator: +class BatchReportGenerator(NarrativeSelection): """Generate batch reports for Polis conversations.""" def __init__(self, conversation_id, model=None, no_cache=False, max_batch_size=20, job_id=None, layers=None, include_moderation=False, exclude_comment_selections=True): @@ -229,7 +179,7 @@ def __init__(self, conversation_id, model=None, no_cache=False, max_batch_size=2 logger.info(f"exclude_comment_selections: {exclude_comment_selections}") endpoint_url = os.environ.get('DYNAMODB_ENDPOINT') or None - self.dynamodb = boto3.resource( + self.dynamodb = result_resource( 'dynamodb', endpoint_url=endpoint_url, region_name=os.environ.get('AWS_REGION', 'us-east-1') @@ -572,328 +522,6 @@ async def get_topics(self): logger.error(f"A critical error occurred in get_topics: {str(e)}", exc_info=True) return [] - def filter_topics(self, comment, topic_cluster_id=None, topic_layer_id=None, topic_citations=None, sample_comments=None, filter_type=None, filter_threshold=None): - """Filter for comments that are part of a specific topic or meet global section criteria.""" - # Get comment ID - comment_id = comment.get('comment_id') - if not comment_id: - return False - - # Handle global section filtering - if filter_type is not None: - return self._apply_global_filter(comment, filter_type, filter_threshold) - - # Handle layer-specific topic filtering (existing logic) - if topic_cluster_id is not None and topic_layer_id is not None: - # Get the cluster ID for the specified layer - layer_cluster_key = f'layer{topic_layer_id}_cluster_id' - comment_cluster_id = comment.get(layer_cluster_key) - if comment_cluster_id is not None: - # Debug logging for cluster 0 - if str(topic_cluster_id) == "0" and comment_id in [1, 2, 3]: # Log first few comments - logger.info(f"DEBUG: Checking comment {comment_id} - layer{topic_layer_id}_cluster_id={comment_cluster_id}, topic_cluster_id={topic_cluster_id}") - logger.info(f"DEBUG: String comparison: '{str(comment_cluster_id)}' == '{str(topic_cluster_id)}' = {str(comment_cluster_id) == str(topic_cluster_id)}") - - # Simple string comparison is more reliable across different numeric types - if str(comment_cluster_id) == str(topic_cluster_id): - return True - - # Check if this comment ID is in our topic citations - if topic_citations and str(comment_id) in [str(c) for c in topic_citations]: - return True - - # If we have sample comments and not enough filtered comments, - # try to match based on text similarity - if sample_comments and len(sample_comments) > 0: - comment_text = comment.get('comment', '') - if not comment_text: - return False - - # Check if this comment text matches any sample comment - for sample in sample_comments: - # Skip non-string samples - if not isinstance(sample, str) or not sample: - continue - - # Simple substring match rather than complex word comparison - if sample.lower() in comment_text.lower() or comment_text.lower() in sample.lower(): - return True - - return False - - def _apply_global_filter(self, comment, filter_type, filter_threshold): - """ - Apply global section filtering based on Polis statistical metrics. - - Args: - comment: Comment data dictionary - filter_type: Type of filter ('comment_extremity', 'group_aware_consensus', 'uncertainty_ratio') - filter_threshold: Threshold value for filtering (or 'dynamic' for group_aware_consensus) - - Returns: - Boolean indicating whether comment passes the filter - """ - try: - if filter_type == "comment_extremity": - # Filter for comments that divide opinion groups (extremity > 1.0) - extremity = comment.get('comment_extremity', 0) - return extremity > filter_threshold - - elif filter_type == "group_aware_consensus": - # Filter for comments with broad cross-group agreement - # Uses dynamic thresholds based on number of groups - consensus = comment.get('group_aware_consensus', 0) - num_groups = comment.get('num_groups', 2) - - # Get dynamic threshold based on group count (matches Node.js logic) - if filter_threshold == "dynamic": - if num_groups == 2: - threshold = 0.7 - elif num_groups == 3: - threshold = 0.47 - elif num_groups == 4: - threshold = 0.32 - else: # 5+ groups - threshold = 0.24 - else: - threshold = filter_threshold - - return consensus > threshold - - elif filter_type == "uncertainty_ratio": - # Filter for comments with high uncertainty/unsure responses (>= 20% pass votes) - passes = comment.get('passes', 0) - votes = comment.get('votes', 0) - - if votes == 0: - return False - - uncertainty_ratio = passes / votes - return uncertainty_ratio >= filter_threshold - - else: - logger.warning(f"Unknown filter type: {filter_type}") - return False - - except Exception as e: - logger.error(f"Error applying global filter {filter_type}: {str(e)}") - return False - - def _get_dynamic_comment_limit(self, layer_id=None, total_layers=None, comment_count=None, filter_type=None): - """ - Calculate dynamic comment limit based on layer granularity and conversation size. - Implements the fractal approach where coarse layers get fewer, higher quality comments. - - Args: - layer_id: Current layer ID (None for global sections) - total_layers: Total number of available layers - comment_count: Total number of comments in conversation - filter_type: Type of filter (for global sections) - - Returns: - Integer comment limit for this section - """ - try: - # Base limits for different categories - base_limits = { - "global_sections": 50, # Fixed limit for global sections - "fine_layers": 100, # More comments for specific topics (layer 0) - "medium_layers": 75, # Balanced approach (middle layers) - "coarse_layers": 50 # Fewer, highest quality comments (top layer) - } - - # Determine category - if filter_type is not None: - # This is a global section - category = "global_sections" - elif layer_id is not None and total_layers is not None: - # This is a layer-specific topic - if layer_id == 0: - category = "fine_layers" # Most specific layer - elif layer_id == total_layers - 1: - category = "coarse_layers" # Most general layer - else: - category = "medium_layers" # Middle layers - else: - # Fallback to medium limit - category = "medium_layers" - - # Get base limit - limit = base_limits[category] - - # Scale down for very large conversations to manage token usage - if comment_count is not None: - if comment_count > 10000: - # Halve limits for huge conversations (>10k comments) - limit = int(limit * 0.5) - elif comment_count > 5000: - # Reduce by 25% for large conversations (5k-10k comments) - limit = int(limit * 0.75) - elif comment_count > 2000: - # Reduce by 10% for medium-large conversations (2k-5k comments) - limit = int(limit * 0.9) - - # Ensure minimum limit - limit = max(limit, 10) - - logger.debug(f"Dynamic comment limit: category={category}, base={base_limits[category]}, " - f"final={limit}, comment_count={comment_count}, layer_id={layer_id}") - - return limit - - except Exception as e: - logger.error(f"Error calculating dynamic comment limit: {str(e)}") - # Fallback to conservative limit - return 50 - - def _select_high_quality_comments(self, comments, limit, filter_type=None): - """ - Select the highest quality comments based on Polis statistical metrics. - - Args: - comments: List of comment dictionaries - limit: Maximum number of comments to select - filter_type: Type of filter being applied (affects sorting priority) - - Returns: - List of selected high-quality comments - """ - if len(comments) <= limit: - return comments - - try: - # Create sorting key based on filter type and available metrics - def get_sort_key(comment): - # Base score starts with vote count (engagement indicator) - votes = comment.get('votes', 0) - vote_score = int(votes) if isinstance(votes, (int, float)) else 0 - - # Add metric-specific scoring - if filter_type == "comment_extremity": - # For extremity filtering, prioritize highly divisive comments - extremity = comment.get('comment_extremity', 0) - metric_score = extremity * 1000 # Scale up for sorting - elif filter_type == "group_aware_consensus": - # For consensus filtering, prioritize high agreement comments - consensus = comment.get('group_aware_consensus', 0) - metric_score = consensus * 1000 # Scale up for sorting - elif filter_type == "uncertainty_ratio": - # For uncertainty filtering, prioritize comments with high pass rates - passes = comment.get('passes', 0) - total_votes = comment.get('votes', 1) - uncertainty = passes / max(total_votes, 1) - metric_score = uncertainty * 1000 # Scale up for sorting - else: - # For topic filtering, use a combination of votes and engagement - agrees = comment.get('agrees', 0) - disagrees = comment.get('disagrees', 0) - total_engagement = int(agrees) + int(disagrees) if isinstance(agrees, (int, float)) and isinstance(disagrees, (int, float)) else 0 - metric_score = total_engagement - - # Combine scores (metric score is primary, vote count is secondary) - return (metric_score, vote_score) - - # Sort comments by quality score (descending) - sorted_comments = sorted(comments, key=get_sort_key, reverse=True) - - # Select top comments up to limit - selected = sorted_comments[:limit] - - logger.info(f"Selected {len(selected)} high-quality comments from {len(comments)} " - f"(filter_type={filter_type}, limit={limit})") - - return selected - - except Exception as e: - logger.error(f"Error selecting high-quality comments: {str(e)}") - # Fallback to simple vote-based selection - try: - sorted_comments = sorted(comments, - key=lambda c: int(c.get('votes', 0)) if isinstance(c.get('votes'), (int, float)) else 0, - reverse=True) - return sorted_comments[:limit] - except Exception: - # Last resort: return first N comments - return comments[:limit] - - async def get_comments_as_xml(self, conversation_data: dict, filter_func=None, filter_args=None): - """Get comments as XML from pre-fetched data.""" - try: - # Use the data passed as an argument - data = conversation_data - - if not data: - logger.error("Received empty conversation data.") - return "" - - # Apply filter if provided - filtered_comments = data["processed_comments"] - - if filter_func: - if filter_args: - filtered_comments = [c for c in filtered_comments if filter_func(c, **filter_args)] - else: - filtered_comments = [c for c in filtered_comments if filter_func(c)] - - # Apply dynamic comment limiting with intelligent selection - if filter_func == self.filter_topics and len(filtered_comments) > 0: - # Get context for dynamic limit calculation - total_comment_count = len(data["processed_comments"]) - - # Extract layer and filter information from filter_args - layer_id = None - total_layers = None - filter_type = None - - if filter_args: - layer_id = filter_args.get('topic_layer_id') - filter_type = filter_args.get('filter_type') - - # Estimate total layers from conversation data (could be improved) - # For now, we'll determine this dynamically or use a reasonable default - if layer_id is not None: - # Try to determine total layers from available cluster data - # This is a heuristic - in practice you might want to pass this explicitly - total_layers = max(layer_id + 1, 3) # Assume at least 3 layers if we have layer data - - # Calculate dynamic limit - comment_limit = self._get_dynamic_comment_limit( - layer_id=layer_id, - total_layers=total_layers, - comment_count=total_comment_count, - filter_type=filter_type - ) - - # Apply intelligent comment selection if we exceed the limit - if len(filtered_comments) > comment_limit: - logger.info(f"Applying dynamic comment limit: {len(filtered_comments)} -> {comment_limit} " - f"(layer_id={layer_id}, filter_type={filter_type}, total_comments={total_comment_count})") - - # Use intelligent selection based on Polis metrics - filtered_comments = self._select_high_quality_comments( - filtered_comments, - comment_limit, - filter_type=filter_type - ) - else: - logger.info(f"No limiting needed: {len(filtered_comments)} comments <= limit of {comment_limit}") - else: - # For non-topic filtering, use a conservative limit to avoid token issues - max_comments = 100 - if len(filtered_comments) > max_comments: - logger.info(f"Applying conservative limit: {len(filtered_comments)} -> {max_comments}") - filtered_comments = self._select_high_quality_comments(filtered_comments, max_comments) - - # Convert to XML - xml = PolisConverter.convert_to_xml(filtered_comments) - - return xml - except Exception as e: - logger.error(f"Error in get_comments_as_xml: {str(e)}") - import traceback - logger.error(traceback.format_exc()) - return "" - async def prepare_batch_requests(self): """Prepare batch requests for all topics.""" logger.info("Fetching all conversation data ONCE...") diff --git a/delphi/umap_narrative/803_check_batch_status.py b/delphi/umap_narrative/803_check_batch_status.py index 5759bff9b9..94b66ee660 100755 --- a/delphi/umap_narrative/803_check_batch_status.py +++ b/delphi/umap_narrative/803_check_batch_status.py @@ -11,6 +11,7 @@ """ import os, sys, json, boto3, logging, argparse, asyncio +from polismath.delphi_storage.resource import result_resource from typing import Dict, Optional from datetime import datetime, timedelta, timezone from botocore.exceptions import ClientError @@ -46,7 +47,7 @@ def __init__(self): raw_endpoint = os.environ.get('DYNAMODB_ENDPOINT') endpoint_url = raw_endpoint if raw_endpoint and raw_endpoint.strip() else None - self.dynamodb = boto3.resource('dynamodb', endpoint_url=endpoint_url, region_name=os.environ.get('AWS_REGION', 'us-east-1')) + self.dynamodb = result_resource('dynamodb', endpoint_url=endpoint_url, region_name=os.environ.get('AWS_REGION', 'us-east-1')) self.job_table = self.dynamodb.Table('Delphi_JobQueue') self.report_table = self.dynamodb.Table('Delphi_NarrativeReports') # Provider token usage summed over the stored results, only when the diff --git a/delphi/umap_narrative/narrative_data.py b/delphi/umap_narrative/narrative_data.py new file mode 100644 index 0000000000..3e1129f96e --- /dev/null +++ b/delphi/umap_narrative/narrative_data.py @@ -0,0 +1,383 @@ +"""Pure narrative selection and XML formatting shared with legacy batch reports.""" +import logging +import xml.etree.ElementTree as ET +from xml.dom.minidom import parseString +logger = logging.getLogger(__name__) + +class PolisConverter: + """Convert between CSV and XML formats for Polis data.""" + + @staticmethod + def convert_to_xml(comment_data): + """ + Convert comment data to XML format. + + Args: + comment_data: List of dictionaries with comment data + + Returns: + String with XML representation of the comment data + """ + # Create root element + root = ET.Element("polis-comments") + + # Process each comment + for record in comment_data: + # Extract base comment data + comment = ET.SubElement(root, "comment", { + "id": str(record.get("comment-id", "")), + "votes": str(record.get("total-votes", 0)), + "agrees": str(record.get("total-agrees", 0)), + "disagrees": str(record.get("total-disagrees", 0)), + "passes": str(record.get("total-passes", 0)), + }) + + # Add comment text + text = ET.SubElement(comment, "text") + text.text = record.get("comment", "") + + # Process group data + group_keys = [] + for key in record.keys(): + if key.startswith("group-") and key.count("-") >= 2: + group_id = key.split("-")[1] + if group_id not in group_keys: + group_keys.append(group_id) + + # Add data for each group + for group_id in group_keys: + group = ET.SubElement(comment, f"group-{group_id}", { + "votes": str(record.get(f"group-{group_id}-votes", 0)), + "agrees": str(record.get(f"group-{group_id}-agrees", 0)), + "disagrees": str(record.get(f"group-{group_id}-disagrees", 0)), + "passes": str(record.get(f"group-{group_id}-passes", 0)), + }) + + # Convert to string with pretty formatting + rough_string = ET.tostring(root, 'utf-8') + reparsed = parseString(rough_string) + return reparsed.toprettyxml(indent=" ") + +class NarrativeSelection: + def filter_topics(self, comment, topic_cluster_id=None, topic_layer_id=None, topic_citations=None, sample_comments=None, filter_type=None, filter_threshold=None): + """Filter for comments that are part of a specific topic or meet global section criteria.""" + # Get comment ID + comment_id = comment.get('comment_id') + if comment_id is None: + return False + + # Handle global section filtering + if filter_type is not None: + return self._apply_global_filter(comment, filter_type, filter_threshold) + + # Handle layer-specific topic filtering (existing logic) + if topic_cluster_id is not None and topic_layer_id is not None: + # Get the cluster ID for the specified layer + layer_cluster_key = f'layer{topic_layer_id}_cluster_id' + comment_cluster_id = comment.get(layer_cluster_key) + if comment_cluster_id is not None: + # Debug logging for cluster 0 + if str(topic_cluster_id) == "0" and comment_id in [1, 2, 3]: # Log first few comments + logger.info(f"DEBUG: Checking comment {comment_id} - layer{topic_layer_id}_cluster_id={comment_cluster_id}, topic_cluster_id={topic_cluster_id}") + logger.info(f"DEBUG: String comparison: '{str(comment_cluster_id)}' == '{str(topic_cluster_id)}' = {str(comment_cluster_id) == str(topic_cluster_id)}") + + # Simple string comparison is more reliable across different numeric types + if str(comment_cluster_id) == str(topic_cluster_id): + return True + + # Check if this comment ID is in our topic citations + if topic_citations and str(comment_id) in [str(c) for c in topic_citations]: + return True + + # If we have sample comments and not enough filtered comments, + # try to match based on text similarity + if sample_comments and len(sample_comments) > 0: + comment_text = comment.get('comment', '') + if not comment_text: + return False + + # Check if this comment text matches any sample comment + for sample in sample_comments: + # Skip non-string samples + if not isinstance(sample, str) or not sample: + continue + + # Simple substring match rather than complex word comparison + if sample.lower() in comment_text.lower() or comment_text.lower() in sample.lower(): + return True + + return False + + def _apply_global_filter(self, comment, filter_type, filter_threshold): + """ + Apply global section filtering based on Polis statistical metrics. + + Args: + comment: Comment data dictionary + filter_type: Type of filter ('comment_extremity', 'group_aware_consensus', 'uncertainty_ratio') + filter_threshold: Threshold value for filtering (or 'dynamic' for group_aware_consensus) + + Returns: + Boolean indicating whether comment passes the filter + """ + try: + if filter_type == "comment_extremity": + # Filter for comments that divide opinion groups (extremity > 1.0) + extremity = comment.get('comment_extremity', 0) + return extremity > filter_threshold + + elif filter_type == "group_aware_consensus": + # Filter for comments with broad cross-group agreement + # Uses dynamic thresholds based on number of groups + consensus = comment.get('group_aware_consensus', 0) + num_groups = comment.get('num_groups', 2) + + # Get dynamic threshold based on group count (matches Node.js logic) + if filter_threshold == "dynamic": + if num_groups == 2: + threshold = 0.7 + elif num_groups == 3: + threshold = 0.47 + elif num_groups == 4: + threshold = 0.32 + else: # 5+ groups + threshold = 0.24 + else: + threshold = filter_threshold + + return consensus > threshold + + elif filter_type == "uncertainty_ratio": + # Filter for comments with high uncertainty/unsure responses (>= 20% pass votes) + passes = comment.get('passes', 0) + votes = comment.get('votes', 0) + + if votes == 0: + return False + + uncertainty_ratio = passes / votes + return uncertainty_ratio >= filter_threshold + + else: + logger.warning(f"Unknown filter type: {filter_type}") + return False + + except Exception as e: + logger.error(f"Error applying global filter {filter_type}: {str(e)}") + return False + + def _get_dynamic_comment_limit(self, layer_id=None, total_layers=None, comment_count=None, filter_type=None): + """ + Calculate dynamic comment limit based on layer granularity and conversation size. + Implements the fractal approach where coarse layers get fewer, higher quality comments. + + Args: + layer_id: Current layer ID (None for global sections) + total_layers: Total number of available layers + comment_count: Total number of comments in conversation + filter_type: Type of filter (for global sections) + + Returns: + Integer comment limit for this section + """ + try: + # Base limits for different categories + base_limits = { + "global_sections": 50, # Fixed limit for global sections + "fine_layers": 100, # More comments for specific topics (layer 0) + "medium_layers": 75, # Balanced approach (middle layers) + "coarse_layers": 50 # Fewer, highest quality comments (top layer) + } + + # Determine category + if filter_type is not None: + # This is a global section + category = "global_sections" + elif layer_id is not None and total_layers is not None: + # This is a layer-specific topic + if layer_id == 0: + category = "fine_layers" # Most specific layer + elif layer_id == total_layers - 1: + category = "coarse_layers" # Most general layer + else: + category = "medium_layers" # Middle layers + else: + # Fallback to medium limit + category = "medium_layers" + + # Get base limit + limit = base_limits[category] + + # Scale down for very large conversations to manage token usage + if comment_count is not None: + if comment_count > 10000: + # Halve limits for huge conversations (>10k comments) + limit = int(limit * 0.5) + elif comment_count > 5000: + # Reduce by 25% for large conversations (5k-10k comments) + limit = int(limit * 0.75) + elif comment_count > 2000: + # Reduce by 10% for medium-large conversations (2k-5k comments) + limit = int(limit * 0.9) + + # Ensure minimum limit + limit = max(limit, 10) + + logger.debug(f"Dynamic comment limit: category={category}, base={base_limits[category]}, " + f"final={limit}, comment_count={comment_count}, layer_id={layer_id}") + + return limit + + except Exception as e: + logger.error(f"Error calculating dynamic comment limit: {str(e)}") + # Fallback to conservative limit + return 50 + + def _select_high_quality_comments(self, comments, limit, filter_type=None): + """ + Select the highest quality comments based on Polis statistical metrics. + + Args: + comments: List of comment dictionaries + limit: Maximum number of comments to select + filter_type: Type of filter being applied (affects sorting priority) + + Returns: + List of selected high-quality comments + """ + if len(comments) <= limit: + return comments + + try: + # Create sorting key based on filter type and available metrics + def get_sort_key(comment): + # Base score starts with vote count (engagement indicator) + votes = comment.get('votes', 0) + vote_score = int(votes) if isinstance(votes, (int, float)) else 0 + + # Add metric-specific scoring + if filter_type == "comment_extremity": + # For extremity filtering, prioritize highly divisive comments + extremity = comment.get('comment_extremity', 0) + metric_score = extremity * 1000 # Scale up for sorting + elif filter_type == "group_aware_consensus": + # For consensus filtering, prioritize high agreement comments + consensus = comment.get('group_aware_consensus', 0) + metric_score = consensus * 1000 # Scale up for sorting + elif filter_type == "uncertainty_ratio": + # For uncertainty filtering, prioritize comments with high pass rates + passes = comment.get('passes', 0) + total_votes = comment.get('votes', 1) + uncertainty = passes / max(total_votes, 1) + metric_score = uncertainty * 1000 # Scale up for sorting + else: + # For topic filtering, use a combination of votes and engagement + agrees = comment.get('agrees', 0) + disagrees = comment.get('disagrees', 0) + total_engagement = int(agrees) + int(disagrees) if isinstance(agrees, (int, float)) and isinstance(disagrees, (int, float)) else 0 + metric_score = total_engagement + + # Combine scores (metric score is primary, vote count is secondary) + return (metric_score, vote_score) + + # Sort comments by quality score (descending) + sorted_comments = sorted(comments, key=get_sort_key, reverse=True) + + # Select top comments up to limit + selected = sorted_comments[:limit] + + logger.info(f"Selected {len(selected)} high-quality comments from {len(comments)} " + f"(filter_type={filter_type}, limit={limit})") + + return selected + + except Exception as e: + logger.error(f"Error selecting high-quality comments: {str(e)}") + # Fallback to simple vote-based selection + try: + sorted_comments = sorted(comments, + key=lambda c: int(c.get('votes', 0)) if isinstance(c.get('votes'), (int, float)) else 0, + reverse=True) + return sorted_comments[:limit] + except Exception: + # Last resort: return first N comments + return comments[:limit] + + async def get_comments_as_xml(self, conversation_data: dict, filter_func=None, filter_args=None): + """Get comments as XML from pre-fetched data.""" + try: + # Use the data passed as an argument + data = conversation_data + + if not data: + logger.error("Received empty conversation data.") + return "" + + # Apply filter if provided + filtered_comments = data["processed_comments"] + + if filter_func: + if filter_args: + filtered_comments = [c for c in filtered_comments if filter_func(c, **filter_args)] + else: + filtered_comments = [c for c in filtered_comments if filter_func(c)] + + # Apply dynamic comment limiting with intelligent selection + if filter_func == self.filter_topics and len(filtered_comments) > 0: + # Get context for dynamic limit calculation + total_comment_count = len(data["processed_comments"]) + + # Extract layer and filter information from filter_args + layer_id = None + total_layers = None + filter_type = None + + if filter_args: + layer_id = filter_args.get('topic_layer_id') + filter_type = filter_args.get('filter_type') + + # Estimate total layers from conversation data (could be improved) + # For now, we'll determine this dynamically or use a reasonable default + if layer_id is not None: + # Try to determine total layers from available cluster data + # This is a heuristic - in practice you might want to pass this explicitly + total_layers = max(layer_id + 1, 3) # Assume at least 3 layers if we have layer data + + # Calculate dynamic limit + comment_limit = self._get_dynamic_comment_limit( + layer_id=layer_id, + total_layers=total_layers, + comment_count=total_comment_count, + filter_type=filter_type + ) + + # Apply intelligent comment selection if we exceed the limit + if len(filtered_comments) > comment_limit: + logger.info(f"Applying dynamic comment limit: {len(filtered_comments)} -> {comment_limit} " + f"(layer_id={layer_id}, filter_type={filter_type}, total_comments={total_comment_count})") + + # Use intelligent selection based on Polis metrics + filtered_comments = self._select_high_quality_comments( + filtered_comments, + comment_limit, + filter_type=filter_type + ) + else: + logger.info(f"No limiting needed: {len(filtered_comments)} comments <= limit of {comment_limit}") + else: + # For non-topic filtering, use a conservative limit to avoid token issues + max_comments = 100 + if len(filtered_comments) > max_comments: + logger.info(f"Applying conservative limit: {len(filtered_comments)} -> {max_comments}") + filtered_comments = self._select_high_quality_comments(filtered_comments, max_comments) + + # Convert to XML + xml = PolisConverter.convert_to_xml(filtered_comments) + + return xml + except Exception as e: + logger.error(f"Error in get_comments_as_xml: {str(e)}") + import traceback + logger.error(traceback.format_exc()) + return "" + diff --git a/delphi/umap_narrative/numerical_stages.py b/delphi/umap_narrative/numerical_stages.py new file mode 100644 index 0000000000..44be4f0fa2 --- /dev/null +++ b/delphi/umap_narrative/numerical_stages.py @@ -0,0 +1,118 @@ +"""Pure numerical stages shared by legacy CLI and independently queued Delphi jobs. + +Extracted without changing parameters, EVOC fallback, or corpus TF-IDF semantics. +No storage clients or provider imports. +""" +import logging +import traceback +import numpy as np +import evoc +from umap import UMAP +from sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer +logger = logging.getLogger(__name__) + + +def project_and_cluster(document_vectors): + # Generate 2D projection with UMAP + logger.info("Generating 2D projection with UMAP...") + document_map = UMAP(n_components=2, metric="cosine", random_state=42).fit_transform( + document_vectors + ) + + # Cluster with EVōC + logger.info("Clustering with EVōC...") + try: + clusterer = evoc.EVoC(min_samples=5) # Set min_samples to avoid empty clusters + cluster_labels = clusterer.fit_predict(document_vectors) + cluster_layers = clusterer.cluster_layers_ + + logger.info( + f"Found {len(np.unique(cluster_labels))} clusters at the finest level" + ) + for i, layer in enumerate(cluster_layers): + unique_clusters = np.unique(layer[layer >= 0]) + logger.info(f"Layer {i}: {len(unique_clusters)} clusters") + + except Exception as e: + logger.error(f"Error during EVōC clustering: {e}") + logger.error(traceback.format_exc()) + # Fallback to simple clustering + from sklearn.cluster import KMeans + + logger.info("Falling back to KMeans clustering...") + kmeans = KMeans(n_clusters=5, random_state=42) + cluster_labels = kmeans.fit_predict(document_vectors) + + # Create a simple layered clustering for demonstration + from sklearn.cluster import AgglomerativeClustering + + layer1 = AgglomerativeClustering(n_clusters=3).fit_predict(document_vectors) + layer2 = AgglomerativeClustering(n_clusters=2).fit_predict(document_vectors) + + cluster_layers = [cluster_labels, layer1, layer2] + logger.info( + f"Created {len(cluster_layers)} cluster layers with fallback clustering" + ) + + return document_map, cluster_layers + + +def characterize_comment_clusters(cluster_layer, comment_texts): + """ + Characterize comment clusters by common themes and keywords. + + Args: + cluster_layer: Cluster assignments for a specific layer + comment_texts: List of comment text strings + + Returns: + cluster_characteristics: Dictionary with cluster characterizations + """ + # Create a dictionary to store cluster characteristics + cluster_characteristics = {} + + # Get unique clusters + unique_clusters = np.unique(cluster_layer) + unique_clusters = unique_clusters[unique_clusters >= 0] # Remove noise points (-1) + + # Create TF-IDF vectorizer + vectorizer = CountVectorizer(max_features=1000, stop_words="english") + transformer = TfidfTransformer() + + # Fit and transform the entire corpus + X = vectorizer.fit_transform(comment_texts) + X_tfidf = transformer.fit_transform(X) + + # Get feature names + feature_names = vectorizer.get_feature_names_out() + + for cluster_id in unique_clusters: + # Get cluster members + cluster_members = np.where(cluster_layer == cluster_id)[0] + + if len(cluster_members) == 0: + continue + + # Get comment texts for this cluster + cluster_comments = [comment_texts[i] for i in cluster_members] + + # Find top words for this cluster by TF-IDF + cluster_tfidf = X_tfidf[cluster_members].toarray().mean(axis=0) + top_indices = np.argsort(cluster_tfidf)[-10:][::-1] # Top 10 words + top_words = [feature_names[i] for i in top_indices] + + # Get sample comments (shortest 3 for readability) + comment_lengths = [len(comment) for comment in cluster_comments] + shortest_indices = np.argsort(comment_lengths)[:3] # 3 shortest comments + sample_comments = [cluster_comments[i] for i in shortest_indices] + + # Add to cluster characteristics + cluster_characteristics[int(cluster_id)] = { + "size": len(cluster_members), + "top_words": top_words, + "top_tfidf_scores": [float(cluster_tfidf[i]) for i in top_indices], + "sample_comments": sample_comments, + } + + return cluster_characteristics + diff --git a/delphi/umap_narrative/polismath_commentgraph/utils/__init__.py b/delphi/umap_narrative/polismath_commentgraph/utils/__init__.py index 19de7a2273..c909bbb546 100644 --- a/delphi/umap_narrative/polismath_commentgraph/utils/__init__.py +++ b/delphi/umap_narrative/polismath_commentgraph/utils/__init__.py @@ -1,11 +1,13 @@ -""" -Utility functions for the Polis comment graph microservice. -""" +"""Utilities; numerical converters do not initialize storage dependencies.""" -from .storage import DynamoDBStorage -from .converter import DataConverter +__all__ = ['DynamoDBStorage', 'DataConverter'] -__all__ = [ - 'DynamoDBStorage', - 'DataConverter' -] \ No newline at end of file + +def __getattr__(name): + if name == 'DataConverter': + from .converter import DataConverter + return DataConverter + if name == 'DynamoDBStorage': + from .storage import DynamoDBStorage + return DynamoDBStorage + raise AttributeError(name) diff --git a/delphi/umap_narrative/polismath_commentgraph/utils/group_data.py b/delphi/umap_narrative/polismath_commentgraph/utils/group_data.py index 1343908f53..4865ef66aa 100644 --- a/delphi/umap_narrative/polismath_commentgraph/utils/group_data.py +++ b/delphi/umap_narrative/polismath_commentgraph/utils/group_data.py @@ -7,6 +7,7 @@ import json import logging import boto3 +from polismath.delphi_storage.resource import result_resource import os from typing import Dict, List, Any, Optional from collections import defaultdict @@ -48,7 +49,7 @@ def init_dynamodb(self): # Set up DynamoDB client WITHOUT explicit credentials. # Boto3 will use its default credential provider chain (env vars -> IAM role). - self.dynamodb = boto3.resource( + self.dynamodb = result_resource( 'dynamodb', endpoint_url=endpoint_url, region_name=region diff --git a/delphi/umap_narrative/polismath_commentgraph/utils/storage.py b/delphi/umap_narrative/polismath_commentgraph/utils/storage.py index b8dd49339c..e488de58b0 100644 --- a/delphi/umap_narrative/polismath_commentgraph/utils/storage.py +++ b/delphi/umap_narrative/polismath_commentgraph/utils/storage.py @@ -3,6 +3,7 @@ """ import boto3 +from polismath.delphi_storage.resource import result_resource import os import json import logging @@ -447,7 +448,7 @@ def __init__(self, region_name: str = None, endpoint_url: str = None): kwargs['aws_secret_access_key'] = aws_secret_access_key # Create the DynamoDB resource - self.dynamodb = boto3.resource('dynamodb', **kwargs) + self.dynamodb = result_resource('dynamodb', **kwargs) # Define table names self.table_names = { diff --git a/delphi/umap_narrative/reset_conversation.py b/delphi/umap_narrative/reset_conversation.py index 5a8f291fbd..aff6c309e6 100644 --- a/delphi/umap_narrative/reset_conversation.py +++ b/delphi/umap_narrative/reset_conversation.py @@ -36,6 +36,9 @@ def get_boto_resource(service_name: str): else: logger.info(f"AWS environment detected for {service_name}. Using IAM role credentials.") + if os.environ.get('DELPHI_RESULT_BACKEND') == 'postgres': + from polismath.delphi_storage.resource import result_resource + return result_resource(service_name, **resource_args) return boto3.resource(service_name, **resource_args) diff --git a/delphi/umap_narrative/run_pipeline.py b/delphi/umap_narrative/run_pipeline.py index e7f7e5e060..497e3cb2e9 100755 --- a/delphi/umap_narrative/run_pipeline.py +++ b/delphi/umap_narrative/run_pipeline.py @@ -33,6 +33,7 @@ ) from sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer from umap import UMAP +from umap_narrative.numerical_stages import characterize_comment_clusters # Configure logging logging.basicConfig( @@ -179,109 +180,12 @@ def process_comments(comments, conversation_id): embedding_model = SentenceTransformer(model_name) document_vectors = embedding_model.encode(comment_texts, show_progress_bar=True) - # Generate 2D projection with UMAP - logger.info("Generating 2D projection with UMAP...") - document_map = UMAP(n_components=2, metric="cosine", random_state=42).fit_transform( - document_vectors - ) - - # Cluster with EVōC - logger.info("Clustering with EVōC...") - try: - clusterer = evoc.EVoC(min_samples=5) # Set min_samples to avoid empty clusters - cluster_labels = clusterer.fit_predict(document_vectors) - cluster_layers = clusterer.cluster_layers_ - - logger.info( - f"Found {len(np.unique(cluster_labels))} clusters at the finest level" - ) - for i, layer in enumerate(cluster_layers): - unique_clusters = np.unique(layer[layer >= 0]) - logger.info(f"Layer {i}: {len(unique_clusters)} clusters") - - except Exception as e: - logger.error(f"Error during EVōC clustering: {e}") - logger.error(traceback.format_exc()) - # Fallback to simple clustering - from sklearn.cluster import KMeans - - logger.info("Falling back to KMeans clustering...") - kmeans = KMeans(n_clusters=5, random_state=42) - cluster_labels = kmeans.fit_predict(document_vectors) - - # Create a simple layered clustering for demonstration - from sklearn.cluster import AgglomerativeClustering - - layer1 = AgglomerativeClustering(n_clusters=3).fit_predict(document_vectors) - layer2 = AgglomerativeClustering(n_clusters=2).fit_predict(document_vectors) - - cluster_layers = [cluster_labels, layer1, layer2] - logger.info( - f"Created {len(cluster_layers)} cluster layers with fallback clustering" - ) + from umap_narrative.numerical_stages import project_and_cluster + document_map, cluster_layers = project_and_cluster(document_vectors) return document_map, document_vectors, cluster_layers, comment_texts, comment_ids -def characterize_comment_clusters(cluster_layer, comment_texts): - """ - Characterize comment clusters by common themes and keywords. - - Args: - cluster_layer: Cluster assignments for a specific layer - comment_texts: List of comment text strings - - Returns: - cluster_characteristics: Dictionary with cluster characterizations - """ - # Create a dictionary to store cluster characteristics - cluster_characteristics = {} - - # Get unique clusters - unique_clusters = np.unique(cluster_layer) - unique_clusters = unique_clusters[unique_clusters >= 0] # Remove noise points (-1) - - # Create TF-IDF vectorizer - vectorizer = CountVectorizer(max_features=1000, stop_words="english") - transformer = TfidfTransformer() - - # Fit and transform the entire corpus - X = vectorizer.fit_transform(comment_texts) - X_tfidf = transformer.fit_transform(X) - - # Get feature names - feature_names = vectorizer.get_feature_names_out() - - for cluster_id in unique_clusters: - # Get cluster members - cluster_members = np.where(cluster_layer == cluster_id)[0] - - if len(cluster_members) == 0: - continue - - # Get comment texts for this cluster - cluster_comments = [comment_texts[i] for i in cluster_members] - - # Find top words for this cluster by TF-IDF - cluster_tfidf = X_tfidf[cluster_members].toarray().mean(axis=0) - top_indices = np.argsort(cluster_tfidf)[-10:][::-1] # Top 10 words - top_words = [feature_names[i] for i in top_indices] - - # Get sample comments (shortest 3 for readability) - comment_lengths = [len(comment) for comment in cluster_comments] - shortest_indices = np.argsort(comment_lengths)[:3] # 3 shortest comments - sample_comments = [cluster_comments[i] for i in shortest_indices] - - # Add to cluster characteristics - cluster_characteristics[int(cluster_id)] = { - "size": len(cluster_members), - "top_words": top_words, - "top_tfidf_scores": [float(cluster_tfidf[i]) for i in top_indices], - "sample_comments": sample_comments, - } - - return cluster_characteristics - def create_comment_hover_info(cluster_layer, cluster_characteristics, comment_texts): """ diff --git a/docker-compose.yml b/docker-compose.yml index 30a6822fc7..748e0b4556 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -129,6 +129,18 @@ services: polis_tag: ${TAG:-dev} environment: - DATABASE_URL=${DATABASE_URL} + - DELPHI_RESULT_BACKEND=${DELPHI_RESULT_BACKEND:-dynamodb} + - DELPHI_RESULT_ENV=${DELPHI_RESULT_ENV:-} + - DELPHI_RESULT_SCOPE=${DELPHI_RESULT_SCOPE:-} + - QUEUE_DATABASE_URL=${QUEUE_DATABASE_URL:-} + - QUEUE_ENV=${DELPHI_RESULT_ENV:-} + - POLIS_JOBS_TRANSPORT=${POLIS_JOBS_TRANSPORT:-tls} + - POLIS_JOBS_CA_FILE=${POLIS_JOBS_CA_FILE:-} + - POLIS_JOBS_HOST_ALLOWLIST=${POLIS_JOBS_HOST_ALLOWLIST:-} + - POLIS_JOBS_WORKER_CLASS=delphi + - POLIS_JOBS_STAGES=delphi_full_pipeline,delphi_narrative + - POLIS_JOBS_JOURNAL_DIR=/app/data/queue-journal + - POLIS_JOBS_WORK_DIR=/app/data/queue-work - LOG_LEVEL=${DELPHI_LOG_LEVEL:-INFO} - DELPHI_DEV_OR_PROD=${DELPHI_DEV_OR_PROD:-prod} # The report pipeline reads math_main/math_ptptstats scoped by math_env, so @@ -183,6 +195,8 @@ services: # `true`. Off by default: with no Datadog agent listening, the tracer only # logs failed sends. Set it (with an agent) to turn tracing back on. - DD_TRACE_ENABLED=${DD_TRACE_ENABLED:-false} + volumes: + - delphi-queue-state:/app/data networks: - "polis-net" extra_hosts: @@ -613,6 +627,8 @@ networks: polis-net: volumes: + delphi-queue-state: + math-backfill-state: labels: polis_tag: ${TAG:-dev} diff --git a/docs/delphi-postgres-results.md b/docs/delphi-postgres-results.md new file mode 100644 index 0000000000..00d7030b18 --- /dev/null +++ b/docs/delphi-postgres-results.md @@ -0,0 +1,71 @@ +# Run-bound Delphi results (#1428) + +M28 stores all 18 result families from frozen `delphi-storage-codec/1`. Queue and +active-guard records remain control-plane records, never executable imported jobs. +No legacy table, vote value, or prior result is updated or deleted. + +`delphi_result_batches` binds one batch to an environment, graph job, graph run, +and attempt. `delphi_result_families` retains each canonical UTF-8 JSONL file and +its SHA-256. `delphi_result_rows` indexes tagged attributes by the family's real +composite key without converting decimal strings through floating point. JSONB +cannot represent NUL strings; importers must quarantine these with original bytes. + +A fenced worker or daemon calls: + +* `pd_result_put_family(env, job, owner, attempt, epoch, family, wire)`; identical + retries return the same digest, changed bytes at the same family key fail. +* `pd_result_seal(env, job, owner, attempt, epoch)`; returns the batch reference + `{schema, batch_id, sha256, families}`. A sealed batch refuses added families. +* The exact reference goes at artifact JSON `results`; graph finalization verifies + it and binds its artifact in the same transaction as queue success. Failed or + superseded attempts cannot affect any served generation. + +Small stage outputs may instead embed `family_files: {family: codecWire}`. The +artifact insert trigger validates, stores, seals and binds these within that same +transaction. The M27 artifact cap remains 512 KiB. The staged path permits up to +64 MiB per family and 256 MiB per batch. It does not require embedding full results +in the graph manifest or handing PostgreSQL credentials to a child process. + +Readers use `pd_result_artifact_family` for pinned dependencies and +`pd_result_served_bundle` for the coherent served generation. The view +`delphi_result_current_rows` provides environment, conversation, scope, generation, +family, tagged key, and tagged item for existing API filters. An entire family +comes from the closest artifact in the published dependency bundle; an explicitly +empty family shadows older ancestor rows. Existing graph publication CAS and +supersession checks determine visibility. Base tables are inaccessible to the +executor role, which receives only read APIs and fenced writers. Mutable human +annotations belong in separate generation-aware overrides, never these artifacts. + +`PostgresResultStore` and `PostgresResultReader` in +`polismath.delphi_storage.postgres` retain caller transaction ownership. They use +the existing frozen codec for Python resource values, validation, and reading. + +Validation commands (run on mm5): + +``` +cd delphi +python -m unittest tests.test_delphi_postgres_results -v +``` + +This exact command is added to the characterization workflow. The supplementary +real PostgreSQL campaign is `python delphi/tests/job_graph/results.py` with the +existing job-graph proof environment and an idle dedicated local queue. It covers +stale ownership, idempotence, changed retry refusal, atomic manifest rejection, +environment isolation, immutable tables, access denial and publication CAS. + +## Default and activation boundary + +M28/M29 are selected additive migrations in the release runner. Merging this +code does not switch readers or launch workers. An absent `DELPHI_RESULT_BACKEND` +uses DynamoDB in both Node and Python; `dynamodb` is the explicit default. +Only `postgres` selects the Postgres reader, with `DELPHI_RESULT_ENV` required +and `DELPHI_RESULT_SCOPE` optional. PostgreSQL errors never fall back to DynamoDB. +The server already receives these settings through its environment file; Python +reader processes need them in their own environment. No production flag is set +by this change. Verify published/imported coverage before Colin activates it. + +`delphi/scripts/import_dynamo_export.py` is an explicitly invoked command. No +startup hook, migration or service automatically runs it. See +`delphi/docs/LEGACY_DYNAMO_IMPORT.md` for its bounded input contract. The proof's +fixed narrative/name outputs demonstrate transport only. Provider execution, +text quality, old Dynamo table retirement and reader activation are separate. diff --git a/docs/delphi-postgres-writers.md b/docs/delphi-postgres-writers.md new file mode 100644 index 0000000000..420681c058 --- /dev/null +++ b/docs/delphi-postgres-writers.md @@ -0,0 +1,67 @@ +# Delphi writers on PostgreSQL + +`DELPHI_RESULT_BACKEND=postgres` now selects PostgreSQL for job admission and +result writes as well as reads. The default remains `dynamodb`. This change does +not activate a deployment or remove a DynamoDB table. + +Apply the release migration chain through the migration runner, including M30. +Set `DELPHI_RESULT_ENV`, `DELPHI_RESULT_SCOPE`, and `DELPHI_WRITER_CODE_SHA` on the +server. The code pin must identify the deployed writer release (40 or 64 lower +case hex characters). Both Delphi submission routes use the same atomic queue +admission function, retaining request binding and scope exclusion. Narrative +batch route IDs become queue UUIDs; the response supplies the authoritative ID. +The existing moderation pin remains true. A checker is a reclaimed attempt of +its original narrative job; submitting an unrelated `AWAITING_NARRATIVE_BATCH` +request is refused. + +Workers use the existing `polis-jobs` daemon, a restricted `polis_queue_executor` +login, and the normal verified TLS configuration (`QUEUE_DATABASE_URL`, +`QUEUE_ENV`, `POLIS_JOBS_TRANSPORT=tls`, `POLIS_JOBS_CA_FILE`, and +`POLIS_JOBS_HOST_ALLOWLIST`). Mount the CA and persistent journal/work directories +according to the worker deployment contract. The Compose Delphi service forwards +the switch and queue settings and retains its journal/work directory in +`delphi-queue-state`. The Postgres image startup selects the daemon and skips the +legacy Dynamo bootstrap/poller. Standalone legacy writers without a bound queue +attempt refuse rather than writing to Dynamo or mutating a publication. + +The daemon hydrates the immutable publication captured at admission. Pipeline +subprocesses share a private SQLite working copy in their attempt directory. +The reset clears computed families in that copy, preserving narrative history, +collective statements and agenda selections. It never deletes served artifacts. +The existing numerical and provider scripts still run; no local narrative model +or fixture is selected in production. The existing provider intent, ambiguous +submission and submit/recheck lifecycle remains the authority for paid work. + +After the process group exits, the daemon imports canonical family files using +its fenced queue token. The v2 manifest binds the sealed batch. Finalization +stores the artifact, completes the job and advances the served generation in one +transaction. Lease loss, failed stages, unresolved provider work or a publication +conflict cannot expose the partial working copy. Limits remain 64 MiB per family +and 256 MiB per result batch; oversize fails without truncation. + +HTTP statement/narrative puts and deletes create a durable producer job, attempt +and immutable artifact within one SQL transaction. Canonical keys bind each edit +to its conversation/report. Concurrent edits preserve unrelated keys. An active +pipeline retains its publication scope; synchronous edits report failure while +that scope is busy. The existing HTTP authorization remains in the routes. +Unknown families still return `ResourceNotFoundException`: the two never-built +topic moderation stores retain their existing unavailable behavior. Building +that separate feature is not part of migrating the live writers. + +Changing the switch back to Dynamo after new Postgres writes is **not a lossless +rollback**. There is no dual write. Stop new admissions and drain/fence workers +before changing writer deployments; retain the Postgres artifacts and publication +pointers. Moving newly written results back to another backend needs an explicit +migration. M30 does not rewrite historical migrations or offer destructive down +migration over result history. + +## Verification + +Hosted CI and workboxes run `bash scripts/test-dynamo-writers.sh` with an isolated +Compose project and port. It runs the Rust library checks, server build/adapter +checks, Python writer/protocol checks, release migrations and real Postgres/daemon +scenarios with the owned DynamoDB container stopped. The process fixture produces +generated results and a fixed provider response: it proves storage and lifecycle, +not numerical equivalence, narrative quality or a live provider call. The real +pipeline orchestration also has a private-reset/manifest test with computation +subprocesses stubbed. Existing numerical suites remain separate. diff --git a/example.env b/example.env index 48b3ce6508..a846f3bc67 100644 --- a/example.env +++ b/example.env @@ -348,3 +348,13 @@ ENCRYPTION_PASSWORD_00001= # (Deprecated) Basic Auth settings for certain requests between math and api services. WEBSERVER_PASS=ws-pass WEBSERVER_USERNAME=ws-user + +###### DELPHI RESULT READERS ###### +# Unset means DynamoDB. Only an explicit postgres value selects the new reader. +# PostgreSQL needs M28/M29 and verified published/imported coverage first. +# DELPHI_RESULT_BACKEND=dynamodb +# DELPHI_RESULT_ENV= +# DELPHI_RESULT_SCOPE= + +# Required code provenance when admitting Postgres Delphi writers (40/64 hex). +# DELPHI_WRITER_CODE_SHA= diff --git a/queue-rs/polis-api/conformance/nginx-routing.sh b/queue-rs/polis-api/conformance/nginx-routing.sh index 48a8219f1f..d3f12eed61 100755 --- a/queue-rs/polis-api/conformance/nginx-routing.sh +++ b/queue-rs/polis-api/conformance/nginx-routing.sh @@ -37,17 +37,48 @@ stub() { # container alias port status body stub "$p-node" server 5000 200 node stub "$p-alpha" client-participation-alpha 4321 200 alpha get() { docker exec "$p-proxy" wget -qO- "http://127.0.0.1$1" 2>/dev/null || echo "(no answer)"; } -# A raw HTTP/1.0 exchange through the proxy; prints the X-Upstream that answered. -upstream_of() { # method path [extra header lines] [body] - printf '%s %s HTTP/1.0\r\nHost: localhost\r\n%s\r\n%s' "$1" "$2" "${3:-}" "${4:-}" \ - | docker exec -i "$p-proxy" nc -w 5 127.0.0.1 80 \ - | tr -d '\r' | sed -n 's/^X-Upstream: //p' +# BusyBox nc can exit on stdin EOF before the HTTP response arrives (nginx +# logs 499). Use a bounded TCP client on an ephemeral loopback-only proxy port. +expect_upstream() { # expected upstream method path [extra header lines] [body] + local want=$1 response got + shift + response=$(python3 - "$p-proxy" "$@" <<'PYTHON' +import socket +import subprocess +import sys + +proxy, method, path, *rest = sys.argv[1:] +headers, body = (rest + ["", ""])[:2] +request = f"{method} {path} HTTP/1.0\r\nHost: localhost\r\n{headers}\r\n{body}".encode() +address = subprocess.check_output(["docker", "port", proxy, "80/tcp"], text=True).strip() +host, port = address.rsplit(":", 1) +assert host == "127.0.0.1", f"test proxy must bind loopback only: {address}" +with socket.create_connection((host, int(port)), timeout=5) as client: + client.sendall(request) + while True: + chunk = client.recv(65536) + if not chunk: + break + sys.stdout.buffer.write(chunk) + sys.stdout.buffer.flush() + +PYTHON + ) || { printf '%s\n' "$response" >&2; fail "raw HTTP transport failed: $1 $2"; } + + got=$(printf '%s\n' "$response" | tr -d '\r' | sed -n 's/^X-Upstream: //p') + if [[ $got != "$want" ]]; then + printf 'Raw response for %s %s (expected X-Upstream: %s):\n%s\n' "$1" "$2" "$want" "$response" >&2 + docker logs "$p-proxy" >&2 2>&1 || true + docker logs "$p-node" >&2 2>&1 || true + fail "$1 $2 answered X-Upstream='$got', expected '$want'" + fi } + start() { # routes [docker run args...] local routes=${1:-} shift || true docker rm -f "$p-proxy" >/dev/null 2>&1 || true - docker run -d --name "$p-proxy" --network "$net" -e RUST_API_ROUTES="$routes" "$@" "$image" >/dev/null + docker run -d --name "$p-proxy" --network "$net" -p 127.0.0.1::80 -e RUST_API_ROUTES="$routes" "$@" "$image" >/dev/null for _ in $(seq 20); do docker exec "$p-proxy" wget -qO- http://127.0.0.1/ >/dev/null 2>&1 && return 0 sleep 0.5 @@ -111,12 +142,12 @@ expect '/api/v3/math/pca2?conversation_id=x' rust "route named twice" [[ $(included | grep -c 'location = /api/v3/math/pca2') == 1 ]] || fail "route named twice: rendered more than once" echo "ok 7 route named twice: rendered once, nginx serves, pca2 -> polis-api" -[[ $(upstream_of GET /api/v3/math/pca2) == polis-api ]] || fail "plain GET should reach polis-api" -[[ $(upstream_of HEAD /api/v3/math/pca2) == polis-api ]] || fail "HEAD should reach polis-api" -[[ $(upstream_of OPTIONS /api/v3/math/pca2) == server ]] || fail "OPTIONS should go to node" -[[ $(upstream_of POST /api/v3/math/pca2 $'Content-Length: 2\r\n' '{}') == server ]] || fail "POST should go to node" -[[ $(upstream_of GET /api/v3/math/pca2 $'Content-Type: application/json\r\nContent-Encoding: gzip\r\nContent-Length: 2\r\n' '{}') == server ]] \ - || fail "GET with a body should go to node" +expect_upstream polis-api GET /api/v3/math/pca2 +expect_upstream polis-api HEAD /api/v3/math/pca2 +expect_upstream server OPTIONS /api/v3/math/pca2 +expect_upstream server POST /api/v3/math/pca2 $'Content-Length: 2\r\n' '{}' +expect_upstream server GET /api/v3/math/pca2 $'Content-Type: application/json\r\nContent-Encoding: gzip\r\nContent-Length: 2\r\n' '{}' + echo "ok 8 only GET/HEAD without a body reach polis-api; OPTIONS, POST, GET with a body -> node" for bad in "RUST_API_UPSTREAM=bad host;" "RUST_API_RESOLVER=bad resolver" "RUST_API_RESOLVER=1.2.3.4; }"; do diff --git a/queue-rs/polis-migrate/tests/prove.py b/queue-rs/polis-migrate/tests/prove.py index 8de921ac41..ef15ef4c23 100644 --- a/queue-rs/polis-migrate/tests/prove.py +++ b/queue-rs/polis-migrate/tests/prove.py @@ -3,6 +3,10 @@ import json, os, pathlib, shutil, subprocess, tempfile, time, unittest ROOT=pathlib.Path(__file__).resolve().parents[3] MIG=ROOT/'server/postgres/migrations' +SELECTED=[s for s in (MIG/'release.txt').read_text().splitlines() if s and not s.startswith('#')] +RETIRED={'000004','000005','000007'} +APPLIED=[s for s in SELECTED if s[:6] not in RETIRED] +PENDING=[s[:6] for s in SELECTED if int(s[:6])>18 and not s.startswith('000022_')] BIN=ROOT/'queue-rs/target/debug/polis-migrate' COMPOSE=['docker','compose','-f',str(pathlib.Path(__file__).with_name('compose.yml'))] assert os.environ.get('COMPOSE_PROJECT_NAME','').startswith('polis-migrate-test-') @@ -42,7 +46,7 @@ def test_18_deploy_legacy_and_unchanged_rerun(self): db=self.legacy('deploy_legacy') first=runner(db,'deploy') self.assertIn('adopted 20 migration(s)',first.stdout) - self.assertEqual([x.split()[1][:6] for x in first.stdout.splitlines() if x.startswith('APPLIED ')],['000019','000023','000024','000027']) + self.assertEqual([x.split()[1][:6] for x in first.stdout.splitlines() if x.startswith('APPLIED ')],PENDING) before=sql(db,'SELECT row_to_json(m) FROM migrations m ORDER BY name') self.assertIn('applied 0 migration(s)',runner(db,'deploy').stdout) self.assertEqual(before,sql(db,'SELECT row_to_json(m) FROM migrations m ORDER BY name')) @@ -62,7 +66,7 @@ def test_20_deploy_concurrent_first_adoption(self): out,err=p.communicate(timeout=350) self.assertEqual(p.returncode,0,out+err);outputs.append(out) self.assertEqual(sum('adopted 20 migration(s)' in x for x in outputs),1) - self.assertEqual(sum('applied 4 migration(s)' in x for x in outputs),1) + self.assertEqual(sum(f'applied {len(PENDING)} migration(s)' in x for x in outputs),1) runner(db,'check') def test_21_deploy_legacy_ledger_preserves_timestamps(self): @@ -79,19 +83,38 @@ def test_22_deploy_rejects_unverified_ledger(self): def test_23_modern_v3_upgrades_to_v5_preserving_receipts(self): db=self.new('modern_v3') + upgrades=[s for s in SELECTED if s[:6] >= '000027'] with tempfile.TemporaryDirectory() as tmp: d=pathlib.Path(tmp)/'m';shutil.copytree(MIG,d) - f=d/'release.txt';f.write_text('\n'.join(x for x in f.read_text().splitlines() if not x.startswith('000027_'))+'\n') - next(d.glob('000027_*.sql')).unlink() + f=d/'release.txt';f.write_text('\n'.join(x for x in f.read_text().splitlines() if x not in upgrades)+'\n') + for name in upgrades: + (d/name).unlink() runner(db,'deploy',dir=d) before=sql(db,'SELECT row_to_json(m) FROM migrations m ORDER BY name') gate(db,False) output=runner(db,'deploy').stdout - self.assertIn('applied 1 migration(s)',output) - self.assertEqual(before,sql(db,"SELECT row_to_json(m) FROM migrations m WHERE name NOT LIKE '000027_%' ORDER BY name")) + self.assertEqual([x.split()[1] for x in output.splitlines() if x.startswith('APPLIED ')], upgrades) + self.assertEqual(before,sql(db,"SELECT row_to_json(m) FROM migrations m WHERE name < '000027' ORDER BY name")) self.assertEqual(sql(db,'SELECT contract_version FROM polis_queue_install'),'polis-queue/5') gate(db,True) + def test_25_release_b_upgrades_results_preserving_receipts(self): + db=self.new('release_b') + upgrades=[s for s in SELECTED if s[:6] >= '000028'] + with tempfile.TemporaryDirectory() as tmp: + d=pathlib.Path(tmp)/'m';shutil.copytree(MIG,d) + f=d/'release.txt';f.write_text('\n'.join(x for x in f.read_text().splitlines() if x not in upgrades)+'\n') + for name in upgrades: + (d/name).unlink() + runner(db,'deploy',dir=d) + before=sql(db,'SELECT row_to_json(m) FROM migrations m ORDER BY name') + gate(db,False) + output=runner(db,'deploy').stdout + self.assertEqual([x.split()[1] for x in output.splitlines() if x.startswith('APPLIED ')], upgrades) + self.assertEqual(before,sql(db,"SELECT row_to_json(m) FROM migrations m WHERE name < '000028' ORDER BY name")) + gate(db,True) + self.assertIn('applied 0 migration(s)',runner(db,'deploy').stdout) + def test_24_unledgered_graph_is_not_readopted(self): db=self.new('unledgered_v5');runner(db,'deploy') sql(db,'DROP TABLE migrations') @@ -130,8 +153,8 @@ def test_16_json_equivalence_is_only_for_legacy_math(self): def test_01_fresh(self): db=self.new('fresh') - self.assertIn('applied 21 migration(s)',runner(db,'apply').stdout) - self.assertEqual(sql(db,"SELECT count(*) FROM migrations WHERE status='APPLIED'"),'21') + self.assertIn(f'applied {len(APPLIED)} migration(s)',runner(db,'apply').stdout) + self.assertEqual(sql(db,"SELECT count(*) FROM migrations WHERE status='APPLIED'"),str(len(APPLIED))) runner(db,'check');gate(db,True) self.assertEqual(sql(db,"SELECT to_regclass('public.polis_coordinator_install') IS NULL"),'t') def test_02_reconcile_then_pending(self): @@ -142,7 +165,7 @@ def test_02_reconcile_then_pending(self): self.assertEqual(sql(db,"SELECT count(*) FROM migrations WHERE status='ADOPTED'"),'20') gate(db,False) p=runner(db,'apply') - self.assertEqual([s.split()[1][:6] for s in p.stdout.splitlines() if s.startswith('APPLIED ')],['000019','000023','000024','000027']) + self.assertEqual([s.split()[1][:6] for s in p.stdout.splitlines() if s.startswith('APPLIED ')],PENDING) runner(db,'check');gate(db,True) self.assertEqual(sql(db,"SELECT hname FROM users"),'generated migration sentinel') def test_03_noop(self): @@ -187,9 +210,9 @@ def test_05_two_runners(self): outputs=[] for p in [a,b]: out,err=p.communicate(timeout=350);self.assertEqual(p.returncode,0,err); outputs.append(out) - self.assertEqual(sorted('applied 21 migration(s)' in o for o in outputs),[False,True]) + self.assertEqual(sorted(f'applied {len(APPLIED)} migration(s)' in o for o in outputs),[False,True]) self.assertEqual(sorted('applied 0 migration(s)' in o for o in outputs),[False,True]) - self.assertEqual(sql(db,'SELECT count(*) FROM migrations'),'24') + self.assertEqual(sql(db,'SELECT count(*) FROM migrations'),str(len(SELECTED))) def test_06_bad_catalog_adopts_nothing(self): db=self.legacy('badcatalog') sql(db,'ALTER TABLE conversations DROP COLUMN topics_enabled') diff --git a/queue-rs/polis-migrate/tests/selection.py b/queue-rs/polis-migrate/tests/selection.py index 27af3896d4..9c64928586 100644 --- a/queue-rs/polis-migrate/tests/selection.py +++ b/queue-rs/polis-migrate/tests/selection.py @@ -34,7 +34,7 @@ def test_02_retired_sql_never_runs(self): for n in ['000004','000005','000007']: next(d.glob(n+'*.sql')).write_text('SELECT 1/0;') output=p.runner(db,'apply',dir=d).stdout - self.assertIn('applied 21 migration(s)',output) + self.assertIn(f'applied {len(p.APPLIED)} migration(s)',output) self.assertEqual(p.sql(db,"SELECT count(*) FROM migrations WHERE status='ADOPTED'"),'3') p.runner(db,'check',dir=d);p.gate(db,True,dir=d) diff --git a/queue-rs/schemas/job-frame-v1.json b/queue-rs/schemas/job-frame-v1.json index cd88e794ef..69b037698c 100644 --- a/queue-rs/schemas/job-frame-v1.json +++ b/queue-rs/schemas/job-frame-v1.json @@ -97,6 +97,14 @@ "integer", "null" ] + }, + "result_backend": { + "const": "postgres" + }, + "result_scope": { + "type": "string", + "minLength": 1, + "maxLength": 128 } } }, @@ -180,6 +188,10 @@ ] } } + }, + "writer_base": { + "type": "object", + "description": "Immutable base for a Postgres writer, hydrated by the credential-holding daemon." } } } diff --git a/queue-rs/schemas/output-manifest-v2.json b/queue-rs/schemas/output-manifest-v2.json new file mode 100644 index 0000000000..f95777b557 --- /dev/null +++ b/queue-rs/schemas/output-manifest-v2.json @@ -0,0 +1,290 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "polis-jobs.output-manifest/2", + "title": "Output manifest written by a Delphi child as its last act", + "description": "Postgres writer manifest. Before finalization the daemon imports bounded spool files, seals them and replaces family_spool with a fenced batch receipt.", + "type": "object", + "additionalProperties": false, + "required": [ + "schema", + "job_id", + "attempt_id", + "stage", + "phase", + "outcome", + "inputs", + "outputs", + "models", + "cost", + "recheck_after", + "duration_ms" + ], + "properties": { + "schema": { + "const": "polis-jobs.output-manifest/2" + }, + "job_id": { + "type": "string", + "format": "uuid" + }, + "attempt_id": { + "type": "string", + "format": "uuid" + }, + "stage": { + "enum": [ + "delphi_full_pipeline", + "delphi_narrative", + "math_rebuild" + ] + }, + "phase": { + "enum": [ + "run", + "submit", + "recheck" + ] + }, + "outcome": { + "enum": [ + "succeeded", + "parked" + ] + }, + "inputs": { + "type": "object", + "additionalProperties": false, + "required": [ + "math_env", + "math_tick", + "math_caching_tick", + "comment_set_sha256", + "vote_hwm" + ], + "properties": { + "math_env": { + "type": [ + "string", + "null" + ] + }, + "math_tick": { + "type": [ + "integer", + "null" + ] + }, + "math_caching_tick": { + "type": [ + "integer", + "null" + ] + }, + "comment_set_sha256": { + "type": [ + "string", + "null" + ], + "pattern": "^[0-9a-f]{64}$" + }, + "vote_hwm": { + "type": [ + "integer", + "null" + ] + } + } + }, + "outputs": { + "description": "Result counts for the attempt-local Postgres working copy.", + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "store", + "family", + "table", + "rows" + ], + "properties": { + "store": { + "const": "postgres" + }, + "family": { + "type": "string", + "minLength": 1 + }, + "table": { + "type": "string", + "minLength": 1 + }, + "key_prefix": { + "type": "object", + "minProperties": 1, + "additionalProperties": { + "type": "string" + } + }, + "keys": { + "type": "array", + "items": { + "type": "object", + "minProperties": 1, + "additionalProperties": { + "type": "string" + } + } + }, + "rows": { + "type": "integer", + "minimum": 0 + } + }, + "oneOf": [ + { + "required": [ + "key_prefix" + ] + }, + { + "required": [ + "keys" + ] + } + ] + } + }, + "models": { + "type": "object", + "additionalProperties": false, + "required": [ + "embed", + "topic", + "narrative" + ], + "properties": { + "embed": { + "type": [ + "string", + "null" + ] + }, + "topic": { + "type": [ + "string", + "null" + ] + }, + "narrative": { + "type": [ + "string", + "null" + ] + } + } + }, + "cost": { + "type": "object", + "additionalProperties": false, + "required": [ + "llm_tokens_in", + "llm_tokens_out", + "provider_batches" + ], + "properties": { + "llm_tokens_in": { + "type": [ + "integer", + "null" + ] + }, + "llm_tokens_out": { + "type": [ + "integer", + "null" + ] + }, + "provider_batches": { + "type": [ + "array", + "null" + ], + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "provider", + "batch_id", + "submitted_at" + ], + "properties": { + "provider": { + "type": "string", + "minLength": 1 + }, + "batch_id": { + "type": "string", + "minLength": 1 + }, + "submitted_at": { + "type": "string", + "minLength": 1 + } + } + } + } + } + }, + "recheck_after": { + "type": [ + "string", + "null" + ], + "format": "date-time", + "description": "Set exactly when outcome is parked." + }, + "duration_ms": { + "type": "integer", + "minimum": 0 + }, + "family_spool": { + "type": "object", + "maxProperties": 18, + "additionalProperties": { + "type": "object", + "additionalProperties": false, + "required": [ + "file", + "sha256" + ], + "properties": { + "file": { + "type": "string", + "pattern": "^[A-Za-z0-9_]+[.]jsonl$" + }, + "sha256": { + "type": "string", + "pattern": "^[0-9a-f]{64}$" + } + } + } + }, + "results": { + "type": "object", + "description": "Fenced delphi-result-batch/1 receipt inserted by daemon, never trusted from child." + } + }, + "oneOf": [ + { + "required": [ + "family_spool" + ] + }, + { + "required": [ + "results" + ] + } + ] +} diff --git a/queue-rs/src/jobs/child.rs b/queue-rs/src/jobs/child.rs index 432914c38e..63d180108c 100644 --- a/queue-rs/src/jobs/child.rs +++ b/queue-rs/src/jobs/child.rs @@ -241,7 +241,7 @@ pub fn frame(claim: &Claim, adm: &Admission, phase: &str, batch_id: Option<&str> "input_sha256":adm.config["input_sha256"],"input_json":adm.config["input_wire"]}), ); } - let config = if claim.stage == "math_rebuild" { + let mut config = if claim.stage == "math_rebuild" { MathConfig::from_admission(adm)?.to_json() } else { json!({ @@ -251,6 +251,10 @@ pub fn frame(claim: &Claim, adm: &Admission, phase: &str, batch_id: Option<&str> "batch_size": adm.config.get("batch_size").cloned().unwrap_or(Value::Null), }) }; + if adm.config["result_backend"] == "postgres" { + config["result_backend"] = json!("postgres"); + config["result_scope"] = adm.config["result_scope"].clone(); + } Ok(json!({ "schema": FRAME_SCHEMA, "env": claim.env, "zid": adm.zid, "report_id": adm.report_id, "job_id": claim.job_id, "run_id": claim.run_id, "attempt_id": claim.attempt_id, @@ -339,7 +343,7 @@ pub fn command_args( app.join("umap_narrative/803_check_batch_status.py"), vec![format!("--job-id={}", claim.job_id)], ), - ("graph_embed" | "graph_cluster" | "graph_narrative", "run") => { + ("graph_embed" | "graph_cluster" | "graph_topics" | "graph_narrative", "run") => { (app.join("scripts/job_graph_stage.py"), vec![]) } ("math_rebuild", "run") => (app.join("scripts/math_poller.py"), vec!["--job".to_owned()]), @@ -352,6 +356,57 @@ pub struct Spawned { pub pgid: i32, } +const GRAPH_ENVIRONMENT: &[&str] = &[ + "PATH", + "HOME", + "TMPDIR", + "TMP", + "TEMP", + "LANG", + "LC_ALL", + "TZ", + "PYTHONPATH", + "PYTHONUNBUFFERED", + "PYTHONDONTWRITEBYTECODE", + "OMP_NUM_THREADS", + "MKL_NUM_THREADS", + "OPENBLAS_NUM_THREADS", + "NUMBA_NUM_THREADS", + "NUMEXPR_NUM_THREADS", + "VECLIB_MAXIMUM_THREADS", + "TOKENIZERS_PARALLELISM", + "HF_HOME", + "HF_HUB_OFFLINE", + "TRANSFORMERS_OFFLINE", + "XDG_CACHE_HOME", + "NUMBA_CACHE_DIR", + "DELPHI_EMBED_MODEL_PATH", + // Existing process-boundary proof fixture paths; neither carries credentials. + "GRAPH_PROOF_ROOT", + "GRAPH_STAGE_SOURCE", +]; + +fn child_environment( + command: &mut Command, + stage: &str, + inherited: impl IntoIterator, +) { + if stage.starts_with("graph_") { + command.env_clear(); + for (key, value) in inherited { + if key + .to_str() + .is_some_and(|key| GRAPH_ENVIRONMENT.contains(&key)) + { + command.env(key, value); + } + } + } + command + .env_remove("QUEUE_DATABASE_URL") + .env_remove("POLIS_JOBS_PASSWORD_FILE"); +} + /// Spawn in a new session (pgid = pid). Queue credentials never reach the child. #[allow(clippy::too_many_arguments)] pub fn spawn( @@ -367,12 +422,11 @@ pub fn spawn( batch_id: Option<&str>, ) -> Result { let mut command = Command::new(python); + child_environment(&mut command, &claim.stage, std::env::vars_os()); command .arg(script) .args(args) .current_dir(app) - .env_remove("QUEUE_DATABASE_URL") - .env_remove("POLIS_JOBS_PASSWORD_FILE") .env("DELPHI_JOB_ID", &claim.job_id) .env("DELPHI_RUN_ID", &claim.run_id) .env("DELPHI_ATTEMPT_ID", &claim.attempt_id) @@ -388,6 +442,15 @@ pub fn spawn( .stdin(Stdio::piped()) .stdout(Stdio::piped()) .stderr(Stdio::piped()); + if adm.config["result_backend"] == "postgres" { + command + .env("DELPHI_RESULT_BACKEND", "postgres") + .env("DELPHI_RESULT_ENV", &claim.env) + .env( + "DELPHI_RESULT_SCOPE", + adm.config["result_scope"].as_str().unwrap_or_default(), + ); + } if let Some(batch) = batch_id { command.env("DELPHI_PROVIDER_BATCH_ID", batch); } @@ -512,6 +575,77 @@ pub fn kill_and_reap( mod tests { use super::*; + #[test] + fn graph_children_receive_only_model_and_runtime_environment() -> Result<()> { + let inherited: Vec<(std::ffi::OsString, std::ffi::OsString)> = [ + "DATABASE_URL", + "DELPHI_RESULT_DATABASE_URL", + "MATH_CAPACITY_QUEUE_DSN", + "QUEUE_DATABASE_URL", + "PGHOST", + "PGPORT", + "PGDATABASE", + "PGUSER", + "PGPASSWORD", + "PGPASSFILE", + "PGSERVICE", + "PGSERVICEFILE", + "POSTGRES_PASSWORD", + "DB_PASSWORD", + "POLIS_JOBS_PASSWORD_FILE", + "UNRECOGNIZED_DATABASE_DSN", + "DELPHI_EMBED_MODEL_PATH", + "OMP_NUM_THREADS", + "PATH", + ] + .into_iter() + .map(|key| (key.into(), format!("generated-{key}").into())) + .collect(); + let mut command = Command::new("/usr/bin/env"); + command.envs(inherited.clone()); + child_environment(&mut command, "graph_embed", inherited); + let output = command.output()?; + assert!(output.status.success()); + let lines = String::from_utf8(output.stdout)?; + let keys: Vec<&str> = lines + .lines() + .filter_map(|line| line.split_once('=').map(|(key, _)| key)) + .collect(); + assert_eq!(keys.len(), 3); + assert!( + keys.iter().all(|key| { + ["DELPHI_EMBED_MODEL_PATH", "OMP_NUM_THREADS", "PATH"].contains(key) + }) + ); + Ok(()) + } + #[test] + fn math_environment_keeps_its_database_access() { + let mut command = Command::new("python"); + command + .env("DATABASE_URL", "generated-main") + .env("MATH_CAPACITY_QUEUE_DSN", "generated-capacity") + .env("QUEUE_DATABASE_URL", "generated-queue") + .env("POLIS_JOBS_PASSWORD_FILE", "generated-file"); + child_environment(&mut command, "math_rebuild", std::iter::empty()); + let actual: std::collections::BTreeMap<_, _> = command.get_envs().collect(); + assert_eq!( + actual.get(std::ffi::OsStr::new("DATABASE_URL")), + Some(&Some(std::ffi::OsStr::new("generated-main"))) + ); + assert_eq!( + actual.get(std::ffi::OsStr::new("MATH_CAPACITY_QUEUE_DSN")), + Some(&Some(std::ffi::OsStr::new("generated-capacity"))) + ); + assert_eq!( + actual.get(std::ffi::OsStr::new("QUEUE_DATABASE_URL")), + Some(&None) + ); + assert_eq!( + actual.get(std::ffi::OsStr::new("POLIS_JOBS_PASSWORD_FILE")), + Some(&None) + ); + } #[test] fn base64url_round_trip() { for input in [ diff --git a/queue-rs/src/jobs/config.rs b/queue-rs/src/jobs/config.rs index 22ee8a5fcb..e18f18e0a2 100644 --- a/queue-rs/src/jobs/config.rs +++ b/queue-rs/src/jobs/config.rs @@ -6,7 +6,12 @@ use std::{collections::BTreeSet, path::PathBuf, time::Duration}; /// `polis_queue_jobs_stage_check`). pub const KNOWN_STAGES: [&str; 2] = ["delphi_full_pipeline", "delphi_narrative"]; /// The stages of class `large`, which `polis-queue/3` admits (000024). -pub const GRAPH_STAGES: [&str; 3] = ["graph_embed", "graph_cluster", "graph_narrative"]; +pub const GRAPH_STAGES: [&str; 4] = [ + "graph_embed", + "graph_cluster", + "graph_topics", + "graph_narrative", +]; pub const LARGE_STAGES: [&str; 1] = ["math_rebuild"]; /// The worker class the daemon claims as (`POLIS_JOBS_WORKER_CLASS`). Each diff --git a/queue-rs/src/jobs/graph.rs b/queue-rs/src/jobs/graph.rs index fb823a4a10..b9683e9849 100644 --- a/queue-rs/src/jobs/graph.rs +++ b/queue-rs/src/jobs/graph.rs @@ -84,9 +84,267 @@ pub fn validate( }) } +/// Hydrate only immutable, digest-bound result bytes; queue credentials stay in daemon. +pub fn hydrate(rpc: &mut super::rpc::Rpc, frame: &mut Value) -> Result<()> { + if frame["stage"] != "graph_cluster" + || frame["input"]["declared"]["model"] != "delphi-umap-evoc/1" + { + return Ok(()); + } + let artifact = &frame["input"]["artifacts"]["embeddings"]; + let payload: Value = serde_json::from_str(artifact["payload"].as_str().unwrap_or_default())?; + if payload.get("family_files").is_some() { + return Ok(()); + } + let family = "Delphi_CommentEmbeddings"; + let reply = rpc.committed( + "pd_result_artifact_wire", + &[ + frame["env"].clone(), + artifact["artifact_id"].clone(), + json!(family), + ], + )?; + let wire = reply["wire"] + .as_str() + .ok_or_else(|| anyhow::anyhow!("result_wire_missing"))?; + ensure!( + wire.len() <= 67_108_864 + && reply["sha256"] == sha256_hex(wire.as_bytes()) + && reply["batch_id"] == payload["results"]["batch_id"] + && reply["batch_sha256"] == payload["results"]["sha256"], + "result_wire_binding" + ); + frame["result_families"] = json!({"embeddings": {family: wire}}); + Ok(()) +} + +/// Read only a bounded regular spool file; a FIFO must never block the daemon. +fn read_family_spool( + directory: &std::path::Path, + family: &str, + value: &Value, + total: &mut u64, +) -> Result { + ensure!( + !family.is_empty() + && family + .bytes() + .all(|c| c.is_ascii_alphanumeric() || c == b'_'), + "result_family_name" + ); + let filename = format!("{family}.jsonl"); + ensure!(value["file"] == filename, "result_spool_path"); + let bytes = super::manifest::read_regular_file(&directory.join(filename), 67_108_864) + .map_err(|error| anyhow::anyhow!("result_spool_file:{error}"))?; + let wire = String::from_utf8(bytes)?; + let next_total = total + .checked_add(wire.len() as u64) + .ok_or_else(|| anyhow::anyhow!("result_spool_bound"))?; + ensure!(next_total <= 268_435_456, "result_spool_bound"); + ensure!( + value["sha256"] == sha256_hex(wire.as_bytes()), + "result_spool_digest" + ); + *total = next_total; + Ok(wire) +} + +/// Persist detached family bytes after process exit and before artifact finalize. +/// The daemon creates the final manifest linking the attempt's sealed batch. +pub fn stage_results( + rpc: &mut super::rpc::Rpc, + claim: &super::child::Claim, + directory: &std::path::Path, + manifest: Manifest, +) -> Result { + let mut document: Value = serde_json::from_str(&manifest.text)?; + let mut payload: Value = + serde_json::from_str(document["output"]["payload"].as_str().unwrap_or_default())?; + let Some(spool) = payload.get("family_spool") else { + return Ok(manifest); + }; + let files = spool + .as_object() + .ok_or_else(|| anyhow::anyhow!("result_spool_shape"))?; + ensure!(!files.is_empty() && files.len() <= 18, "result_spool_count"); + let mut total = 0_u64; + for (family, value) in files { + let wire = read_family_spool(directory, family, value, &mut total)?; + let mut args = super::task::identity(claim); + args.extend([json!(family), json!(wire)]); + let reply = rpc.committed("pd_result_put_family", &args)?; + ensure!(reply["sha256"] == value["sha256"], "result_stage_receipt"); + } + let results = rpc.committed("pd_result_seal", &super::task::identity(claim))?; + ensure!( + results["schema"] == "delphi-result-batch/1" && results["batch_id"] == claim.attempt_id, + "result_batch_receipt" + ); + payload + .as_object_mut() + .ok_or_else(|| anyhow::anyhow!("result_payload"))? + .remove("family_spool"); + payload["results"] = results; + let wire = serde_json::to_string(&payload)?; + document["output"]["sha256"] = json!(sha256_hex(wire.as_bytes())); + document["output"]["payload"] = json!(wire); + let bytes = serde_json::to_vec(&document)?; + let result = validate(Some(&bytes), &claim.job_id, &claim.attempt_id, &claim.stage) + .map_err(|_| anyhow::anyhow!("result_manifest_invalid"))?; + let temporary = directory.join("output-manifest.sealed.tmp"); + std::fs::write(&temporary, &bytes)?; + std::fs::rename(temporary, directory.join("output-manifest.json"))?; + Ok(result) +} + +/// The legacy provider lifecycle uses the same bounded, fenced family spool. +pub fn stage_writer_results( + rpc: &mut super::rpc::Rpc, + claim: &super::child::Claim, + directory: &std::path::Path, + manifest: Manifest, +) -> Result { + let mut document: Value = serde_json::from_str(&manifest.text)?; + ensure!( + document["schema"] == "polis-jobs.output-manifest/2", + "writer_manifest_schema" + ); + let files = document["family_spool"] + .as_object() + .ok_or_else(|| anyhow::anyhow!("writer_spool_shape"))?; + ensure!(!files.is_empty() && files.len() <= 18, "writer_spool_count"); + let mut total = 0_u64; + for (family, value) in files { + let wire = read_family_spool(directory, family, value, &mut total)?; + let mut args = super::task::identity(claim); + args.extend([json!(family), json!(wire)]); + let reply = rpc.committed("pd_result_put_family", &args)?; + ensure!(reply["sha256"] == value["sha256"], "writer_stage_receipt"); + } + let results = rpc.committed("pd_result_seal", &super::task::identity(claim))?; + ensure!( + results["schema"] == "delphi-result-batch/1" && results["batch_id"] == claim.attempt_id, + "writer_batch_receipt" + ); + document + .as_object_mut() + .ok_or_else(|| anyhow::anyhow!("writer_manifest"))? + .remove("family_spool"); + document["results"] = results; + let bytes = serde_json::to_vec(&document)?; + let result = + super::manifest::validate(Some(&bytes), &claim.job_id, &claim.attempt_id, &claim.stage) + .map_err(|_| anyhow::anyhow!("writer_manifest_invalid"))?; + let temporary = directory.join("output-manifest.sealed.tmp"); + std::fs::write(&temporary, &bytes)?; + std::fs::rename(temporary, directory.join("output-manifest.json"))?; + Ok(result) +} + #[cfg(test)] mod tests { use super::*; + struct SpoolDirectory(std::path::PathBuf); + impl SpoolDirectory { + fn new() -> Result { + let path = + std::env::temp_dir().join(format!("polis-graph-spool-{}", uuid::Uuid::new_v4())); + std::fs::create_dir(&path)?; + Ok(Self(path)) + } + } + impl Drop for SpoolDirectory { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } + } + fn spool_description(bytes: &[u8]) -> Value { + json!({"file":"Family.jsonl","sha256":sha256_hex(bytes)}) + } + #[test] + fn spool_regular_file_preserves_exact_bytes_and_counts() -> Result<()> { + let dir = SpoolDirectory::new()?; + let wire = b"{\"number\":1}\n"; + std::fs::write(dir.0.join("Family.jsonl"), wire)?; + let mut total = 4; + assert_eq!( + read_family_spool(&dir.0, "Family", &spool_description(wire), &mut total)?.as_bytes(), + wire + ); + assert_eq!(total, 4 + wire.len() as u64); + Ok(()) + } + #[test] + fn spool_symlink_is_refused() -> Result<()> { + let dir = SpoolDirectory::new()?; + std::fs::write(dir.0.join("target"), b"data")?; + std::os::unix::fs::symlink(dir.0.join("target"), dir.0.join("Family.jsonl"))?; + assert!(read_family_spool(&dir.0, "Family", &spool_description(b"data"), &mut 0).is_err()); + Ok(()) + } + #[test] + fn spool_fifo_is_refused_without_waiting_for_writer() -> Result<()> { + let dir = SpoolDirectory::new()?; + use std::os::unix::ffi::OsStrExt; + let name = std::ffi::CString::new(dir.0.join("Family.jsonl").as_os_str().as_bytes())?; + // SAFETY: name is a live, NUL-terminated path; mkfifo retains no pointer. + assert_eq!(unsafe { libc::mkfifo(name.as_ptr(), 0o600) }, 0); + let path = dir.0.clone(); + let (tx, rx) = std::sync::mpsc::channel(); + std::thread::spawn(move || { + let _ = tx + .send(read_family_spool(&path, "Family", &spool_description(b""), &mut 0).is_err()); + }); + assert!(rx.recv_timeout(std::time::Duration::from_secs(2))?); + Ok(()) + } + #[test] + fn spool_digest_mismatch_is_refused() -> Result<()> { + let dir = SpoolDirectory::new()?; + std::fs::write(dir.0.join("Family.jsonl"), b"changed")?; + assert!( + read_family_spool(&dir.0, "Family", &spool_description(b"original"), &mut 0).is_err() + ); + Ok(()) + } + #[test] + fn spool_sparse_oversize_is_refused_without_reading() -> Result<()> { + let dir = SpoolDirectory::new()?; + std::fs::File::create(dir.0.join("Family.jsonl"))?.set_len(67_108_865)?; + let error = read_family_spool(&dir.0, "Family", &spool_description(b""), &mut 0).err(); + assert_eq!( + error.map(|e| e.to_string()), + Some("result_spool_file:manifest TooLarge".into()) + ); + Ok(()) + } + #[test] + fn spool_path_and_total_bound_are_refused() -> Result<()> { + let dir = SpoolDirectory::new()?; + std::fs::write(dir.0.join("Family.jsonl"), b"data")?; + assert!( + read_family_spool(&dir.0, "../Family", &spool_description(b"data"), &mut 0).is_err() + ); + assert!( + read_family_spool(&dir.0, "Family", &json!({"file":"../Family.jsonl"}), &mut 0) + .is_err() + ); + assert!( + read_family_spool( + &dir.0, + "Family", + &spool_description(b"data"), + &mut 268_435_456 + ) + .is_err() + ); + let mut total = u64::MAX; + assert!( + read_family_spool(&dir.0, "Family", &spool_description(b"data"), &mut total).is_err() + ); + Ok(()) + } #[test] fn resolved_bytes_are_checked_before_dispatch() { let wire = r#"{"schema":"polis-job-input/1","declared":{},"artifacts":{}}"#; diff --git a/queue-rs/src/jobs/manifest.rs b/queue-rs/src/jobs/manifest.rs index e7480b0ba6..560c9e7365 100644 --- a/queue-rs/src/jobs/manifest.rs +++ b/queue-rs/src/jobs/manifest.rs @@ -49,6 +49,43 @@ impl std::fmt::Display for Invalid { } } +/// Open a bounded regular child output without following links or waiting on FIFOs. +pub(super) fn read_regular_file( + path: &std::path::Path, + maximum: usize, +) -> Result, Invalid> { + use std::io::Read; + use std::os::unix::fs::OpenOptionsExt; + let file = std::fs::OpenOptions::new() + .read(true) + .custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK) + .open(path) + .map_err(|error| { + if error.kind() == std::io::ErrorKind::NotFound { + Invalid::Missing + } else { + Invalid::Field("file_open") + } + })?; + let metadata = file + .metadata() + .map_err(|_| Invalid::Field("file_metadata"))?; + if !metadata.is_file() { + return Err(Invalid::Field("file_type")); + } + if metadata.len() > maximum as u64 { + return Err(Invalid::TooLarge); + } + let mut bytes = Vec::new(); + file.take(maximum as u64 + 1) + .read_to_end(&mut bytes) + .map_err(|_| Invalid::Field("file_read"))?; + if bytes.len() > maximum { + return Err(Invalid::TooLarge); + } + Ok(bytes) +} + fn is_timestamp(s: &str) -> bool { // RFC 3339 date-time with an explicit offset, e.g. 2026-10-03T12:00:00Z. let b = s.as_bytes(); @@ -144,8 +181,19 @@ pub fn validate( let text = std::str::from_utf8(bytes).map_err(|_| Invalid::NotUtf8)?; let m: Value = serde_json::from_str(text).map_err(|_| Invalid::NotJson)?; let root = m.as_object().ok_or(Invalid::Field("root"))?; - closed(root, &ROOT_KEYS, "root")?; - if m["schema"] != SCHEMA { + let writer = m["schema"] == "polis-jobs.output-manifest/2"; + if writer { + let mut keys = ROOT_KEYS.to_vec(); + keys.push(if root.contains_key("family_spool") { + "family_spool" + } else { + "results" + }); + closed(root, &keys, "root")?; + } else { + closed(root, &ROOT_KEYS, "root")?; + } + if m["schema"] != SCHEMA && !writer { return Err(Invalid::Field("schema")); } if m["job_id"] != job_id { @@ -202,7 +250,7 @@ pub fn validate( } _ => return Err(Invalid::Field("outputs")), }; - let ok = o["store"] == "dynamodb" + let ok = o["store"] == if writer { "postgres" } else { "dynamodb" } && o["family"].as_str().is_some_and(|t| !t.is_empty()) && o["table"].as_str().is_some_and(|t| !t.is_empty()) && o["rows"].as_u64().is_some() @@ -319,6 +367,23 @@ mod tests { validate(Some(&bytes), J, A, "delphi_full_pipeline") } + #[test] + fn manifest_file_reader_enforces_the_limit_before_validation() -> anyhow::Result<()> { + let path = std::env::temp_dir().join(format!("polis-manifest-{}", uuid::Uuid::new_v4())); + assert_eq!(read_regular_file(&path, MAX_BYTES), Err(Invalid::Missing)); + let file = std::fs::File::create(&path)?; + file.set_len(MAX_BYTES as u64)?; + assert_eq!( + read_regular_file(&path, MAX_BYTES) + .map_err(|e| anyhow::anyhow!("{e}"))? + .len(), + MAX_BYTES + ); + file.set_len(MAX_BYTES as u64 + 1)?; + assert_eq!(read_regular_file(&path, MAX_BYTES), Err(Invalid::TooLarge)); + std::fs::remove_file(path)?; + Ok(()) + } #[test] fn valid_manifest_keeps_exact_bytes_and_hash() { let bytes = b"{\"schema\": \"polis-jobs.output-manifest/1\" }"; @@ -340,7 +405,7 @@ mod tests { Some(Invalid::Missing) ); let mut v = fixture(J, A, "delphi_full_pipeline", "succeeded"); - v["schema"] = "polis-jobs.output-manifest/2".into(); + v["schema"] = "polis-jobs.output-manifest/99".into(); assert_eq!(check(&v).err(), Some(Invalid::Field("schema"))); let big = vec![b' '; MAX_BYTES + 1]; assert_eq!( diff --git a/queue-rs/src/jobs/task.rs b/queue-rs/src/jobs/task.rs index de48b8d87a..dc76c0034d 100644 --- a/queue-rs/src/jobs/task.rs +++ b/queue-rs/src/jobs/task.rs @@ -405,7 +405,7 @@ pub fn run(ctx: Arc, claim: Claim, reply: Value) { let manifest_path = dir.join("output-manifest.json"); let frame_path = dir.join("frame.json"); let batch = recheck.as_ref().map(|(_, b)| b.as_str()); - let frame = match child::frame(&claim, &admission, &phase, batch) { + let mut frame = match child::frame(&claim, &admission, &phase, batch) { Ok(f) => f, Err(_) => { refuse_without_child(&ctx, &mut rpc, &claim, true, "math_config_invalid"); @@ -413,6 +413,22 @@ pub fn run(ctx: Arc, claim: Claim, reply: Value) { return; } }; + if graph && super::graph::hydrate(&mut rpc, &mut frame).is_err() { + refuse_without_child(&ctx, &mut rpc, &claim, false, "result_hydration_failed"); + let _ = ctx.journal.remove(&claim.attempt_id); + return; + } + let result_writer = admission.config["result_backend"] == "postgres"; + if result_writer { + match rpc.committed("pd_writer_base", &[json!(claim.env), json!(claim.job_id)]) { + Ok(base) if base["schema"] == "delphi-writer-base/1" => frame["writer_base"] = base, + _ => { + refuse_without_child(&ctx, &mut rpc, &claim, false, "writer_base_unavailable"); + let _ = ctx.journal.remove(&claim.attempt_id); + return; + } + } + } let prepared = std::fs::create_dir_all(&dir).and_then(|_| std::fs::write(&frame_path, frame.to_string())); let command = child::command_args(&cfg.app_path, &claim, &admission, &phase); @@ -617,18 +633,26 @@ pub fn run(ctx: Arc, claim: Claim, reply: Value) { None => Exit::Code(255), }, }; - let manifest_bytes = std::fs::read(&manifest_path).ok(); + let manifest_bytes = manifest::read_regular_file(&manifest_path, manifest::MAX_BYTES); let validate = if graph { super::graph::validate } else { manifest::validate }; - let manifest: Result = validate( - manifest_bytes.as_deref(), - &claim.job_id, - &claim.attempt_id, - &claim.stage, - ); + let mut manifest: Result = manifest_bytes + .and_then(|bytes| validate(Some(&bytes), &claim.job_id, &claim.attempt_id, &claim.stage)); + if graph && outcome::decide(&exit, &manifest) == Action::Finalize { + manifest = manifest.and_then(|value| { + super::graph::stage_results(&mut rpc, &claim, &dir, value) + .map_err(|_| manifest::Invalid::Field("result_staging_failed")) + }); + } + if result_writer && outcome::decide(&exit, &manifest) == Action::Finalize { + manifest = manifest.and_then(|value| { + super::graph::stage_writer_results(&mut rpc, &claim, &dir, value) + .map_err(|_| manifest::Invalid::Field("writer_result_staging_failed")) + }); + } let mut action = outcome::decide(&exit, &manifest); let id = identity(&claim); // Provider bookkeeping while still the owner (or as the late submitter). diff --git a/queue-rs/src/lib.rs b/queue-rs/src/lib.rs index e9df6b8ced..100cb04649 100644 --- a/queue-rs/src/lib.rs +++ b/queue-rs/src/lib.rs @@ -70,6 +70,10 @@ pub fn signature_v2(name: &str) -> Result<&'static [&'static str]> { ], "pd_graph_finalize" => &["text", "uuid", "uuid", "uuid", "bigint", "text", "text"], "pd_graph_reconcile" => &["text"], + "pd_result_put_family" => &["text", "uuid", "uuid", "uuid", "bigint", "text", "text"], + "pd_result_seal" => &["text", "uuid", "uuid", "uuid", "bigint"], + "pd_writer_base" => &["text", "uuid"], + "pd_result_artifact_wire" => &["text", "uuid", "text"], "pq_claim" => &["text", "smallint", "uuid", "uuid", "integer", "text"], "pq_class_depth" => &["text", "text"], "pq_heartbeat" => &["text", "uuid", "uuid", "uuid", "bigint", "integer"], diff --git a/scripts/prove-dynamo-report.sh b/scripts/prove-dynamo-report.sh new file mode 100755 index 0000000000..54ca189d6b --- /dev/null +++ b/scripts/prove-dynamo-report.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# Local mm5 report smoke against an already published generated Postgres run. +set -euo pipefail +cd "$(dirname "$0")/.." +repo_root=$PWD +: "${COMPOSE_PROJECT_NAME:?owned local project required}" +: "${DYNAMO_PROOF_OUTPUT:?absolute evidence directory required}" +: "${DYNAMO_PROOF_REPORT_ID:?generated local report id required}" +: "${DYNAMO_PROOF_PG_DATABASE:?local generated database required}" +[[ "$COMPOSE_PROJECT_NAME" == polis-graph-test-* ]] +[[ "$DYNAMO_PROOF_REPORT_ID" == rlocal* ]] +[[ "$DYNAMO_PROOF_PG_DATABASE" == dynamo1424* ]] +[[ "$DYNAMO_PROOF_OUTPUT" = /* ]] +proof_network="${COMPOSE_PROJECT_NAME}_default" +proof_pg="${COMPOSE_PROJECT_NAME}-postgres-1" +proof_prefix="${DYNAMO_PROOF_CONTAINER_PREFIX:-astra-dynamo1424-${COMPOSE_PROJECT_NAME#polis-graph-test-}}" +[[ "$proof_prefix" == astra-dynamo1424-* ]] +proof_server="$proof_prefix-server" +proof_web="$proof_prefix-web" +proof_port="${DYNAMO_PROOF_PORT:-58244}" +proof_image="${DYNAMO_PROOF_SERVER_IMAGE:-$proof_prefix:server}" +mkdir -p "$DYNAMO_PROOF_OUTPUT" +# Each invocation owns these exact names. Refuse to replace existing processes. +if docker container inspect "$proof_server" >/dev/null 2>&1 || docker container inspect "$proof_web" >/dev/null 2>&1; then + echo "proof container name already exists; choose DYNAMO_PROOF_CONTAINER_PREFIX" >&2 + exit 2 +fi +cleanup() { + docker logs "$proof_server" > "$DYNAMO_PROOF_OUTPUT/server.log" 2>&1 || true + docker logs "$proof_web" > "$DYNAMO_PROOF_OUTPUT/nginx.log" 2>&1 || true + if [[ "${DYNAMO_PROOF_KEEP_RUNNING:-0}" != 1 ]]; then + docker rm -f "$proof_server" "$proof_web" >/dev/null 2>&1 || true + fi +} +trap cleanup EXIT +# Optional dependency caches are read-only inputs. Hosted CI can start empty. +if [[ -z "${DYNAMO_PROOF_SERVER_IMAGE:-}" ]]; then + docker build --target dev -t "$proof_image" server > "$DYNAMO_PROOF_OUTPUT/server-image-build.log" 2>&1 +fi +if [[ ! -d client-report/node_modules ]]; then + (cd client-report && npm ci --no-audit --no-fund) > "$DYNAMO_PROOF_OUTPUT/report-dependencies.log" 2>&1 +fi +if [[ -z "${DYNAMO_PROOF_PLAYWRIGHT:-}" ]]; then + proof_qa="$DYNAMO_PROOF_OUTPUT/qa" + mkdir -p "$proof_qa" + npm install --prefix "$proof_qa" --no-save --package-lock=false --no-audit --no-fund playwright@1.64.0 > "$DYNAMO_PROOF_OUTPUT/browser-dependencies.log" 2>&1 + export DYNAMO_PROOF_PLAYWRIGHT="$proof_qa/node_modules/playwright" + export PLAYWRIGHT_BROWSERS_PATH="$proof_qa/browsers" + browser_install=(install chromium) + if [[ "${CI:-}" == true ]]; then browser_install=(install --with-deps chromium); fi + node "$DYNAMO_PROOF_PLAYWRIGHT/cli.js" "${browser_install[@]}" >> "$DYNAMO_PROOF_OUTPUT/browser-dependencies.log" 2>&1 +fi +node --test ci/dynamo-removal/report-quality.test.cjs > "$DYNAMO_PROOF_OUTPUT/report-quality-unit.log" 2>&1 +# Same public OIDC settings as docker-compose.test.yml; anonymous report reads +# still use the real AuthProvider. These must exist at Webpack build time. +(cd client-report && AUTH_AUDIENCE=users AUTH_CLIENT_ID=dev-client-id \ + AUTH_ISSUER=https://localhost:3000/ AUTH_NAMESPACE=https://pol.is/ npm run build:prod) > "$DYNAMO_PROOF_OUTPUT/report-build.log" 2>&1 +sed "s/astra-dynamo1424-server/$proof_server/" ci/dynamo-removal/report-nginx.conf > "$DYNAMO_PROOF_OUTPUT/nginx.conf" +docker run -d --name "$proof_server" --network "$proof_network" \ + --label polis.proof=dynamo1424 --env-file test.env \ + -e NODE_ENV=development -e "DATABASE_URL=postgresql://postgres@$proof_pg:5432/$DYNAMO_PROOF_PG_DATABASE" \ + -e DATABASE_SSL=false -e "MATH_ENV=${DYNAMO_PROOF_ENV:-demo1424}" \ + -e DELPHI_RESULT_BACKEND=postgres -e "DELPHI_RESULT_ENV=${DYNAMO_PROOF_ENV:-demo1424}" \ + -e "DELPHI_RESULT_SCOPE=${DYNAMO_PROOF_SCOPE:-delphi}" -e POLIS_QUEUE_SUBSTRATE_ENABLED=true \ + -e DYNAMODB_ENDPOINT=http://127.0.0.1:9 -e AWS_EC2_METADATA_DISABLED=true -e DD_TRACE_ENABLED=false \ + -e "SERVICE_URL=http://localhost:$proof_port" -v "$repo_root/server:/candidate:ro" \ + --entrypoint sh "$proof_image" -c 'cp -a /candidate/. /app/; cd /app; npm run build && node dist/index.js' \ + > "$DYNAMO_PROOF_OUTPUT/server-container.txt" +for _ in $(seq 1 90); do + if docker logs "$proof_server" 2>&1 | grep 'Server started on port' >/dev/null; then break; fi + if [[ "$(docker inspect "$proof_server" --format '{{.State.Running}}')" != true ]]; then + docker logs "$proof_server" >&2;exit 1 + fi + sleep 1 +done +docker logs "$proof_server" 2>&1 | grep 'Server started on port' >/dev/null +docker run -d --name "$proof_web" --network "$proof_network" --label polis.proof=dynamo1424 \ + -p "127.0.0.1:$proof_port:8080" \ + -v "$DYNAMO_PROOF_OUTPUT/nginx.conf:/etc/nginx/nginx.conf:ro" \ + -v "$repo_root/client-report/dist:/report:ro" nginx:1.21.5-alpine > "$DYNAMO_PROOF_OUTPUT/web-container.txt" +export DYNAMO_PROOF_REPORT_URL="http://127.0.0.1:$proof_port" +for _ in $(seq 1 30); do curl -fsS "$DYNAMO_PROOF_REPORT_URL/" >/dev/null && break; sleep 1; done +node ci/dynamo-removal/report-proof.cjs > "$DYNAMO_PROOF_OUTPUT/browser.log" 2>&1 +cat "$DYNAMO_PROOF_OUTPUT/browser.log" diff --git a/scripts/test-dynamo-removal.sh b/scripts/test-dynamo-removal.sh new file mode 100755 index 0000000000..c6081520cf --- /dev/null +++ b/scripts/test-dynamo-removal.sh @@ -0,0 +1,121 @@ +#!/usr/bin/env bash +# The hosted CI and mm5 entry point. Generated/public fixture data only. +set -euo pipefail +cd "$(dirname "$0")/.." +repo=$PWD +: "${COMPOSE_PROJECT_NAME:?set a unique polis-graph-test-* project}" +: "${POLIS_RECOVERY_PG_PORT:?set an unused local port}" +: "${RECOVERY_PG_PORT:?set the same local port}" +[[ "$COMPOSE_PROJECT_NAME" == polis-graph-test-* ]] +[[ "$POLIS_RECOVERY_PG_PORT" == "$RECOVERY_PG_PORT" ]] +proof=${DYNAMO_PROOF_ROOT:-$repo/.dynamo-proof/$COMPOSE_PROJECT_NAME} +mkdir -p "$proof"; proof=$(cd "$proof" && pwd) +if [[ -e "$proof/demo-admitted.json" || -e "$proof/cluster-failed-once" ]]; then echo "Proof directory already contains a run; choose a new directory" >&2; exit 2; fi +pg="${COMPOSE_PROJECT_NAME}-postgres-1" +image="${COMPOSE_PROJECT_NAME}:worker" +base=${DELPHI_PROOF_BASE:-polis-dynamo-deps:local} +compose=(docker compose -f delphi/tests/job_graph/compose.yml) +if docker inspect "$pg" >/dev/null 2>&1; then echo 'Proof project already exists; choose a new project' >&2; exit 2; fi +owned=() +cleanup() { + for container in "${owned[@]-}"; do [[ -n "$container" ]] || continue; docker logs "$container" > "$proof/$container.log" 2>&1 || true; done + if [[ "${DYNAMO_PROOF_KEEP_RUNNING:-0}" != 1 ]]; then + for container in "${owned[@]-}"; do [[ -n "$container" ]] || continue; docker rm -f "$container" >/dev/null 2>&1 || true; done + "${compose[@]}" down -v > "$proof/cleanup.log" 2>&1 || true + fi +} +trap cleanup EXIT +# Only dependency layers may be reused. Candidate source + daemon are rebuilt. +if [[ -z "${DELPHI_PROOF_BASE:-}" ]]; then + docker build --build-context queue-rs=queue-rs -t "$base" delphi > "$proof/dependencies-build.log" 2>&1 +fi +mkdir -p "$proof/binary" "$proof/target-linux" +(cd queue-rs && cargo fmt --check && cargo test --locked --lib && cargo build --locked -p polis-migrate) > "$proof/rust-unit.log" 2>&1 +# Keep Linux process-group fencing intact: the actual daemon and Python child +# execute together in the dedicated worker container, never through docker exec. +docker run --rm --name "${COMPOSE_PROJECT_NAME}-build" \ + -v "$repo/queue-rs:/source:ro" -v "$proof/target-linux:/target" \ + -v "${DYNAMO_PROOF_CARGO_REGISTRY:-${COMPOSE_PROJECT_NAME}-cargo}:/usr/local/cargo/registry" -w /source \ + -e CARGO_TARGET_DIR=/target \ + "${DYNAMO_PROOF_RUST_IMAGE:-rust:1.98.1-slim-bookworm}" \ + bash -c 'unset RUSTUP_TOOLCHAIN; export RUSTUP_TOOLCHAIN=$(basename /usr/local/rustup/toolchains/*); apt-get update -qq && apt-get install -y -qq pkg-config libssl-dev && cargo build --locked --bin polis-jobs' > "$proof/linux-build.log" 2>&1 +cp "$proof/target-linux/debug/polis-jobs" "$proof/binary/" +docker build -f delphi/tests/dynamo_removal/Dockerfile --build-arg "DELPHI_BASE=$base" \ + --build-context "queue-binary=$proof/binary" -t "$image" . > "$proof/worker-build.log" 2>&1 +"${compose[@]}" up -d --wait +# Use the release runner and real receipts, including M28/M29. The ordinary +# API startup check below stays enabled and verifies this exact source chain. +docker exec "$pg" psql -X -U postgres -v ON_ERROR_STOP=1 -c 'CREATE DATABASE dynamo1424' +DATABASE_URL="postgresql://postgres@127.0.0.1:$POLIS_RECOVERY_PG_PORT/dynamo1424?sslmode=disable" \ + POLIS_MIGRATIONS_DIR="$repo/server/postgres/migrations" \ + "$repo/queue-rs/target/debug/polis-migrate" apply > "$proof/install.log" 2>&1 +DATABASE_URL="postgresql://postgres@127.0.0.1:$POLIS_RECOVERY_PG_PORT/dynamo1424?sslmode=disable" \ + POLIS_MIGRATIONS_DIR="$repo/server/postgres/migrations" \ + "$repo/queue-rs/target/debug/polis-migrate" check >> "$proof/install.log" 2>&1 +python3 - "$proof/sql-source-sha256.json" <<'PYHASH' +import hashlib,json,pathlib,sys +paths=sorted(pathlib.Path('server/postgres/migrations').glob('*.sql')) +pathlib.Path(sys.argv[1]).write_text(json.dumps({str(p):hashlib.sha256(p.read_bytes()).hexdigest() for p in paths},indent=2)+'\n') +PYHASH +docker exec "$pg" psql -U postgres -v ON_ERROR_STOP=1 -c 'CREATE ROLE dynamo1424_worker LOGIN; GRANT polis_queue_executor TO dynamo1424_worker' +common=(--network "container:$pg" -e DATABASE_URL=postgresql://postgres@127.0.0.1/dynamo1424 \ + -e 'QUEUE_DATABASE_URL=postgresql://dynamo1424_worker@127.0.0.1/dynamo1424?sslmode=disable' \ + -e 'MATH_CAPACITY_QUEUE_DSN=postgresql://dynamo1424_worker@127.0.0.1/dynamo1424?sslmode=disable' \ + -e MATH_CAPACITY_QUEUE_ENV=demo1424 -e DATABASE_SSL_MODE=disable -v "$proof:/proof") +docker run --rm "${common[@]}" "$image" python tests/dynamo_removal/seed.py --source-agree=+1 | tee "$proof/seed.log" +docker run --rm --network none "$image" python -m pytest --noconftest -o addopts= -q \ + tests/poller/test_enqueue_math_rebuild.py tests/test_delphi_legacy_import.py tests/test_delphi_storage_codec.py \ + tests/test_delphi_postgres_results.py tests/test_delphi_result_resource.py tests/job_graph/test_numerical_stages.py \ + tests/test_narrative_audit.py > "$proof/python-unit.log" 2>&1 +# The same server tests and SQL boundary campaign run in hosted CI and on mm5. +server_image=${DYNAMO_PROOF_SERVER_IMAGE:-${COMPOSE_PROJECT_NAME}:server} +if [[ -z "${DYNAMO_PROOF_SERVER_IMAGE:-}" ]]; then + docker build --target dev -t "$server_image" server > "$proof/server-image-build.log" 2>&1 +fi +export DYNAMO_PROOF_SERVER_IMAGE="$server_image" +docker run --rm --network none -v "$repo/server:/candidate:ro" -v "$repo/delphi:/delphi:ro" \ + -e DATABASE_URL=postgresql://postgres@127.0.0.1:1/generated --entrypoint sh "$server_image" \ + -c 'cp -a /candidate/. /app/; cd /app; npm run build && npx jest --config characterization/delphi/jest.codec.config.json --runInBand' > "$proof/server-unit.log" 2>&1 +GRAPH_PROOF_ROOT="$proof/results-sql" GRAPH_PROOF_DB=dynamo1424 uv run --no-project --with PyYAML==6.0.2 python delphi/tests/job_graph/results.py > "$proof/results-sql.log" 2>&1 +worker_common=("${common[@]}" --memory 4g --cpus 3 -e POLIS_JOBS_ENABLED=1 -e QUEUE_ENV=demo1424 \ + -e POLIS_JOBS_TRANSPORT=loopback -e POLIS_JOBS_POLL_SECONDS=1 -e POLIS_JOBS_LEASE_SECONDS=60 \ + -e POLIS_JOBS_HEARTBEAT_SECONDS=5 -e POLIS_JOBS_JOURNAL_DIR=/worker/journal -e POLIS_JOBS_WORK_DIR=/worker/jobs \ + -e DELPHI_APP_PATH=/app) +math="${COMPOSE_PROJECT_NAME}-math"; owned+=("$math") +docker run --rm "${common[@]}" "$image" python scripts/enqueue_math_rebuild.py --zid 1424 \ + --staged-label demo1424-stage --target-label demo1424 --source-commit ce1038340d608e568681d33aa3fb31f59b366dee > "$proof/math-admission.json" +docker run -d --name "$math" "${worker_common[@]}" -v "$proof/math-worker:/worker" \ + -e POLIS_JOBS_WORKER_CLASS=large -e POLIS_JOBS_STAGES=math_rebuild -e POLIS_JOBS_PYTHON=python \ + -e MATH_ENV=demo1424-stage -e MATH_POLLER_MEMORY_LIMIT_MB=4096 \ + -e MATH_POLLER_SOURCE_COMMIT=ce1038340d608e568681d33aa3fb31f59b366dee "$image" polis-jobs +job=$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["job_id"])' "$proof/math-admission.json") +docker run --rm "${common[@]}" "$image" python tests/dynamo_removal/math_proof.py --job-id "$job" \ + --zid 1424 --staged-label demo1424-stage --target-label demo1424 --wait-seconds 300 | tee "$proof/math-verification.log" +# Fixed narrative/name stand-ins exercise the same queue and result transport. +numerical=("${common[@]}" -e MATH_ENV=demo1424-stage) +docker run --rm "${numerical[@]}" "$image" python tests/dynamo_removal/admit_demo.py | tee "$proof/demo-admission.log" +delphi="${COMPOSE_PROJECT_NAME}-delphi"; owned+=("$delphi") +docker run -d --name "$delphi" "${worker_common[@]}" -v "$proof/delphi-worker:/worker" \ + -e POLIS_JOBS_WORKER_CLASS=delphi -e POLIS_JOBS_STAGES=graph_embed,graph_cluster,graph_topics,graph_narrative \ + -e POLIS_JOBS_PYTHON=/app/tests/dynamo_removal/fault_python.py \ + "$image" polis-jobs +docker run --rm "${numerical[@]}" "$image" python tests/dynamo_removal/wait_publish.py | tee "$proof/demo-verification.log" +docker run --rm "${numerical[@]}" "$image" python tests/dynamo_removal/audit_narrative.py | tee "$proof/narrative-audit.log" +# Actual DynamoDB export import; stop Dynamo before the queued import completes. +dynamo="${COMPOSE_PROJECT_NAME}-dynamo"; owned+=("$dynamo") +docker run -d --name "$dynamo" --network "container:$pg" amazon/dynamodb-local:3.3.1 -jar DynamoDBLocal.jar -sharedDb -inMemory +docker run --rm "${common[@]}" "$image" python tests/dynamo_removal/wait_dynamo.py +docker run --rm "${common[@]}" "$image" python tests/dynamo_removal/import_proof.py prepare --directory /proof/import --endpoint http://127.0.0.1:8000 +docker stop "$dynamo" +import_worker="${COMPOSE_PROJECT_NAME}-import"; owned+=("$import_worker") +docker run -d --name "$import_worker" "${worker_common[@]}" -v "$proof/import-worker:/worker" \ + -e QUEUE_ENV=proof-import -e POLIS_JOBS_WORKER_CLASS=delphi -e POLIS_JOBS_STAGES=graph_narrative -e POLIS_JOBS_PYTHON=python "$image" polis-jobs +docker run --rm "${common[@]}" "$image" python tests/dynamo_removal/import_proof.py verify --directory /proof/import --wait-seconds 300 +test "$(docker inspect "$dynamo" --format '{{.State.Running}}')" = false +# Build and render the actual report against the standard API. +export DYNAMO_PROOF_PG_DATABASE=dynamo1424 DYNAMO_PROOF_REPORT_ID=rlocaldynamo1424 +export DYNAMO_PROOF_ENV=demo1424 DYNAMO_PROOF_SCOPE=delphi DYNAMO_PROOF_OUTPUT="$proof/report" +bash scripts/prove-dynamo-report.sh +test "$(docker inspect "$dynamo" --format '{{.State.Running}}')" = false +docker inspect "$dynamo" --format '{{json .State}}' > "$proof/dynamo-stopped.json" +printf 'PASS math queue, full Delphi graph, retry/reuse, PostgreSQL report, stopped Dynamo import\n' diff --git a/scripts/test-dynamo-writers.sh b/scripts/test-dynamo-writers.sh new file mode 100644 index 0000000000..654ea9812e --- /dev/null +++ b/scripts/test-dynamo-writers.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash +# One entry point for hosted CI and isolated workbox proof. Generated data only. +set -euo pipefail +cd "$(dirname "$0")/.." +repo=$PWD +: "${COMPOSE_PROJECT_NAME:?unique polis-graph-test-* project required}" +: "${POLIS_RECOVERY_PG_PORT:?owned port required}" +: "${RECOVERY_PG_PORT:?same owned port required}" +[[ "$COMPOSE_PROJECT_NAME" == polis-graph-test-* && "$POLIS_RECOVERY_PG_PORT" == "$RECOVERY_PG_PORT" ]] +proof=${WRITER_PROOF_ROOT:-$repo/.writer-proof/$COMPOSE_PROJECT_NAME} +mkdir -p "$proof"; proof=$(cd "$proof" && pwd) +pg=${COMPOSE_PROJECT_NAME}-postgres-1 +compose=(docker compose -f delphi/tests/job_graph/compose.yml) +[[ -z "$(docker ps -aq --filter "label=com.docker.compose.project=$COMPOSE_PROJECT_NAME")" ]] +[[ ! -f "$proof/completed.json" ]] +image=${COMPOSE_PROJECT_NAME}:writer +base=${WRITER_PROOF_BASE:-${COMPOSE_PROJECT_NAME}:dependencies} +server_image=${WRITER_PROOF_SERVER_IMAGE:-${COMPOSE_PROJECT_NAME}:server} +cleanup() { + docker rm -f "${COMPOSE_PROJECT_NAME}-proof" "${COMPOSE_PROJECT_NAME}-dynamo" >/dev/null 2>&1 || true + "${compose[@]}" down -v > "$proof/cleanup.log" 2>&1 || true +} +trap cleanup EXIT +(cd queue-rs && cargo fmt --all --check && cargo test --locked --lib && cargo build --locked -p polis-migrate) 2>&1 | tee "$proof/rust.log" +if [[ -z "${WRITER_PROOF_BASE:-}" ]]; then + docker build --build-context queue-rs=queue-rs -t "$base" delphi > "$proof/dependencies-build.log" 2>&1 +fi +mkdir -p "$proof/binary" "$proof/linux-target" +docker run --rm --name "${COMPOSE_PROJECT_NAME}-build" \ + -v "$repo/queue-rs:/source:ro" -v "$proof/linux-target:/target" \ + -v "${COMPOSE_PROJECT_NAME}-cargo:/usr/local/cargo/registry" -w /source -e CARGO_TARGET_DIR=/target \ + rust:1.98.1-slim-bookworm bash -c 'unset RUSTUP_TOOLCHAIN; export RUSTUP_TOOLCHAIN=$(basename /usr/local/rustup/toolchains/*); apt-get update -qq && apt-get install -y -qq pkg-config libssl-dev && cargo build --locked --bin polis-jobs' \ + > "$proof/linux-build.log" 2>&1 +cp "$proof/linux-target/debug/polis-jobs" "$proof/binary/" +docker build -f delphi/tests/dynamo_removal/Dockerfile --build-arg "DELPHI_BASE=$base" \ + --build-context "queue-binary=$proof/binary" -t "$image" . > "$proof/worker-build.log" 2>&1 +if [[ -z "${WRITER_PROOF_SERVER_IMAGE:-}" ]]; then + docker build --target dev -t "$server_image" server > "$proof/server-build.log" 2>&1 +fi +docker run --rm --network none -v "$repo/server:/candidate:ro" -v "$repo/delphi:/delphi:ro" \ + -e DATABASE_URL=postgresql://postgres@127.0.0.1:1/generated --entrypoint sh "$server_image" \ + -c 'cp -a /candidate/. /app/; cd /app; npm run build && npx jest --config characterization/delphi/jest.codec.config.json --runInBand' \ + 2>&1 | tee "$proof/server.log" +docker run --rm --network none "$image" python -m pytest --noconftest -o addopts= -q \ + tests/test_delphi_writer.py tests/test_delphi_result_resource.py tests/test_delphi_postgres_results.py \ + tests/test_delphi_storage_codec.py tests/test_run_delphi_exit_codes.py tests/test_entrypoint_boundaries.py \ + tests/test_job_child_protocol.py 2>&1 | tee "$proof/python.log" +"${compose[@]}" up -d --wait +docker exec "$pg" psql -X -U postgres -v ON_ERROR_STOP=1 -c 'CREATE DATABASE writerproof' +DATABASE_URL="postgresql://postgres@127.0.0.1:$POLIS_RECOVERY_PG_PORT/writerproof?sslmode=disable" \ + POLIS_MIGRATIONS_DIR="$repo/server/postgres/migrations" queue-rs/target/debug/polis-migrate apply 2>&1 | tee "$proof/migrations.log" +docker exec "$pg" psql -X -U postgres -d writerproof -v ON_ERROR_STOP=1 -c 'CREATE ROLE writer_executor LOGIN; GRANT polis_queue_executor TO writer_executor' +docker run -d --name "${COMPOSE_PROJECT_NAME}-dynamo" --network "container:$pg" amazon/dynamodb-local:3.3.1 -jar DynamoDBLocal.jar -sharedDb -inMemory +docker stop "${COMPOSE_PROJECT_NAME}-dynamo" +test "$(docker inspect "${COMPOSE_PROJECT_NAME}-dynamo" --format '{{.State.Running}}')" = false +docker run --rm --name "${COMPOSE_PROJECT_NAME}-proof" --network "container:$pg" \ + -v "$proof:/proof" -e DATABASE_URL=postgresql://postgres@127.0.0.1/writerproof \ + -e 'QUEUE_DATABASE_URL=postgresql://writer_executor@127.0.0.1/writerproof?sslmode=disable' \ + -e DYNAMODB_ENDPOINT=http://127.0.0.1:8000 -e AWS_EC2_METADATA_DISABLED=true \ + "$image" python tests/dynamo_writers/prove.py 2>&1 | tee "$proof/writers.log" +docker run --rm --network "container:$pg" -v "$repo/server:/candidate:ro" -v "$repo/delphi:/delphi:ro" \ + -e DATABASE_URL=postgresql://postgres@127.0.0.1/writerproof -e DATABASE_SSL=false -e DELPHI_WRITER_PROOF=1 \ + --entrypoint sh "$server_image" -c 'cp -a /candidate/. /app/; cd /app; npx jest --config characterization/delphi/jest.codec.config.json --runInBand --forceExit delphi-postgres-writers' \ + 2>&1 | tee "$proof/server-db.log" +test "$(docker inspect "${COMPOSE_PROJECT_NAME}-dynamo" --format '{{.State.Running}}')" = false +docker inspect "${COMPOSE_PROJECT_NAME}-dynamo" --format '{{json .State}}' > "$proof/dynamo-stopped.json" +printf 'PASS PostgreSQL writers, immutable history, fenced publication, stopped DynamoDB\n' diff --git a/server/__tests__/integration/delphi-job-table.test.ts b/server/__tests__/integration/delphi-job-table.test.ts index 8350433fef..767f137af1 100644 --- a/server/__tests__/integration/delphi-job-table.test.ts +++ b/server/__tests__/integration/delphi-job-table.test.ts @@ -2,7 +2,7 @@ * The Delphi job table: migration 000023, contract polis-queue/2, and the * large worker class on top of it: migration 000024, contract polis-queue/3. * Migration 000027 adds graph stages under the install contract polis-queue/5; - * existing RPC wire versions and worker classes remain unchanged. + * M29 adds graph_topics; existing RPC wire versions and worker classes remain unchanged. * * The migrations reach the test database the way every migration does: the * postgres image applies server/postgres/migrations/*.sql once at initdb. So @@ -110,7 +110,7 @@ describe("the Delphi job table (000023) with large class (000024) and graph stag "WHERE conrelid = 'public.polis_queue_jobs'::regclass AND conname = 'polis_queue_jobs_stage_check'", ), ).toBe( - "CHECK ((stage = ANY (ARRAY['noop'::text, 'delphi_full_pipeline'::text, 'delphi_narrative'::text, 'math_rebuild'::text, 'graph_embed'::text, 'graph_cluster'::text, 'graph_narrative'::text])))", + "CHECK ((stage = ANY (ARRAY['noop'::text, 'delphi_full_pipeline'::text, 'delphi_narrative'::text, 'math_rebuild'::text, 'graph_embed'::text, 'graph_cluster'::text, 'graph_topics'::text, 'graph_narrative'::text])))", ); expect( await one( diff --git a/server/__tests__/unit/delphi-job-guard.test.ts b/server/__tests__/unit/delphi-job-guard.test.ts index 2b99284546..50f0595910 100644 --- a/server/__tests__/unit/delphi-job-guard.test.ts +++ b/server/__tests__/unit/delphi-job-guard.test.ts @@ -1,3 +1,4 @@ +jest.mock("../../src/db/pg-query", () => ({__esModule:true,default:{queryP:jest.fn()}})); /** * P-003 S3 — failure and resolution paths of the Delphi active-work guard. * diff --git a/server/__tests__/unit/delphi-postgres-liveness.test.ts b/server/__tests__/unit/delphi-postgres-liveness.test.ts new file mode 100644 index 0000000000..1f0d56d91c --- /dev/null +++ b/server/__tests__/unit/delphi-postgres-liveness.test.ts @@ -0,0 +1,61 @@ +jest.mock("../../src/db/pg-query", () => ({__esModule:true,default:{queryP:jest.fn()}})); +jest.mock("../../src/utils/logger", () => ({__esModule:true,default:{warn:jest.fn(),info:jest.fn()}})); +jest.mock("@aws-sdk/client-dynamodb", () => ({DynamoDB:jest.fn(() => {throw new Error("AWS must not be constructed");})})); +jest.mock("@aws-sdk/lib-dynamodb", () => ({DynamoDBDocument:{from:jest.fn(() => {throw new Error("AWS must not be constructed");})}})); +import pg from "../../src/db/pg-query"; +import {assessConversationLiveness} from "../../src/routes/delphi/jobGuard"; + +beforeEach(() => { + jest.clearAllMocks(); + process.env.DELPHI_RESULT_BACKEND="postgres"; + process.env.DELPHI_RESULT_ENV="generated"; + process.env.DELPHI_RESULT_SCOPE="published"; +}); +afterAll(() => { + delete process.env.DELPHI_RESULT_BACKEND; + delete process.env.DELPHI_RESULT_ENV; + delete process.env.DELPHI_RESULT_SCOPE; +}); + +test("Postgres liveness uses bound namespace and one authoritative snapshot without AWS",async () => { + (pg.queryP as jest.Mock).mockResolvedValue([ + {job_id:"finished",status:"COMPLETED",work_live:false}, + {job_id:"unconfirmed-process",status:"FAILED",work_live:true}, + {job_id:"provider-pending",status:"COMPLETED",work_live:true} + ]); + const store={sweepConversation:jest.fn()}; + const result=await assessConversationLiveness("42' OR TRUE",store as any); + expect(result.complete).toBe(true); + expect([...result.liveByJobId]).toEqual([["finished",false],["unconfirmed-process",true],["provider-pending",true]]); + expect(result.rowsByJobId.get("finished").status).toBe("COMPLETED"); + expect(store.sweepConversation).not.toHaveBeenCalled(); + expect(pg.queryP).toHaveBeenCalledTimes(1); + const [sql,values]=(pg.queryP as jest.Mock).mock.calls[0]; + expect(values).toEqual(["generated","42' OR TRUE","published"]); + expect(sql).not.toContain("42' OR TRUE"); + expect(sql).toContain("process_exit_confirmed_at IS NULL"); + expect(sql).toContain("submission_unknown"); +}); + +test("Postgres failure stays uncertain without falling back to Dynamo",async () => { + (pg.queryP as jest.Mock).mockRejectedValue(new Error("database unavailable")); + const store={sweepConversation:jest.fn()}; + const result=await assessConversationLiveness("42",store as any); + expect(result.complete).toBe(false); + expect(result.liveByJobId.size).toBe(0); + expect(store.sweepConversation).not.toHaveBeenCalled(); +}); + +test("missing result environment refuses an unscoped query",async () => { + delete process.env.DELPHI_RESULT_ENV; + await expect(assessConversationLiveness("42")).rejects.toThrow("DELPHI_RESULT_ENV"); + expect(pg.queryP).not.toHaveBeenCalled(); +}); + +test("legacy backend preserves the supplied admission store",async () => { + process.env.DELPHI_RESULT_BACKEND="dynamodb"; + const store={sweepConversation:jest.fn().mockResolvedValue({kind:"none"})}; + expect((await assessConversationLiveness("42",store as any)).complete).toBe(true); + expect(store.sweepConversation).toHaveBeenCalledWith("42"); + expect(pg.queryP).not.toHaveBeenCalled(); +}); diff --git a/server/__tests__/unit/delphi-postgres-read-routes.test.ts b/server/__tests__/unit/delphi-postgres-read-routes.test.ts new file mode 100644 index 0000000000..9aad88977f --- /dev/null +++ b/server/__tests__/unit/delphi-postgres-read-routes.test.ts @@ -0,0 +1,36 @@ +jest.mock("../../src/db/pg-query",()=>({__esModule:true,default:{queryP:jest.fn()}})); +jest.mock("../../src/utils/logger",()=>({__esModule:true,default:{info:jest.fn(),warn:jest.fn(),error:jest.fn(),debug:jest.fn()}})); +jest.mock("../../src/config",()=>({__esModule:true,default:jest.requireActual("../../src/config").default})); +jest.mock("../../src/utils/parameter",()=>({getZidFromReport:jest.fn().mockResolvedValue(1)})); +jest.mock("../../src/routes/delphi/jobGuard",()=>({assessConversationLiveness:jest.fn()})); +jest.mock("@aws-sdk/client-s3",()=>({S3Client:jest.fn(()=>{throw new Error("no S3 allowed")}),ListObjectsV2Command:jest.fn()})); +import pg from "../../src/db/pg-query"; +import {S3Client} from "@aws-sdk/client-s3"; +import {DynamoDBClient} from "@aws-sdk/client-dynamodb"; +import {assessConversationLiveness} from "../../src/routes/delphi/jobGuard"; +import {handle_GET_delphi_visualizations} from "../../src/routes/delphi/visualizations"; +import {getCurrentDelphiJobId} from "../../src/routes/delphi/topicAgenda"; +import {encodeFamily} from "../../src/utils/delphiStorageCodec"; +beforeEach(()=>{ + jest.clearAllMocks();jest.spyOn(DynamoDBClient.prototype,"send").mockRejectedValue(new Error("no DynamoDB allowed") as never);process.env.DELPHI_RESULT_BACKEND="postgres";process.env.DELPHI_RESULT_ENV="test";process.env.DELPHI_RESULT_SCOPE="delphi"; +}); +afterAll(()=>{delete process.env.DELPHI_RESULT_BACKEND;delete process.env.DELPHI_RESULT_ENV;delete process.env.DELPHI_RESULT_SCOPE;}); +test("agenda attributes selections to the published root, not a newer completed math job",async()=>{ + (pg.queryP as jest.Mock).mockResolvedValue([{job_id:"served-root"}]); + expect(await getCurrentDelphiJobId("1")).toBe("served-root"); + expect((pg.queryP as jest.Mock).mock.calls[0][0]).toContain("delphi_result_publications"); + expect((pg.queryP as jest.Mock).mock.calls[0][1]).toEqual(["test",1,"delphi"]); + expect(DynamoDBClient.prototype.send).not.toHaveBeenCalled(); +}); +test("visualization metadata is local and archived processing controls are never live work",async()=>{ + const codec_wire=encodeFamily("Delphi_JobQueue",[{job_id:{S:"historical"},conversation_id:{S:"1"},created_at:{S:"2026-01-01T00:00:00Z"},status:{S:"PROCESSING"}}]).toString(); + (pg.queryP as jest.Mock).mockResolvedValueOnce([{job_id:"current",conversation_id:"1",created_at:"2026-01-02T00:00:00Z",status:"COMPLETED"}]) + .mockResolvedValueOnce([{zid:1,scope_key:"delphi",generation:"1",codec_wire}]); + (assessConversationLiveness as jest.Mock).mockResolvedValue({complete:true,liveByJobId:new Map([["current",false]]),rowsByJobId:new Map()}); + const res:any={status:jest.fn().mockReturnThis(),json:jest.fn().mockReturnThis()}; + await handle_GET_delphi_visualizations({query:{report_id:"rlocaltest"}} as any,res); + expect(res.json.mock.calls[0][0]).toMatchObject({status:"success",visualizations:[],jobs:expect.arrayContaining([ + expect.objectContaining({jobId:"historical",status:"PROCESSING",archived:true,workLive:false}),expect.objectContaining({jobId:"current",workLive:false}) + ])}); + expect(S3Client).not.toHaveBeenCalled();expect(DynamoDBClient.prototype.send).not.toHaveBeenCalled(); +}); diff --git a/server/__tests__/unit/delphi-postgres-writers.test.ts b/server/__tests__/unit/delphi-postgres-writers.test.ts new file mode 100644 index 0000000000..dd5f4d213b --- /dev/null +++ b/server/__tests__/unit/delphi-postgres-writers.test.ts @@ -0,0 +1,50 @@ +import {randomUUID} from "crypto"; +import {PutCommand,DeleteCommand,GetCommand} from "@aws-sdk/lib-dynamodb"; + +const campaign = process.env.DELPHI_WRITER_PROOF === "1" ? describe : describe.skip; +campaign("Postgres writer real SQL boundary with Dynamo stopped",()=>{ + let pg: typeof import("../../src/db/pg-query").default; + let admitDelphiJob: typeof import("../../src/routes/delphi/jobGuard").admitDelphiJob; + let resultClient: typeof import("../../src/utils/delphiResults").resultClient; + let sendPostgresResult: typeof import("../../src/utils/delphiResults").sendPostgresResult; + beforeAll(async()=>{ + // A skipped real-DB campaign must not construct a pool at collection time. + process.env.DELPHI_RESULT_BACKEND="postgres"; + process.env.DELPHI_RESULT_ENV="writer-node-proof"; + process.env.DELPHI_RESULT_SCOPE="delphi"; + process.env.DELPHI_WRITER_CODE_SHA="d6f9ed6093e46b3c07db98f3c644bbbd9c4e87cd"; + pg = (await import("../../src/db/pg-query")).default; + ({admitDelphiJob} = await import("../../src/routes/delphi/jobGuard")); + ({resultClient,sendPostgresResult} = await import("../../src/utils/delphiResults")); + }); + test("synchronous adapters put and delete with no legacy client construction",async()=>{ + const factory=jest.fn(()=>{throw new Error("Dynamo constructed");}); + const client=resultClient(factory as any); + const key={zid_topic_jobid:"1#node#generated"}; + await client.send(new PutCommand({TableName:"Delphi_CollectiveStatement",Item:{...key,text:"node-generated"}})); + expect((await sendPostgresResult(new GetCommand({TableName:"Delphi_CollectiveStatement",Key:key}))).Item.text).toBe("node-generated"); + await client.send(new DeleteCommand({TableName:"Delphi_CollectiveStatement",Key:key})); + expect((await sendPostgresResult(new GetCommand({TableName:"Delphi_CollectiveStatement",Key:key}))).Item).toBeUndefined(); + expect(factory).not.toHaveBeenCalled(); + }); + test("both route payloads admit; non-UUID batch ids normalize; nested model/config survives",async()=>{ + const key=randomUUID(); + const request={scope:{conversationId:"2",jobType:"FULL_PIPELINE",jobConfig:JSON.stringify({include_moderation:false})}, + jobItem:{job_id:randomUUID()},idempotencyKey:key}; + const first=await admitDelphiJob(request); + expect(first.outcome).toBe("created"); + expect((await admitDelphiJob(request)).jobId).toBe(first.jobId); + expect((await admitDelphiJob({...request,scope:{...request.scope,jobConfig:'{"changed":true}'}})).outcome).toBe("idempotency_conflict"); + const nested={job_type:"CREATE_NARRATIVE_BATCH",stages:[{config:{model:"fixed-proof-model",max_batch_size:7,no_cache:true}}]}; + const batch=await admitDelphiJob({scope:{conversationId:"1",reportId:"rlocalwriter",jobType:"CREATE_NARRATIVE_BATCH",jobConfig:JSON.stringify(nested)}, + jobItem:{job_id:"batch_report_generated_123_suffix"},idempotencyKey:randomUUID()}); + expect(batch.outcome).toBe("created"); + expect(batch.jobId).toMatch(/^[0-9a-f-]{36}$/); + const config=await pg.queryP<{job_config:any}>("SELECT job_config FROM delphi_result_jobs WHERE env='writer-node-proof' AND job_id=$1",[batch.jobId]); + expect(config[0].job_config).toMatchObject({model:"fixed-proof-model",batch_size:7,no_cache:true,include_moderation:true,result_backend:"postgres"}); + }); + test("Postgres selection refuses unsupported and invalid writes without fallback",async()=>{ + await expect(sendPostgresResult(new PutCommand({TableName:"Delphi_JobQueue",Item:{job_id:"generated"}}))).rejects.toThrow(); + await expect(sendPostgresResult(new PutCommand({TableName:"Delphi_CollectiveStatement",Item:{zid_topic_jobid:"bad#key"}}))).rejects.toThrow(); + }); +}); diff --git a/server/__tests__/unit/delphiResultSnapshot.test.ts b/server/__tests__/unit/delphiResultSnapshot.test.ts new file mode 100644 index 0000000000..9a36207f3f --- /dev/null +++ b/server/__tests__/unit/delphiResultSnapshot.test.ts @@ -0,0 +1,55 @@ +import { EventEmitter } from "events"; +import pg from "../../src/db/pg-query"; +import { delphiResultSnapshot, resultQuery } from "../../src/utils/delphiResultSnapshot"; +jest.mock("../../src/db/pg-query", () => ({__esModule:true, default:{connectResultSnapshot:jest.fn(),queryP:jest.fn()}})); +jest.mock("../../src/utils/logger", () => ({__esModule:true,default:{error:jest.fn()}})); +const tick = () => new Promise(resolve => setImmediate(resolve)); +beforeEach(() => { jest.clearAllMocks(); process.env.DELPHI_RESULT_BACKEND="postgres"; }); +afterAll(() => { delete process.env.DELPHI_RESULT_BACKEND; }); +test("parallel family reads share a transaction and finish/close release once", async () => { + const client = Object.assign(new EventEmitter(), {query:jest.fn().mockResolvedValue({rows:[{value:1}]}),release:jest.fn()}); + (pg.connectResultSnapshot as jest.Mock).mockResolvedValue(client); + const res = new EventEmitter(); + await new Promise((resolve,reject) => delphiResultSnapshot({} as any,res as any, (() => { + Promise.all([resultQuery("SELECT topics"),resultQuery("SELECT assignments")]).then(() => resolve(),reject); + }) as any)); + expect(pg.connectResultSnapshot).toHaveBeenCalledTimes(1); + expect(client.query.mock.calls[0][0]).toBe("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY"); + res.emit("finish"); res.emit("close"); await tick(); + expect(client.query.mock.calls.filter(([sql]) => sql === "ROLLBACK")).toHaveLength(1); + expect(client.release).toHaveBeenCalledTimes(1); +}); +test("requests without results acquire no client, and failed setup releases it", async () => { + const empty = new EventEmitter(); + delphiResultSnapshot({} as any,empty as any, (()=>{}) as any); + empty.emit("close"); expect(pg.connectResultSnapshot).not.toHaveBeenCalled(); + const failure = new Error("begin failed"); + const client = Object.assign(new EventEmitter(), {query:jest.fn().mockRejectedValue(failure),release:jest.fn()}); + (pg.connectResultSnapshot as jest.Mock).mockResolvedValue(client); + const res = new EventEmitter(); + await new Promise((resolve,reject) => delphiResultSnapshot({} as any,res as any, (() => { + resultQuery("SELECT topics").then(()=>reject(new Error("expected failure")),error=>{ + expect(error).toBe(failure); resolve(); + }); + }) as any)); + res.emit("close"); await tick(); + expect(client.release).toHaveBeenCalledTimes(1); + expect(client.release).toHaveBeenCalledWith(failure); +}); + +test("connection errors cannot crash the process or reuse a failed snapshot", async () => { + const client = Object.assign(new EventEmitter(), {query:jest.fn().mockResolvedValue({rows:[]}),release:jest.fn()}); + (pg.connectResultSnapshot as jest.Mock).mockResolvedValue(client); + const res = new EventEmitter(); + const failure = new Error("idle transaction terminated"); + await new Promise((resolve,reject) => delphiResultSnapshot({} as any,res as any, (() => { + (async () => { + await resultQuery("SELECT topics"); + client.emit("error",failure); + await expect(resultQuery("SELECT assignments")).rejects.toBe(failure); + resolve(); + })().catch(reject); + }) as any)); + res.emit("finish"); await tick(); + expect(client.release).toHaveBeenCalledWith(failure); +}); diff --git a/server/__tests__/unit/delphiResults.test.ts b/server/__tests__/unit/delphiResults.test.ts new file mode 100644 index 0000000000..fd60535362 --- /dev/null +++ b/server/__tests__/unit/delphiResults.test.ts @@ -0,0 +1,103 @@ +jest.mock("../../src/db/pg-query", () => ({__esModule:true,default:{queryP:jest.fn()}})); +import pg from "../../src/db/pg-query"; +import {sendPostgresResult,sqlPredicate,resultClient,decodeAttribute} from "../../src/utils/delphiResults"; +import {encodeFamily} from "../../src/utils/delphiStorageCodec"; +class QueryCommand { constructor(public input:any) {} } +class GetCommand { constructor(public input:any) {} } +const family="Delphi_CommentEmbeddings"; +const row=(id:number,generation="1",scope_key="scope")=>({zid:1,scope_key,generation,item:{conversation_id:{S:"1"},comment_id:{N:String(id)}}}); + +beforeEach(()=>{jest.clearAllMocks();process.env.DELPHI_RESULT_BACKEND="postgres";process.env.DELPHI_RESULT_ENV="generated";delete process.env.DELPHI_RESULT_SCOPE;}); +afterAll(()=>{delete process.env.DELPHI_RESULT_BACKEND;delete process.env.DELPHI_RESULT_ENV;}); + +test("key expressions use parameters for untrusted values",()=>{ + const values:any[]=[]; + const text=sqlPredicate("#c = :c AND begins_with(topic_key, :prefix)",{ExpressionAttributeNames:{"#c":"conversation_id"},ExpressionAttributeValues:{":c":"1' OR TRUE",":prefix":"%_"}},values); + expect(text).not.toContain("1' OR TRUE");expect(text).toContain("starts_with");expect(values).toContain('%_'); +}); +test("pagination carries generation and refuses mixed snapshots",async()=>{ + (pg.queryP as jest.Mock).mockResolvedValue([row(0),row(1)]); + const first=await sendPostgresResult(new QueryCommand({TableName:family,Limit:1})); + expect(first.Items[0].comment_id).toBe(0); + const second=await sendPostgresResult(new QueryCommand({TableName:family,Limit:1,ExclusiveStartKey:first.LastEvaluatedKey})); + expect(second.Items[0].comment_id).toBe(1); + (pg.queryP as jest.Mock).mockResolvedValue([row(0,"2"),row(1,"2")]); + await expect(sendPostgresResult(new QueryCommand({TableName:family,ExclusiveStartKey:first.LastEvaluatedKey}))).rejects.toThrow("generation changed"); +}); +test("SQL is scoped to environment and configured publication",async()=>{ + process.env.DELPHI_RESULT_SCOPE="selected";(pg.queryP as jest.Mock).mockResolvedValue([row(0)]); + await sendPostgresResult(new QueryCommand({TableName:family,KeyConditionExpression:"conversation_id = :id",ExpressionAttributeValues:{":id":"1"}})); + expect((pg.queryP as jest.Mock).mock.calls[0][1]).toEqual(['generated',family,'selected','conversation_id',JSON.stringify({S:'1'})]); +}); +test("conflicting published scopes fail visibly",async()=>{ + (pg.queryP as jest.Mock).mockResolvedValue([row(0),{...row(0,"1","other"),item:{...row(0).item,text:{S:"different"}}}]); + await expect(sendPostgresResult(new QueryCommand({TableName:family}))).rejects.toThrow("Ambiguous"); +}); +test("Postgres error never creates an AWS client",async()=>{ + const factory=jest.fn();(pg.queryP as jest.Mock).mockRejectedValue(new Error("database unavailable")); + const client=resultClient(factory); + await expect(client.send(new QueryCommand({TableName:family}))).rejects.toThrow("database unavailable"); + expect(factory).not.toHaveBeenCalled(); +}); +test("unsafe integral numeric values fail instead of rounding",()=>{ + expect(()=>decodeAttribute({N:"9007199254740993"})).toThrow("precision"); +}); + +test("ConversationIndex selects newest job by creation time, not UUID",async()=>{ + (pg.queryP as jest.Mock).mockResolvedValueOnce([ + {job_id:"z-older",conversation_id:"1",created_at:"2025-01-01T00:00:00Z",status:"COMPLETED"}, + {job_id:"a-newest",conversation_id:"1",created_at:"2026-01-01T00:00:00Z",status:"COMPLETED"} + ]).mockResolvedValueOnce([]); + const reply=await sendPostgresResult(new QueryCommand({TableName:"Delphi_JobQueue",IndexName:"ConversationIndex",ScanIndexForward:false,Limit:1})); + expect(reply.Items[0].job_id).toBe("a-newest"); +}); + +test("archived metadata preserves NUL, exposes exact unsafe decimal and never replaces a current queue row",async()=>{ + const codec_wire=encodeFamily("Delphi_JobQueue",[ + {job_id:{S:"legacy"},conversation_id:{S:"1"},status:{S:"PROCESSING"},logs:{S:"a\0b"},job_config:{M:{large:{N:"9007199254740993"}}}}, + {job_id:{S:"current"},status:{S:"PROCESSING"}} + ]).toString(); + (pg.queryP as jest.Mock).mockResolvedValueOnce([{job_id:"current",status:"COMPLETED"}]) + .mockResolvedValueOnce([{zid:1,scope_key:"scope",generation:"4",codec_wire}]); + const reply=await sendPostgresResult(new QueryCommand({TableName:"Delphi_JobQueue"})); + const legacy=reply.Items.find((item:any)=>item.job_id==="legacy"); + expect(legacy).toMatchObject({archived:true,logs:"a\0b",status:"PROCESSING",job_config:{large:"9007199254740993"}}); + expect(legacy.unreadable_metadata_reason).toContain("precision"); + expect(legacy.legacy_control_item.job_config.M.large.N).toBe("9007199254740993"); + expect(reply.Items.find((item:any)=>item.job_id==="current").status).toBe("COMPLETED"); +}); + +test("modern job metadata uses the same graph scope as liveness and archived controls",async()=>{ + process.env.DELPHI_RESULT_SCOPE="delphi"; + (pg.queryP as jest.Mock).mockResolvedValue([]); + await sendPostgresResult(new QueryCommand({TableName:"Delphi_JobQueue"})); + expect((pg.queryP as jest.Mock).mock.calls).toHaveLength(2); + for (const [sql,values] of (pg.queryP as jest.Mock).mock.calls) { + expect(sql).toContain("scope_key=$2");expect(values).toEqual(["generated","delphi"]); + } +}); + +test("job get binds current attempt logs while job listings do not fetch log payloads",async()=>{ + const job_id="11111111-1111-1111-1111-111111111111",attempt="22222222-2222-2222-2222-222222222222"; + (pg.queryP as jest.Mock).mockResolvedValueOnce([{job_id,conversation_id:"1"}]).mockResolvedValueOnce([]) + .mockResolvedValueOnce([{value:{attempt_id:attempt}}]).mockResolvedValueOnce([{timestamp:"2026-01-01T00:00:00Z",level:"INFO",message:"actual child output"}]); + const reply=await sendPostgresResult(new GetCommand({TableName:"Delphi_JobQueue",Key:{job_id}})); + expect(reply.Item.logs.entries[0].message).toBe("actual child output");expect(reply.Item.log_attempt_id).toBe(attempt); + const calls=(pg.queryP as jest.Mock).mock.calls; + expect(calls[2][1]).toEqual(["generated",job_id]);expect(calls[3][1]).toEqual(["generated",attempt]); + expect(calls[3][0]).toContain("pq_attempt_logs($1::text,$2::uuid,NULL,1000)"); + expect(calls[3][0]).toContain("stream IN ('stdout','stderr')"); +}); + +// A merge must preserve the existing reader without any activation setting. +test.each([undefined, "dynamodb"])("backend %s forwards unchanged to DynamoDB", async backend => { + if (backend === undefined) delete process.env.DELPHI_RESULT_BACKEND; + else process.env.DELPHI_RESULT_BACKEND = backend; + const command = new QueryCommand({TableName:family}); + const response = {Items:[{legacy:true}]}; + const send = jest.fn().mockResolvedValue(response); + const factory = jest.fn(() => ({send})); + expect(await resultClient(factory).send(command)).toBe(response); + expect(send).toHaveBeenCalledWith(command); + expect(pg.queryP).not.toHaveBeenCalled(); +}); diff --git a/server/__tests__/unit/delphiStoragePagination.test.ts b/server/__tests__/unit/delphiStoragePagination.test.ts new file mode 100644 index 0000000000..3b5bc6e2a6 --- /dev/null +++ b/server/__tests__/unit/delphiStoragePagination.test.ts @@ -0,0 +1,49 @@ +jest.mock("../../src/db/pg-query", () => ({__esModule:true,default:{queryP:jest.fn()}})); +jest.mock("../../src/config", () => ({__esModule:true,default:jest.requireActual("../../src/config").default})); +jest.mock("../../src/utils/logger", () => ({__esModule:true,default:{debug:jest.fn(),error:jest.fn(),warn:jest.fn()}})); +jest.mock("../../src/utils/dynamoClient", () => ({makeDynamoClient:jest.fn(() => {throw new Error("DynamoDB must not be constructed");})})); +import pg from "../../src/db/pg-query"; +import {makeDynamoClient} from "../../src/utils/dynamoClient"; +import DynamoStorageService from "../../src/utils/storage"; + +const row=(key:string,report_data:string,generation="1") => ({ + zid:1,scope_key:"report",generation,item:{ + rid_section_model:{S:key},timestamp:{S:"2026-01-01T00:00:00Z"},report_data:{S:report_data} + } +}); +beforeEach(() => { + jest.clearAllMocks(); + process.env.DELPHI_RESULT_BACKEND="postgres"; + process.env.DELPHI_RESULT_ENV="generated"; + process.env.DELPHI_RESULT_SCOPE="report"; +}); +afterAll(() => { + delete process.env.DELPHI_RESULT_BACKEND; + delete process.env.DELPHI_RESULT_ENV; + delete process.env.DELPHI_RESULT_SCOPE; +}); + +test("report reader follows an empty filtered page and retains later row order and JSON string bytes",async () => { + const bytes=['{"text":"first Ω","nested":[1,2]}','{"text":"second\\nline"}']; + const rows=Array.from({length:1000},(_,i)=>row(`generated-other#${String(i).padStart(4,"0")}`,"unrelated")); + rows.push(row("rlocalpage#section-a",bytes[0]),row("rlocalpage#section-b",bytes[1])); + (pg.queryP as jest.Mock).mockResolvedValue(rows); + const result=await new DynamoStorageService("report_narrative_store").getAllByReportID("rlocalpage#"); + expect(result.success).toBe(true); + expect(result.data?.map(item=>item.rid_section_model)).toEqual(["rlocalpage#section-a","rlocalpage#section-b"]); + expect(result.data?.map(item=>item.report_data)).toEqual(bytes); + expect(pg.queryP).toHaveBeenCalledTimes(2); + expect(makeDynamoClient).not.toHaveBeenCalled(); +}); + +test("a changed generation on a later page refuses partial report success",async () => { + const rows=Array.from({length:1001},(_,i)=>row(`rlocalpage#${String(i).padStart(4,"0")}`,"stored")); + (pg.queryP as jest.Mock).mockResolvedValueOnce(rows) + .mockResolvedValueOnce(rows.map(item=>({...item,generation:"2"}))); + const result=await new DynamoStorageService("report_narrative_store").getAllByReportID("rlocalpage#"); + expect(result.success).toBe(false); + expect(result.data).toBeUndefined(); + expect(result.error?.message).toContain("generation changed"); + expect(pg.queryP).toHaveBeenCalledTimes(2); + expect(makeDynamoClient).not.toHaveBeenCalled(); +}); diff --git a/server/app.ts b/server/app.ts index 23121b6f39..60f2ab2a39 100644 --- a/server/app.ts +++ b/server/app.ts @@ -14,6 +14,7 @@ import morgan from "morgan"; import timeout from "connect-timeout"; import server from "./src/server"; +import { delphiResultSnapshot } from "./src/utils/delphiResultSnapshot"; import Config from "./src/config"; import { makeFileFetcher } from "./src/utils/file-fetcher"; import logger from "./src/utils/logger"; @@ -325,6 +326,7 @@ export const appReady = helpersInitialized.then( //////////////////////////////////////////// app.use(middleware_responseTime_start); + app.use(delphiResultSnapshot); app.use(redirectIfNotHttps); app.use(express.bodyParser({ limit: "50mb" })); diff --git a/server/bin/build-migration-report.py b/server/bin/build-migration-report.py index 1acf91ce4e..d6969b5c0f 100644 --- a/server/bin/build-migration-report.py +++ b/server/bin/build-migration-report.py @@ -81,7 +81,8 @@ def manifest(path): selected = manifest(MIG/'release.txt') held = manifest(MIG/'held.txt') pending = {'000019_create_polis_queue.sql', '000023_create_delphi_foundation.sql', - '000024_create_polis_queue_large_class.sql', '000027_create_sealed_job_graphs.sql'} + '000024_create_polis_queue_large_class.sql', '000027_create_sealed_job_graphs.sql', '000028_create_delphi_results.sql', + '000029_extend_delphi_graph_stages.sql', '000030_create_delphi_writers.sql'} assert selected == {p.name for p in files} | pending, 'report scope differs from release selection' assert held == {'000021_create_polis_coordinator.sql'}, 'review changed hold policy' numbered = sorted(MIG.glob('*.sql')) @@ -149,7 +150,10 @@ def manifest(path): ELSE 'WOULD_APPLY' END FROM (VALUES ('000019_create_polis_queue.sql'),('000023_create_delphi_foundation.sql'), ('000024_create_polis_queue_large_class.sql'), - ('000027_create_sealed_job_graphs.sql')) p(migration) CROSS JOIN readiness + ('000027_create_sealed_job_graphs.sql'), + ('000028_create_delphi_results.sql'), + ('000029_extend_delphi_graph_stages.sql'), + ('000030_create_delphi_writers.sql')) p(migration) CROSS JOIN readiness UNION ALL SELECT migration, NULL::boolean, 'OUTSIDE_RELEASE' FROM (VALUES ('000020'),('000021'),('000025'),('000026')) p(migration) diff --git a/server/characterization/delphi/jest.codec.config.json b/server/characterization/delphi/jest.codec.config.json index f94ac9f2db..3f2c327207 100644 --- a/server/characterization/delphi/jest.codec.config.json +++ b/server/characterization/delphi/jest.codec.config.json @@ -1,6 +1,22 @@ { "rootDir": "../../", - "transform": { "^.+\\.ts$": ["ts-jest", { "tsconfig": "./tsconfig.json" }] }, + "transform": { + "^.+\\.ts$": [ + "ts-jest", + { + "tsconfig": "./tsconfig.json" + } + ] + }, "testEnvironment": "node", - "testMatch": ["**/__tests__/unit/delphiStorageCodec.test.ts"] + "testMatch": [ + "**/__tests__/unit/delphiStorageCodec.test.ts", + "**/__tests__/unit/delphiStoragePagination.test.ts", + "**/__tests__/unit/delphiResults.test.ts", + "**/__tests__/unit/delphiResultSnapshot.test.ts", + "**/__tests__/unit/delphi-postgres-liveness.test.ts", + "**/__tests__/unit/delphi-job-guard.test.ts", + "**/__tests__/unit/delphi-postgres-read-routes.test.ts", + "**/__tests__/unit/delphi-postgres-writers.test.ts" + ] } diff --git a/server/postgres/migrations/000028_create_delphi_results.sql b/server/postgres/migrations/000028_create_delphi_results.sql new file mode 100644 index 0000000000..a25a384c18 --- /dev/null +++ b/server/postgres/migrations/000028_create_delphi_results.sql @@ -0,0 +1,255 @@ +-- Draft #1428, step 7. Run-bound, immutable Delphi results on graph contract /5. +-- Does not rewrite historical migrations or mutate any vote or result row. +BEGIN; +SET LOCAL lock_timeout='5s'; +GRANT SELECT(report_id,zid) ON public.reports TO polis_queue_owner; +SET LOCAL ROLE polis_queue_owner; +SET LOCAL search_path=pg_catalog,pg_temp; +CREATE TABLE public.delphi_result_catalog(family text PRIMARY KEY, key_spec jsonb NOT NULL); +INSERT INTO public.delphi_result_catalog VALUES +('Delphi_PCAConversationConfig','[["zid", "S"]]'::jsonb), +('Delphi_PCAResults','[["zid", "S"], ["math_tick", "N"]]'::jsonb), +('Delphi_KMeansClusters','[["zid_tick", "S"], ["group_id", "N"]]'::jsonb), +('Delphi_CommentRouting','[["zid_tick", "S"], ["comment_id", "S"]]'::jsonb), +('Delphi_RepresentativeComments','[["zid_tick_gid", "S"], ["comment_id", "S"]]'::jsonb), +('Delphi_PCAParticipantProjections','[["zid_tick", "S"], ["participant_id", "S"]]'::jsonb), +('Delphi_UMAPConversationConfig','[["conversation_id", "S"]]'::jsonb), +('Delphi_CommentEmbeddings','[["conversation_id", "S"], ["comment_id", "N"]]'::jsonb), +('Delphi_CommentHierarchicalClusterAssignments','[["conversation_id", "S"], ["comment_id", "N"]]'::jsonb), +('Delphi_CommentClustersStructureKeywords','[["conversation_id", "S"], ["cluster_key", "S"]]'::jsonb), +('Delphi_UMAPGraph','[["conversation_id", "S"], ["edge_id", "S"]]'::jsonb), +('Delphi_CommentClustersFeatures','[["conversation_id", "S"], ["cluster_key", "S"]]'::jsonb), +('Delphi_CommentClustersLLMTopicNames','[["conversation_id", "S"], ["topic_key", "S"]]'::jsonb), +('Delphi_NarrativeReports','[["rid_section_model", "S"], ["timestamp", "S"]]'::jsonb), +('Delphi_CommentExtremity','[["conversation_id", "S"], ["comment_id", "S"]]'::jsonb), +('Delphi_TopicAgendaSelections','[["conversation_id", "S"], ["participant_id", "S"]]'::jsonb), +('Delphi_CollectiveStatement','[["zid_topic_jobid", "S"]]'::jsonb), +('report_narrative_store','[["rid_section_model", "S"], ["timestamp", "S"]]'::jsonb); +CREATE TABLE public.delphi_result_batches( + env text NOT NULL, batch_id uuid NOT NULL, job_id uuid NOT NULL, run_id uuid NOT NULL, + sealed_sha text, artifact_id uuid, created_at timestamptz NOT NULL DEFAULT clock_timestamp(), + PRIMARY KEY(env,batch_id), UNIQUE(env,artifact_id), + FOREIGN KEY(env,job_id,run_id) REFERENCES public.delphi_graph_nodes(env,job_id,run_id), + FOREIGN KEY(env,job_id,batch_id) REFERENCES public.polis_queue_attempts(env,job_id,attempt_id), + FOREIGN KEY(env,artifact_id) REFERENCES public.delphi_artifacts(env,artifact_id) +); +CREATE TABLE public.delphi_result_families( + env text NOT NULL,batch_id uuid NOT NULL,family text NOT NULL REFERENCES public.delphi_result_catalog, + codec_wire text NOT NULL, content_sha text NOT NULL, row_count integer NOT NULL CHECK(row_count>=0), + PRIMARY KEY(env,batch_id,family), FOREIGN KEY(env,batch_id) REFERENCES public.delphi_result_batches, + CHECK(content_sha=encode(sha256(convert_to(codec_wire,'UTF8')),'hex')) +); +CREATE TABLE public.delphi_result_rows( + env text NOT NULL,batch_id uuid NOT NULL,family text NOT NULL,item_key jsonb NOT NULL, + ordinal integer NOT NULL CHECK(ordinal>0),item jsonb NOT NULL CHECK(jsonb_typeof(item)='object'), + PRIMARY KEY(env,batch_id,family,item_key), UNIQUE(env,batch_id,family,ordinal), + FOREIGN KEY(env,batch_id,family) REFERENCES public.delphi_result_families +); +CREATE TRIGGER result_family_immutable BEFORE UPDATE OR DELETE ON public.delphi_result_families + FOR EACH ROW EXECUTE FUNCTION public.pd_graph_immutable(); +CREATE TRIGGER result_row_immutable BEFORE UPDATE OR DELETE ON public.delphi_result_rows + FOR EACH ROW EXECUTE FUNCTION public.pd_graph_immutable(); +-- Validate AttributeValue tags before making a result readable. No numeric value +-- passes through a float; the immutable original bytes remain the audit record. +CREATE FUNCTION public.pd_result_valid_value(v jsonb) RETURNS boolean +LANGUAGE plpgsql IMMUTABLE SET search_path=pg_catalog,pg_temp AS $$ +DECLARE tag text;body jsonb;x jsonb;t text;n numeric; +BEGIN + IF jsonb_typeof(v)<>'object' OR (SELECT count(*) FROM jsonb_object_keys(v))<>1 THEN RETURN false; END IF; + SELECT key,value INTO tag,body FROM jsonb_each(v); + IF tag='S' THEN RETURN jsonb_typeof(body)='string'; + ELSIF tag='BOOL' THEN RETURN jsonb_typeof(body)='boolean'; + ELSIF tag='NULL' THEN RETURN body='true'::jsonb; + ELSIF tag='N' THEN + IF jsonb_typeof(body)<>'string' THEN RETURN false; END IF; + t=body#>>'{}'; + IF length(t)>256 OR t !~ '^-?(0|[1-9][0-9]*)([.][0-9]*[1-9])?$' OR t='-0' THEN RETURN false; END IF; + n=t::numeric; + RETURN (n=0 OR (abs(n)>=1e-130::numeric AND abs(n)<1e126::numeric)) + AND length(trim(both '0' from replace(replace(t,'-',''),'.','')))<=38; + ELSIF tag='B' THEN + IF jsonb_typeof(body)<>'string' THEN RETURN false; END IF; + t=body#>>'{}';RETURN replace(encode(decode(t,'base64'),'base64'),chr(10),'')=t; + ELSIF tag='M' THEN + IF jsonb_typeof(body)<>'object' THEN RETURN false; END IF; + FOR x IN SELECT value FROM jsonb_each(body) LOOP IF NOT public.pd_result_valid_value(x) THEN RETURN false; END IF; END LOOP; + ELSIF tag IN ('L','SS','NS','BS') THEN + IF jsonb_typeof(body)<>'array' THEN RETURN false; END IF; + IF tag<>'L' AND (jsonb_array_length(body)=0 OR (SELECT count(DISTINCT value) FROM jsonb_array_elements(body))<>jsonb_array_length(body)) THEN RETURN false; END IF; + FOR x IN SELECT value FROM jsonb_array_elements(body) LOOP + IF NOT public.pd_result_valid_value(CASE tag WHEN 'L' THEN x ELSE jsonb_build_object(left(tag,1),x) END) THEN RETURN false; END IF; + END LOOP; + ELSE RETURN false; END IF; + RETURN true; +EXCEPTION WHEN OTHERS THEN RETURN false; +END $$; +-- Internal insertion helper; caller has already acquired the queue row lock and fence. +CREATE FUNCTION public.pd_result_insert_family(p_env text,p_job uuid,p_attempt uuid,p_family text,p_wire text) +RETURNS jsonb LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp AS $$ +#variable_conflict use_column +DECLARE n public.delphi_graph_nodes;b public.delphi_result_batches;spec jsonb;head jsonb; + lines text[];rowdoc jsonb;k jsonb;part jsonb;i integer;sha text;old public.delphi_result_families; +BEGIN + SELECT * INTO STRICT n FROM public.delphi_graph_nodes WHERE env=p_env AND job_id=p_job; + SELECT key_spec INTO STRICT spec FROM public.delphi_result_catalog WHERE family=p_family; + IF p_wire IS NULL OR octet_length(p_wire)>67108864 OR right(p_wire,1)<>chr(10) THEN RAISE EXCEPTION 'invalid result family bytes'; END IF; + sha=encode(sha256(convert_to(p_wire,'UTF8')),'hex'); + INSERT INTO public.delphi_result_batches(env,batch_id,job_id,run_id) VALUES(p_env,p_attempt,p_job,n.run_id) ON CONFLICT DO NOTHING; + SELECT * INTO STRICT b FROM public.delphi_result_batches WHERE env=p_env AND batch_id=p_attempt FOR UPDATE; + IF b.job_id<>p_job OR b.run_id<>n.run_id THEN RAISE EXCEPTION 'result run binding'; END IF; + SELECT * INTO old FROM public.delphi_result_families WHERE env=p_env AND batch_id=p_attempt AND family=p_family; + IF FOUND THEN + IF old.codec_wire<>p_wire THEN RAISE EXCEPTION 'immutable result family conflict'; END IF; + RETURN jsonb_build_object('sha256',old.content_sha,'row_count',old.row_count); + END IF; + IF b.sealed_sha IS NOT NULL THEN RAISE EXCEPTION 'result batch sealed'; END IF; + IF COALESCE((SELECT sum(octet_length(codec_wire)) FROM public.delphi_result_families WHERE env=p_env AND batch_id=p_attempt),0)+octet_length(p_wire)>268435456 THEN RAISE EXCEPTION 'result batch byte limit'; END IF; + lines=string_to_array(left(p_wire,length(p_wire)-1),chr(10));head=lines[1]::jsonb; + IF head IS DISTINCT FROM jsonb_build_object('codec','delphi-storage-codec/1','family',p_family,'key',(SELECT jsonb_agg(x->0) FROM jsonb_array_elements(spec) x)) + THEN RAISE EXCEPTION 'invalid result codec header'; END IF; + INSERT INTO public.delphi_result_families VALUES(p_env,p_attempt,p_family,p_wire,sha,cardinality(lines)-1); + FOR i IN 2..cardinality(lines) LOOP + rowdoc=lines[i]::jsonb;k='[]'::jsonb; + IF jsonb_typeof(rowdoc)<>'object' OR EXISTS(SELECT 1 FROM jsonb_each(rowdoc) a WHERE a.key='' OR NOT public.pd_result_valid_value(a.value)) THEN RAISE EXCEPTION 'invalid result row'; END IF; + FOR part IN SELECT value FROM jsonb_array_elements(spec) LOOP + IF rowdoc->(part->>0) IS NULL OR jsonb_typeof(rowdoc->(part->>0))<>'object' + OR NOT (rowdoc->(part->>0) ? (part->>1)) OR (SELECT count(*) FROM jsonb_object_keys(rowdoc->(part->>0)))<>1 + OR jsonb_typeof(rowdoc->(part->>0)->(part->>1))<>'string' THEN RAISE EXCEPTION 'invalid result key'; END IF; + k=k||jsonb_build_array(rowdoc->(part->>0)); + END LOOP; + IF (rowdoc ? 'conversation_id' AND rowdoc->'conversation_id'->>'S' IS DISTINCT FROM n.zid::text) + OR (rowdoc ? 'zid' AND COALESCE(rowdoc->'zid'->>'S',rowdoc->'zid'->>'N') IS DISTINCT FROM n.zid::text) + OR (rowdoc ? 'zid_tick' AND split_part(rowdoc->'zid_tick'->>'S',':',1)<>n.zid::text) + OR (rowdoc ? 'zid_tick_gid' AND split_part(rowdoc->'zid_tick_gid'->>'S',':',1)<>n.zid::text) + OR (rowdoc ? 'zid_topic_jobid' AND split_part(rowdoc->'zid_topic_jobid'->>'S','#',1)<>n.zid::text) + THEN RAISE EXCEPTION 'result conversation mismatch'; END IF; + IF rowdoc ? 'rid_section_model' AND NOT EXISTS( + SELECT 1 FROM public.reports r WHERE r.zid=n.zid AND r.report_id=split_part(rowdoc->'rid_section_model'->>'S','#',1) + AND (NOT rowdoc ? 'report_id' OR rowdoc->'report_id'->>'S'=r.report_id)) + THEN RAISE EXCEPTION 'result report conversation mismatch'; END IF; + INSERT INTO public.delphi_result_rows VALUES(p_env,p_attempt,p_family,k,i-1,rowdoc); + END LOOP; + RETURN jsonb_build_object('sha256',sha,'row_count',cardinality(lines)-1); +END $$; +CREATE FUNCTION public.pd_result_seal_internal(p_env text,p_attempt uuid) RETURNS jsonb +LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp AS $$ +DECLARE digest text;names jsonb; +BEGIN + SELECT encode(sha256(convert_to(COALESCE(string_agg(family||':'||content_sha||':'||row_count::text,chr(10) ORDER BY family COLLATE "C"),''),'UTF8')),'hex'), + COALESCE(jsonb_agg(family ORDER BY family COLLATE "C"),'[]'::jsonb) INTO digest,names + FROM public.delphi_result_families WHERE env=p_env AND batch_id=p_attempt; + UPDATE public.delphi_result_batches SET sealed_sha=digest WHERE env=p_env AND batch_id=p_attempt AND sealed_sha IS NULL; + IF NOT EXISTS(SELECT 1 FROM public.delphi_result_batches WHERE env=p_env AND batch_id=p_attempt AND sealed_sha=digest) THEN RAISE EXCEPTION 'result seal mismatch'; END IF; + RETURN jsonb_build_object('schema','delphi-result-batch/1','batch_id',p_attempt,'sha256',digest,'families',names); +END $$; +CREATE FUNCTION public.pd_result_put_family(p_env text,p_job uuid,p_owner uuid,p_attempt uuid,p_epoch bigint,p_family text,p_wire text) +RETURNS jsonb LANGUAGE plpgsql SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ +DECLARE j public.polis_queue_jobs; +BEGIN + j=public.pd_lock(p_env,p_job); + IF NOT public.pq_owns(j,p_owner,p_attempt,p_epoch) THEN RAISE EXCEPTION 'result writer fenced'; END IF; + RETURN public.pd_result_insert_family(p_env,p_job,p_attempt,p_family,p_wire); +END $$; +CREATE FUNCTION public.pd_result_seal(p_env text,p_job uuid,p_owner uuid,p_attempt uuid,p_epoch bigint) +RETURNS jsonb LANGUAGE plpgsql SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ +DECLARE j public.polis_queue_jobs; +BEGIN + j=public.pd_lock(p_env,p_job); + IF NOT public.pq_owns(j,p_owner,p_attempt,p_epoch) THEN RAISE EXCEPTION 'result writer fenced'; END IF; + RETURN public.pd_result_seal_internal(p_env,p_attempt); +END $$; +CREATE FUNCTION public.pd_result_bind_artifact() RETURNS trigger LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp AS $$ +DECLARE doc jsonb;ref jsonb;pair record;b public.delphi_result_batches;expected jsonb; +BEGIN + doc=NEW.payload::jsonb; + IF doc ? 'family_files' THEN + IF doc ? 'results' OR jsonb_typeof(doc->'family_files')<>'object' OR doc->'family_files'='{}'::jsonb THEN RAISE EXCEPTION 'invalid inline result families'; END IF; + FOR pair IN SELECT key,value FROM jsonb_each(doc->'family_files') LOOP + IF jsonb_typeof(pair.value)<>'string' THEN RAISE EXCEPTION 'invalid inline codec bytes'; END IF; + PERFORM public.pd_result_insert_family(NEW.env,NEW.job_id,NEW.attempt_id,pair.key,pair.value#>>'{}'); + END LOOP; + ref=public.pd_result_seal_internal(NEW.env,NEW.attempt_id); + ELSIF doc ? 'results' THEN ref=doc->'results'; + ELSE RETURN NEW; END IF; -- Reference graph artifacts remain compatible. + SELECT * INTO STRICT b FROM public.delphi_result_batches WHERE env=NEW.env AND batch_id=NEW.attempt_id FOR UPDATE; + expected=public.pd_result_seal_internal(NEW.env,NEW.attempt_id); + IF ref IS DISTINCT FROM expected OR b.job_id<>NEW.job_id OR b.run_id<>NEW.run_id OR b.artifact_id IS NOT NULL THEN RAISE EXCEPTION 'artifact result binding'; END IF; + UPDATE public.delphi_result_batches SET artifact_id=NEW.artifact_id WHERE env=NEW.env AND batch_id=NEW.attempt_id; + RETURN NEW; +END $$; +-- AFTER INSERT permits the batch's FK to the newly fenced artifact. +CREATE TRIGGER result_artifact_bind AFTER INSERT ON public.delphi_artifacts FOR EACH ROW EXECUTE FUNCTION public.pd_result_bind_artifact(); +CREATE FUNCTION public.pd_result_bundle_rows(p_env text,p_root uuid) +RETURNS TABLE(family text,item_key jsonb,item jsonb,ordinal integer) LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + WITH RECURSIVE upstream(id,depth,path) AS ( + SELECT p_root,0,ARRAY[p_root] UNION ALL + SELECT e.artifact_id,u.depth+1,u.path||e.artifact_id FROM upstream u + JOIN public.delphi_artifacts a ON a.env=p_env AND a.artifact_id=u.id + JOIN public.delphi_graph_edges e ON e.env=a.env AND e.consumer=a.job_id + WHERE e.artifact_id IS NOT NULL AND NOT e.artifact_id=ANY(u.path) + ), chosen AS ( + SELECT DISTINCT ON(f.family) f.family,b.batch_id FROM upstream u + JOIN public.delphi_result_batches b ON b.env=p_env AND b.artifact_id=u.id + JOIN public.delphi_result_families f ON f.env=b.env AND f.batch_id=b.batch_id + ORDER BY f.family,u.depth,u.id + ) SELECT r.family,r.item_key,r.item,r.ordinal FROM chosen c JOIN public.delphi_result_rows r + ON r.env=p_env AND r.batch_id=c.batch_id AND r.family=c.family +$$; +CREATE FUNCTION public.pd_result_artifact_wire(p_env text,p_artifact uuid,p_family text) RETURNS jsonb +LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + SELECT jsonb_build_object('wire',f.codec_wire,'sha256',f.content_sha,'batch_sha256',b.sealed_sha,'batch_id',b.batch_id) + FROM public.delphi_result_batches b JOIN public.delphi_result_families f USING(env,batch_id) + WHERE b.env=p_env AND b.artifact_id=p_artifact AND f.family=p_family +$$; +CREATE FUNCTION public.pd_result_artifact_family(p_env text,p_artifact uuid,p_family text) RETURNS jsonb +LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + SELECT COALESCE(jsonb_agg(item ORDER BY ordinal),'[]'::jsonb) FROM public.pd_result_bundle_rows(p_env,p_artifact) WHERE family=p_family +$$; +CREATE VIEW public.delphi_result_current_rows WITH(security_barrier=true) AS + SELECT s.env,s.zid,s.scope_key,s.generation,r.family,r.item_key,r.item + FROM public.delphi_graph_served s CROSS JOIN LATERAL public.pd_result_bundle_rows(s.env,s.artifact_id) r; +CREATE FUNCTION public.pd_result_served(p_env text,p_zid integer,p_scope text,p_family text) RETURNS jsonb +LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + SELECT public.pd_result_artifact_family(p_env,s.artifact_id,p_family) FROM public.delphi_graph_served s + WHERE s.env=p_env AND s.zid=p_zid AND s.scope_key=p_scope +$$; +CREATE FUNCTION public.pd_result_served_bundle(p_env text,p_zid integer,p_scope text) RETURNS jsonb +LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + SELECT jsonb_build_object('generation',s.generation,'families',COALESCE((SELECT jsonb_object_agg(family,items) FROM + (SELECT family,jsonb_agg(item ORDER BY ordinal) items FROM public.pd_result_bundle_rows(p_env,s.artifact_id) GROUP BY family) f),'{}'::jsonb)) + FROM public.delphi_graph_served s WHERE s.env=p_env AND s.zid=p_zid AND s.scope_key=p_scope +$$; +-- Archive wire stays text: PostgreSQL JSONB cannot represent NUL in a legacy +-- attribute. Decode the canonical inner file in the language codec, not SQL. +CREATE VIEW public.delphi_result_legacy_controls WITH(security_barrier=true) AS + SELECT s.env,s.zid,s.scope_key,s.generation,c.key AS family,c.value AS codec_wire + FROM public.delphi_graph_served s JOIN public.delphi_artifacts a USING(env,artifact_id) + CROSS JOIN LATERAL jsonb_each_text(COALESCE(a.payload::jsonb->'legacy_control_files','{}'::jsonb)) c + WHERE c.key IN ('Delphi_JobQueue','Delphi_JobActiveGuard'); +CREATE VIEW public.delphi_result_publications WITH(security_barrier=true) AS + SELECT s.env,s.zid,s.scope_key,s.generation,s.artifact_id,a.job_id::text + FROM public.delphi_graph_served s JOIN public.delphi_artifacts a USING(env,artifact_id); +REVOKE ALL ON public.delphi_result_legacy_controls,public.delphi_result_publications FROM PUBLIC,polis_queue_executor; +GRANT SELECT ON public.delphi_result_legacy_controls,public.delphi_result_publications TO polis_queue_executor; +CREATE VIEW public.delphi_result_jobs WITH(security_barrier=true) AS + SELECT env,job_id::text,zid::text AS conversation_id,report_id, + CASE status WHEN 'succeeded' THEN 'COMPLETED' WHEN 'dead' THEN 'FAILED' WHEN 'running' THEN 'PROCESSING' ELSE upper(status) END AS status, + kind AS job_type,config_effective AS job_config, + to_char(created_at AT TIME ZONE 'UTC','YYYY-MM-DD"T"HH24:MI:SS.US"Z"') AS created_at, + to_char(completed_at AT TIME ZONE 'UTC','YYYY-MM-DD"T"HH24:MI:SS.US"Z"') AS completed_at,error, + (SELECT g.scope_key FROM public.delphi_graph_nodes n JOIN public.delphi_graphs g USING(env,graph_id) + WHERE n.env=delphi_jobs.env AND n.job_id=delphi_jobs.job_id) AS scope_key + FROM public.delphi_jobs; +REVOKE ALL ON public.delphi_result_jobs FROM PUBLIC,polis_queue_executor; +GRANT SELECT ON public.delphi_result_jobs TO polis_queue_executor; +REVOKE ALL ON public.delphi_result_catalog,public.delphi_result_batches,public.delphi_result_families,public.delphi_result_rows,public.delphi_result_current_rows FROM PUBLIC,polis_queue_executor; +DO $$ DECLARE f record; BEGIN + FOR f IN SELECT oid::regprocedure name FROM pg_proc WHERE pronamespace='public'::regnamespace AND starts_with(proname,'pd_result_') LOOP + EXECUTE format('REVOKE ALL ON FUNCTION %s FROM PUBLIC,polis_queue_executor',f.name); END LOOP; +END $$; +GRANT SELECT ON public.delphi_result_current_rows TO polis_queue_executor; +GRANT EXECUTE ON FUNCTION public.pd_result_put_family(text,uuid,uuid,uuid,bigint,text,text), + public.pd_result_seal(text,uuid,uuid,uuid,bigint),public.pd_result_artifact_family(text,uuid,text), + public.pd_result_bundle_rows(text,uuid),public.pd_result_artifact_wire(text,uuid,text), + public.pd_result_served(text,integer,text,text),public.pd_result_served_bundle(text,integer,text) TO polis_queue_executor; +COMMIT; diff --git a/server/postgres/migrations/000029_extend_delphi_graph_stages.sql b/server/postgres/migrations/000029_extend_delphi_graph_stages.sql new file mode 100644 index 0000000000..ee474570b2 --- /dev/null +++ b/server/postgres/migrations/000029_extend_delphi_graph_stages.sql @@ -0,0 +1,96 @@ +-- #1427: actual bounded Delphi numerical graph on #1432 core M27. +-- Superseding dead branches, scoped breakers and provider remediation remain deferred. +BEGIN; +SET LOCAL lock_timeout='5s'; +SET LOCAL ROLE polis_queue_owner; +SET LOCAL search_path=pg_catalog,pg_temp; +ALTER TABLE public.polis_queue_jobs DROP CONSTRAINT polis_queue_jobs_stage_check; +ALTER TABLE public.polis_queue_jobs ADD CHECK(stage IN ('noop','delphi_full_pipeline','delphi_narrative','math_rebuild','graph_embed','graph_cluster','graph_topics','graph_narrative')); +ALTER TABLE public.delphi_jobs DROP CONSTRAINT delphi_jobs_kind_check; +ALTER TABLE public.delphi_jobs ADD CHECK(kind IN ('full_pipeline','embed','snapshot','umap','cluster','keywords','topics','topic_name','narrative','collective_statement','visualize','math_rebuild','legacy_import','legacy_queue_record')); +CREATE OR REPLACE FUNCTION public.pd_graph_admit(p_env text,p_zid integer,p_scope text,p_key text,p_spec jsonb,p_supersedes uuid DEFAULT NULL) +RETURNS jsonb LANGUAGE plpgsql SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ +#variable_conflict use_column +DECLARE g public.delphi_graphs; gid uuid=gen_random_uuid(); root uuid; + n jsonb; e jsonb; jid uuid; rid uuid; prod uuid; aid uuid; decl jsonb; reply jsonb; + v_stage text; cls text; +BEGIN + IF p_supersedes IS NOT NULL THEN RAISE EXCEPTION 'superseding dead branches is not supported'; END IF; + IF p_env IS NULL OR p_env !~ '^[a-z0-9_-]{1,64}$' OR p_scope IS NULL OR length(p_scope) NOT BETWEEN 1 AND 128 + OR p_key IS NULL OR length(p_key) NOT BETWEEN 1 AND 128 OR p_spec->>'schema' IS DISTINCT FROM 'polis-job-graph/1' + OR jsonb_typeof(p_spec->'nodes') IS DISTINCT FROM 'array' OR jsonb_array_length(p_spec->'nodes') NOT BETWEEN 1 AND 32 + OR p_spec-'schema'-'nodes'<>'{}'::jsonb OR octet_length(p_spec::text)>1048576 THEN RAISE EXCEPTION 'invalid graph'; END IF; + PERFORM 1 FROM public.conversations WHERE zid=p_zid FOR KEY SHARE; + IF NOT FOUND THEN RAISE EXCEPTION 'unknown conversation'; END IF; + PERFORM pg_advisory_xact_lock(hashtextextended(jsonb_build_array('pd:scope',p_env,p_scope)::text,0)); + SELECT * INTO g FROM public.delphi_graphs WHERE env=p_env AND scope_key=p_scope AND request_key=p_key; + IF FOUND THEN + IF g.zid<>p_zid OR g.request IS DISTINCT FROM p_spec OR g.supersedes IS DISTINCT FROM p_supersedes THEN RAISE EXCEPTION 'graph request conflict'; END IF; + RETURN jsonb_build_object('outcome','existing','graph_id',g.graph_id,'root_job_id',g.root_job_id); + END IF; + IF EXISTS(SELECT 1 FROM public.delphi_job_guards WHERE env=p_env AND scope_key=p_scope) THEN RAISE EXCEPTION 'scope busy'; END IF; + root=gen_random_uuid(); + INSERT INTO public.delphi_graphs(env,graph_id,zid,scope_key,request_key,request,root_job_id,supersedes) + VALUES(p_env,gid,p_zid,p_scope,p_key,p_spec,root,p_supersedes); + FOR n IN SELECT value FROM jsonb_array_elements(p_spec->'nodes') LOOP + v_stage=n->>'stage'; cls=n->>'class'; decl=n->'declared'; + IF n-'key'-'stage'-'class'-'declared'-'inputs'-'max_attempts'<>'{}'::jsonb + OR n->>'key' IS NULL OR v_stage IS NULL OR v_stage NOT IN ('graph_embed','graph_cluster','graph_topics','graph_narrative') + OR cls IS NULL OR NOT ((v_stage='graph_cluster' AND cls IN ('delphi','large')) OR (v_stage<>'graph_cluster' AND cls='delphi')) + OR jsonb_typeof(decl) IS DISTINCT FROM 'object' + OR decl-'snapshot'-'code'-'model'-'runtime'-'seed'-'config'-'mode'-'memory_bytes'-'work_units'<>'{}'::jsonb + OR decl->>'mode' IS DISTINCT FROM 'full' + OR decl->>'code' IS NULL OR decl->>'code' !~ '^[0-9a-f]{40,64}$' + OR NOT COALESCE((CASE v_stage WHEN 'graph_embed' THEN decl->>'model' IN ('local-token-count/1','sentence-transformers/all-MiniLM-L6-v2') WHEN 'graph_cluster' THEN decl->>'model' IN ('local-nearest-centroid/1','delphi-umap-evoc/1') WHEN 'graph_topics' THEN decl->>'model' = 'delphi-tfidf-keywords/1' WHEN 'graph_narrative' THEN decl->>'model' IN ('local-cluster-summary/1','local-narrative-fixture/1','legacy-dynamo-export/1') ELSE false END),false) OR COALESCE(decl->>'runtime','')='' + OR jsonb_typeof(decl->'seed') IS DISTINCT FROM 'number' OR jsonb_typeof(decl->'config') IS DISTINCT FROM 'object' + OR jsonb_typeof(decl->'snapshot') IS DISTINCT FROM 'object' + OR jsonb_typeof(decl->'snapshot'->'data'->'texts') IS DISTINCT FROM 'array' + OR jsonb_array_length(decl->'snapshot'->'data'->'texts') NOT BETWEEN 1 AND 2000 + OR (decl->'snapshot'->'data')-'texts'<>'{}'::jsonb + OR public.pd_graph_hash(decl->'snapshot'->'data') IS DISTINCT FROM decl->'snapshot'->>'sha256' + OR (decl->'snapshot')-'data'-'sha256'<>'{}'::jsonb + OR COALESCE((decl->>'memory_bytes')::bigint,0) NOT BETWEEN 1 AND (CASE cls WHEN 'delphi' THEN 4294967296 ELSE 2147483648 END) + OR COALESCE((decl->>'work_units')::integer,0) NOT BETWEEN 1 AND 10000 + OR jsonb_typeof(n->'inputs') IS DISTINCT FROM 'array' OR jsonb_array_length(n->'inputs')>4 + OR COALESCE((n->>'max_attempts')::integer,0) NOT BETWEEN 1 AND 10 + THEN RAISE EXCEPTION 'invalid stage contract (incremental not supported)'; END IF; + IF (v_stage='graph_embed' OR decl->>'model'='legacy-dynamo-export/1') AND jsonb_array_length(n->'inputs')<>0 OR (v_stage<>'graph_embed' AND decl->>'model'<>'legacy-dynamo-export/1') AND jsonb_array_length(n->'inputs')<>1 + THEN RAISE EXCEPTION 'stage input arity'; END IF; + jid=CASE WHEN NOT EXISTS(SELECT 1 FROM public.delphi_graph_nodes WHERE env=p_env AND graph_id=gid) THEN root ELSE gen_random_uuid() END; + rid=gen_random_uuid(); + reply=public.pq_enqueue(p_env,p_zid,'graph:'||p_scope||':'||(n->>'key'),'graph-admission',p_key, + public.pd_graph_hash(n),rid,jid,'graph://'||gid::text||'/'||(n->>'key'),public.pd_graph_hash(decl),public.pd_graph_hash(decl->'config'), + decl->>'code',1::smallint,(n->>'max_attempts')::integer); + IF reply->>'outcome'<>'enqueued' THEN RAISE EXCEPTION 'graph product conflict'; END IF; + UPDATE public.polis_queue_runs SET contract_version='polis-queue/5' WHERE env=p_env AND run_id=rid; + INSERT INTO public.delphi_jobs(job_id,env,zid,kind,parent_job_id,run_id,origin,replayable,reuse_eligible,status,config_effective,code_version,model_versions) + VALUES(jid,p_env,p_zid,substring(v_stage from 7),CASE WHEN jid<>root THEN root END,rid,'queued',true,true,'queued',decl->'config',decl->>'code',jsonb_build_object('exact',decl->>'model')); + INSERT INTO public.delphi_graph_nodes VALUES(p_env,p_zid,gid,jid,rid,n->>'key',decl,NULL,NULL); + UPDATE public.polis_queue_jobs SET stage=n->>'stage',worker_class=cls WHERE env=p_env AND job_id=jid; + END LOOP; + FOR n IN SELECT value FROM jsonb_array_elements(p_spec->'nodes') LOOP + SELECT job_id INTO STRICT jid FROM public.delphi_graph_nodes WHERE env=p_env AND graph_id=gid AND node_key=n->>'key'; + FOR e IN SELECT value FROM jsonb_array_elements(n->'inputs') LOOP + aid=NULL; + IF e-'node'-'artifact_id'-'sha256'-'contract_sha256'-'role'<>'{}'::jsonb + OR e->>'role' IS DISTINCT FROM (CASE n->>'stage' WHEN 'graph_cluster' THEN 'embeddings' WHEN 'graph_topics' THEN 'clusters' WHEN 'graph_narrative' THEN CASE WHEN n->'declared'->>'model'='local-cluster-summary/1' THEN 'clusters' ELSE 'topics' END END) + OR (e ? 'node')=(e ? 'artifact_id') THEN RAISE EXCEPTION 'invalid input role or reference'; END IF; + IF e ? 'node' THEN + SELECT job_id INTO STRICT prod FROM public.delphi_graph_nodes WHERE env=p_env AND graph_id=gid AND node_key=e->>'node'; + ELSE + aid=(e->>'artifact_id')::uuid; + SELECT a.job_id INTO STRICT prod FROM public.delphi_artifacts a JOIN public.delphi_graph_nodes pn USING(env,job_id) + WHERE a.env=p_env AND a.artifact_id=aid AND pn.zid=p_zid AND a.content_sha=e->>'sha256' AND public.pd_graph_hash(pn.declared)=e->>'contract_sha256'; + END IF; + IF NOT EXISTS(SELECT 1 FROM public.polis_queue_jobs qp WHERE qp.env=p_env AND qp.job_id=prod AND qp.stage=CASE n->>'stage' WHEN 'graph_cluster' THEN 'graph_embed' WHEN 'graph_topics' THEN 'graph_cluster' WHEN 'graph_narrative' THEN CASE WHEN n->'declared'->>'model'='local-cluster-summary/1' THEN 'graph_cluster' ELSE 'graph_topics' END END) + THEN RAISE EXCEPTION 'input stage mismatch'; END IF; + IF (SELECT gn.declared->'snapshot'->>'sha256' FROM public.delphi_graph_nodes gn WHERE gn.env=p_env AND gn.job_id=prod) IS DISTINCT FROM n->'declared'->'snapshot'->>'sha256' THEN RAISE EXCEPTION 'upstream snapshot mismatch'; END IF; + INSERT INTO public.delphi_graph_edges VALUES(p_env,jid,prod,e->>'role',e->>'sha256',aid); + END LOOP; + END LOOP; + UPDATE public.delphi_graphs SET sealed=true WHERE env=p_env AND graph_id=gid; + INSERT INTO public.delphi_job_guards VALUES(p_env,p_scope,p_zid,root,public.pd_graph_hash(p_spec)); + RETURN jsonb_build_object('outcome','enqueued','graph_id',gid,'root_job_id',root); +END $$; + +COMMIT; diff --git a/server/postgres/migrations/000030_create_delphi_writers.sql b/server/postgres/migrations/000030_create_delphi_writers.sql new file mode 100644 index 0000000000..796e7acb3c --- /dev/null +++ b/server/postgres/migrations/000030_create_delphi_writers.sql @@ -0,0 +1,196 @@ +-- #1448: durable writers behind DELPHI_RESULT_BACKEND=postgres. +-- Prior artifacts and migration bytes remain immutable. No vote writes. +BEGIN; +SET LOCAL lock_timeout='5s'; +SET LOCAL ROLE polis_queue_owner; +SET LOCAL search_path=pg_catalog,pg_temp; +CREATE TABLE public.delphi_writer_runs( + env text NOT NULL,job_id uuid NOT NULL,zid integer NOT NULL,scope_key text NOT NULL, + base_generation bigint NOT NULL,base_artifact uuid,actor text NOT NULL, + PRIMARY KEY(env,job_id), FOREIGN KEY(env,job_id) REFERENCES public.delphi_graph_nodes, + FOREIGN KEY(env,base_artifact) REFERENCES public.delphi_artifacts +); +CREATE TRIGGER writer_run_immutable BEFORE UPDATE OR DELETE ON public.delphi_writer_runs + FOR EACH ROW EXECUTE FUNCTION public.pd_graph_immutable(); +-- Called only by admission under the publication lock. Bind the existing /2 +-- provider lifecycle to /5 immutable result storage without changing its stage. +CREATE FUNCTION public.pd_writer_bind(p_env text,p_job uuid,p_scope text,p_actor text) RETURNS void +LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp AS $$ +DECLARE j public.delphi_jobs; s public.delphi_graph_served; gid uuid=gen_random_uuid(); +BEGIN + SELECT * INTO STRICT j FROM public.delphi_jobs WHERE env=p_env AND job_id=p_job; + SELECT * INTO s FROM public.delphi_graph_served WHERE env=p_env AND zid=j.zid AND scope_key=p_scope; + INSERT INTO public.delphi_graphs(env,graph_id,zid,scope_key,request_key,request,root_job_id) + VALUES(p_env,gid,j.zid,p_scope,p_job::text,jsonb_build_object('schema','delphi-writer/1','actor',p_actor),p_job); + INSERT INTO public.delphi_graph_nodes(env,zid,graph_id,job_id,run_id,node_key,declared) + VALUES(p_env,j.zid,gid,p_job,j.run_id,'writer',jsonb_build_object('schema','delphi-writer/1','config',j.config_effective)); + IF s.artifact_id IS NOT NULL THEN + INSERT INTO public.delphi_graph_edges(env,consumer,producer,input_role,expected_sha,artifact_id) + SELECT p_env,p_job,a.job_id,'previous',a.content_sha,a.artifact_id FROM public.delphi_artifacts a + WHERE a.env=p_env AND a.artifact_id=s.artifact_id; + END IF; + UPDATE public.delphi_graphs SET sealed=true WHERE env=p_env AND graph_id=gid; + INSERT INTO public.delphi_writer_runs VALUES(p_env,p_job,j.zid,p_scope,COALESCE(s.generation,0),s.artifact_id,p_actor); +END $$; +CREATE FUNCTION public.pd_writer_admit(p_env text,p_zid integer,p_scope text,p_actor text,p_key text,p_sha text, + p_job uuid,p_run uuid,p_stage text,p_report text,p_config jsonb,p_code text) RETURNS jsonb +LANGUAGE plpgsql SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ +DECLARE frame text; conf jsonb; reply jsonb;guard text; +BEGIN + IF p_scope IS NULL OR length(p_scope) NOT BETWEEN 1 AND 128 OR p_stage NOT IN ('delphi_full_pipeline','delphi_narrative') + OR p_code !~ '^[0-9a-f]{40,64}$' OR p_actor IS NULL OR length(p_actor) NOT BETWEEN 1 AND 128 + OR p_key IS NULL OR length(p_key) NOT BETWEEN 1 AND 128 OR p_sha !~ '^[0-9a-f]{64}$' + OR jsonb_typeof(p_config) IS DISTINCT FROM 'object' THEN RAISE EXCEPTION 'invalid writer admission'; END IF; + IF p_report IS NOT NULL AND NOT EXISTS(SELECT 1 FROM public.reports WHERE zid=p_zid AND report_id=p_report) + THEN RAISE EXCEPTION 'report conversation mismatch'; END IF; + -- One publication scope per conversation, including HTTP edits. Serialize + -- admission with edits; an active producer retains its snapshot until completion. + PERFORM pg_advisory_xact_lock(hashtextextended(jsonb_build_array('pd:writer',p_env,p_zid,p_scope)::text,0)); + guard='writer:'||public.pd_graph_hash(jsonb_build_array(p_zid,p_scope)); + PERFORM public.pd_release_scope(p_env,guard); + conf=p_config||jsonb_build_object('result_backend','postgres','result_scope',p_scope); + frame=jsonb_build_object('schema','polis-jobs.admission/1','zid',p_zid,'report_id',p_report,'config',conf,'inputs','{}'::jsonb)::text; + reply=public.pd_enqueue(p_env,p_zid,guard,p_actor,p_key,p_sha,p_run,p_job, + 'frame://inline/'||rtrim(translate(replace(encode(convert_to(frame,'UTF8'),'base64'),chr(10),''),'+/','-_'),'='), + encode(sha256(convert_to(frame,'UTF8')),'hex'),public.pd_graph_hash(conf),p_code,1::smallint,3, + p_stage,p_report,guard,conf); + IF reply->>'outcome'='enqueued' THEN PERFORM public.pd_writer_bind(p_env,p_job,p_scope,p_actor); END IF; + RETURN reply; +END $$; +-- Daemon reads the immutable base; no queue credential reaches Python. +CREATE FUNCTION public.pd_writer_base(p_env text,p_job uuid) RETURNS jsonb +LANGUAGE sql STABLE SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ + SELECT jsonb_build_object('schema','delphi-writer-base/1','zid',w.zid,'scope',w.scope_key, + 'generation',w.base_generation,'artifact_id',w.base_artifact, + 'families',COALESCE((SELECT jsonb_object_agg(family,rows) FROM ( + SELECT family,jsonb_agg(item ORDER BY ordinal) rows FROM public.pd_result_bundle_rows(w.env,w.base_artifact) GROUP BY family) f),'{}'::jsonb)) + FROM public.delphi_writer_runs w WHERE env=p_env AND job_id=p_job +$$; +CREATE FUNCTION public.pd_writer_publish(p_env text,p_job uuid,p_attempt uuid,p_results jsonb) RETURNS void +LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp AS $$ +DECLARE w public.delphi_writer_runs; aid uuid;gen bigint;wire text;j public.polis_queue_jobs; +BEGIN + SELECT * INTO STRICT w FROM public.delphi_writer_runs WHERE env=p_env AND job_id=p_job; + SELECT * INTO STRICT j FROM public.polis_queue_jobs WHERE env=p_env AND job_id=p_job; + PERFORM pg_advisory_xact_lock(hashtextextended(jsonb_build_array('pd:writer',p_env,w.zid,w.scope_key)::text,0)); + SELECT generation INTO gen FROM public.delphi_graph_served WHERE env=p_env AND zid=w.zid AND scope_key=w.scope_key FOR UPDATE; + IF COALESCE(gen,0)<>w.base_generation THEN RAISE EXCEPTION 'writer publication conflict'; END IF; + IF p_results IS DISTINCT FROM public.pd_result_seal_internal(p_env,p_attempt) THEN RAISE EXCEPTION 'writer result seal mismatch'; END IF; + wire=jsonb_build_object('schema','delphi-writer-result/1','results',p_results,'legacy_control_files',COALESCE((SELECT payload::jsonb->'legacy_control_files' FROM public.delphi_artifacts WHERE env=p_env AND artifact_id=w.base_artifact),'{}'::jsonb))::text; + INSERT INTO public.delphi_artifacts(env,job_id,run_id,attempt_id,output_role,schema_version,payload,content_sha,byte_count) + VALUES(p_env,p_job,j.run_id,p_attempt,'result','delphi-writer-result/1',wire,encode(sha256(convert_to(wire,'UTF8')),'hex'),octet_length(wire)) RETURNING artifact_id INTO aid; + INSERT INTO public.delphi_graph_served VALUES(p_env,w.zid,w.scope_key,w.base_generation+1,aid) + ON CONFLICT(env,zid,scope_key) DO UPDATE SET generation=EXCLUDED.generation,artifact_id=EXCLUDED.artifact_id; +END $$; +CREATE OR REPLACE FUNCTION public.pd_finalize(p_env text,p_job uuid,p_owner uuid,p_attempt uuid,p_epoch bigint,p_uri text,p_sha text) +RETURNS jsonb LANGUAGE plpgsql SET search_path=pg_catalog,pg_temp SET TimeZone='UTC' AS $$ +DECLARE j public.polis_queue_jobs; a public.polis_queue_attempts; r public.polis_queue_runs; raw text; m jsonb; +BEGIN + j=public.pd_lock(p_env,p_job); + IF j.stage LIKE 'graph_%' THEN RAISE EXCEPTION 'graph manifest required'; END IF; + SELECT * INTO a FROM public.polis_queue_attempts WHERE env=p_env AND job_id=p_job AND attempt_id=p_attempt; + SELECT * INTO r FROM public.polis_queue_runs WHERE env=p_env AND run_id=j.run_id; + IF j.state='succeeded' AND j.terminal_attempt_id=p_attempt AND a.owner_id=p_owner AND a.lease_epoch=p_epoch THEN + RETURN public.pq_result(CASE WHEN a.output_sha256=p_sha AND r.expected_output_uri=p_uri THEN 'already_succeeded' ELSE 'invalid_output' END,j); + END IF; + IF NOT public.pq_owns(j,p_owner,p_attempt,p_epoch) THEN RETURN public.pq_result('fenced',j); END IF; + IF a.process_exit_confirmed_at IS NULL THEN RAISE EXCEPTION 'process exit proof required'; END IF; + IF p_uri IS NULL OR p_uri !~ '^file://.+' OR length(p_uri)>2048 OR p_sha IS NULL OR p_sha !~ '^[0-9a-f]{64}$' + THEN RETURN public.pq_result('invalid_output',j); END IF; + SELECT line INTO raw FROM public.polis_queue_logs WHERE env=p_env AND attempt_id=p_attempt AND stream='manifest' + AND encode(sha256(convert_to(line,'UTF8')),'hex')=p_sha ORDER BY seq DESC LIMIT 1; + IF raw IS NULL THEN RETURN public.pq_result('invalid_output',j); END IF; + BEGIN m=raw::jsonb; EXCEPTION WHEN invalid_text_representation THEN RETURN public.pq_result('invalid_output',j); END; + IF (m->>'schema' IS DISTINCT FROM 'polis-jobs.output-manifest/1' AND m->>'schema' IS DISTINCT FROM 'polis-jobs.output-manifest/2') + OR m->>'job_id' IS DISTINCT FROM p_job::text OR m->>'attempt_id' IS DISTINCT FROM p_attempt::text + OR m->>'stage' IS DISTINCT FROM j.stage OR m->>'outcome' IS DISTINCT FROM 'succeeded' + OR COALESCE(m->>'phase','') NOT IN ('submit','recheck','run') + OR jsonb_typeof(m->'outputs') IS DISTINCT FROM 'array' + OR jsonb_typeof(m->'inputs') IS DISTINCT FROM 'object' + OR jsonb_typeof(m->'models') IS DISTINCT FROM 'object' + OR jsonb_typeof(m->'cost') IS DISTINCT FROM 'object' + OR (m ? 'artifacts' AND m->'artifacts'<>'[]'::jsonb) + THEN RETURN public.pq_result('invalid_output',j); END IF; + IF EXISTS(SELECT 1 FROM jsonb_array_elements(m->'outputs') o WHERE o->>'store' IS DISTINCT FROM CASE WHEN m->>'schema'='polis-jobs.output-manifest/2' THEN 'postgres' ELSE 'dynamodb' END + OR COALESCE(o->>'table','')='' OR NOT (o ? 'keys' OR o ? 'key_prefix') OR COALESCE(o->>'rows','') !~ '^[0-9]+$') + THEN RETURN public.pq_result('invalid_output',j); END IF; + IF EXISTS(SELECT 1 FROM public.delphi_provider_requests WHERE env=p_env AND job_id=p_job AND state IN ('intent','submission_unknown','submitted')) + THEN RAISE EXCEPTION 'provider request unresolved'; END IF; + IF EXISTS(SELECT 1 FROM public.delphi_writer_runs WHERE env=p_env AND job_id=p_job) THEN + IF m->>'schema' IS DISTINCT FROM 'polis-jobs.output-manifest/2' OR NOT (m ? 'results') OR m ? 'family_spool' + THEN RETURN public.pq_result('invalid_output',j); END IF; + PERFORM public.pd_writer_publish(p_env,p_job,p_attempt,m->'results'); + ELSIF m->>'schema'<>'polis-jobs.output-manifest/1' THEN RETURN public.pq_result('invalid_output',j); + END IF; + UPDATE public.polis_queue_attempts SET outcome='succeeded',ended_at=clock_timestamp(),output_sha256=p_sha WHERE env=p_env AND attempt_id=p_attempt; + UPDATE public.polis_queue_jobs SET state='succeeded',terminal_attempt_id=p_attempt,owner_id=NULL,attempt_id=NULL,locked_until=NULL, + output_sha256=p_sha,version=version+1,mgmt_version=mgmt_version+1,updated_at=clock_timestamp() WHERE env=p_env AND job_id=p_job RETURNING * INTO j; + UPDATE public.delphi_jobs SET output_manifest_digest=decode(p_sha,'hex') WHERE env=p_env AND job_id=p_job; + UPDATE public.polis_queue_runs SET state='succeeded',expected_output_uri=p_uri,expected_output_sha256=p_sha,output_sha256=p_sha WHERE env=p_env AND run_id=j.run_id; + RETURN public.pq_result('succeeded',j,false); +END $$; + +-- Canonical JSON for codec bytes (JSONB's display order is not canonical). +CREATE FUNCTION public.pd_writer_json(v jsonb) RETURNS text LANGUAGE plpgsql IMMUTABLE + SET search_path=pg_catalog,pg_temp AS $$ +BEGIN + CASE jsonb_typeof(v) + WHEN 'object' THEN RETURN '{'||COALESCE((SELECT string_agg(to_jsonb(key)::text||':'||public.pd_writer_json(value),',' ORDER BY key COLLATE "C") FROM jsonb_each(v)),'')||'}'; + WHEN 'array' THEN RETURN '['||COALESCE((SELECT string_agg(public.pd_writer_json(value),',' ORDER BY ord) FROM jsonb_array_elements(v) WITH ORDINALITY a(value,ord)),'')||']'; + ELSE RETURN v::text; END CASE; +END $$; +-- Synchronous API edits use a completed producer/attempt in the same transaction. +-- Only three audited HTTP families are writable here. The canonical key derives +-- the conversation; caller-supplied zids cannot redirect a report's results. +CREATE FUNCTION public.pd_result_mutate(p_env text,p_scope text,p_actor text,p_request uuid,p_family text,p_operation text,p_item jsonb) +RETURNS jsonb LANGUAGE plpgsql SECURITY DEFINER SET search_path=pg_catalog,pg_temp AS $$ +#variable_conflict use_column +DECLARE v_zid integer; jid uuid=gen_random_uuid();rid uuid=gen_random_uuid();aid uuid=gen_random_uuid();owner uuid=gen_random_uuid(); + reply jsonb; k jsonb; spec jsonb; wire text;ref jsonb;guard text;old_item jsonb;line jsonb; +BEGIN + IF p_family NOT IN ('Delphi_CollectiveStatement','Delphi_NarrativeReports','report_narrative_store') OR p_operation NOT IN ('put','delete') + OR jsonb_typeof(p_item) IS DISTINCT FROM 'object' OR octet_length(p_item::text)>1048576 + THEN RAISE EXCEPTION 'invalid synchronous result mutation'; END IF; + IF p_family='Delphi_CollectiveStatement' THEN v_zid=split_part(p_item->'zid_topic_jobid'->>'S','#',1)::integer; + ELSE SELECT r.zid INTO STRICT v_zid FROM public.reports r WHERE report_id=split_part(p_item->'rid_section_model'->>'S','#',1); END IF; + SELECT key_spec INTO STRICT spec FROM public.delphi_result_catalog WHERE family=p_family; + SELECT jsonb_agg(p_item->(part->>0)) INTO k FROM jsonb_array_elements(spec) part; + IF EXISTS(SELECT 1 FROM jsonb_array_elements(k) v WHERE v='null'::jsonb) THEN RAISE EXCEPTION 'missing result key'; END IF; + reply=public.pd_writer_admit(p_env,v_zid,p_scope,p_actor,p_request::text, + public.pd_graph_hash(jsonb_build_array(p_family,p_operation,p_item)),jid,rid,'delphi_full_pipeline',NULL, + jsonb_build_object('synchronous',true),encode(sha256(convert_to(pg_get_functiondef('public.pd_result_mutate(text,text,text,uuid,text,text,jsonb)'::regprocedure),'UTF8')),'hex')); + IF reply->>'outcome'<>'enqueued' THEN + IF reply->>'outcome'='existing' AND reply->>'state'='succeeded' THEN RETURN reply; END IF; + RAISE EXCEPTION 'result scope busy or request conflict'; + END IF; + -- No child or provider is started. The in-transaction producer owns only jid. + UPDATE public.polis_queue_jobs SET state='running',owner_id=owner,attempt_id=aid,lease_epoch=lease_epoch+1, + attempt_count=attempt_count+1,locked_until=clock_timestamp()+interval '60 seconds',version=version+1,mgmt_version=mgmt_version+1 + WHERE env=p_env AND job_id=jid; + INSERT INTO public.polis_queue_attempts(env,attempt_id,job_id,owner_id,lease_epoch,outcome,process_exit_confirmed_at) + VALUES(p_env,aid,jid,owner,1,'running',clock_timestamp()); + SELECT item INTO old_item FROM public.delphi_result_current_rows WHERE env=p_env AND scope_key=p_scope AND family=p_family AND item_key=k AND delphi_result_current_rows.zid=v_zid; + wire=public.pd_writer_json(jsonb_build_object('codec','delphi-storage-codec/1','family',p_family,'key',(SELECT jsonb_agg(part->0) FROM jsonb_array_elements(spec) part)))||chr(10); + FOR line IN SELECT doc FROM ( + SELECT item AS doc,item_key AS sort_key FROM public.delphi_result_current_rows + WHERE env=p_env AND scope_key=p_scope AND family=p_family AND item_key<>k AND delphi_result_current_rows.zid=v_zid + UNION ALL SELECT p_item,k WHERE p_operation='put' + ) items ORDER BY public.pd_writer_json(sort_key) COLLATE "C" LOOP + wire=wire||public.pd_writer_json(line)||chr(10); + END LOOP; + PERFORM public.pd_result_put_family(p_env,jid,owner,aid,1,p_family,wire); + ref=public.pd_result_seal(p_env,jid,owner,aid,1); + PERFORM public.pd_writer_publish(p_env,jid,aid,ref); + UPDATE public.polis_queue_attempts SET outcome='succeeded',ended_at=clock_timestamp(),output_sha256=ref->>'sha256' WHERE env=p_env AND attempt_id=aid; + UPDATE public.polis_queue_jobs SET state='succeeded',terminal_attempt_id=aid,owner_id=NULL,attempt_id=NULL,locked_until=NULL, + output_sha256=ref->>'sha256',version=version+1,mgmt_version=mgmt_version+1 WHERE env=p_env AND job_id=jid; + UPDATE public.polis_queue_runs SET state='succeeded',output_sha256=ref->>'sha256' WHERE env=p_env AND run_id=rid; + guard='writer:'||public.pd_graph_hash(jsonb_build_array(v_zid,p_scope)); + PERFORM public.pd_release_scope(p_env,guard); + RETURN jsonb_build_object('outcome','succeeded','job_id',jid,'previous',old_item); +END $$; +REVOKE ALL ON public.delphi_writer_runs FROM PUBLIC,polis_queue_executor; +REVOKE ALL ON FUNCTION public.pd_writer_bind(text,uuid,text,text),public.pd_writer_publish(text,uuid,uuid,jsonb) FROM PUBLIC,polis_queue_executor; +REVOKE ALL ON FUNCTION public.pd_writer_admit(text,integer,text,text,text,text,uuid,uuid,text,text,jsonb,text),public.pd_writer_base(text,uuid),public.pd_result_mutate(text,text,text,uuid,text,text,jsonb) FROM PUBLIC; +GRANT EXECUTE ON FUNCTION public.pd_writer_admit(text,integer,text,text,text,text,uuid,uuid,text,text,jsonb,text),public.pd_writer_base(text,uuid),public.pd_result_mutate(text,text,text,uuid,text,text,jsonb) TO polis_queue_executor; +COMMIT; diff --git a/server/postgres/migrations/release.txt b/server/postgres/migrations/release.txt index 8b46a0c527..cb4593eea4 100644 --- a/server/postgres/migrations/release.txt +++ b/server/postgres/migrations/release.txt @@ -24,3 +24,6 @@ 000023_create_delphi_foundation.sql 000024_create_polis_queue_large_class.sql 000027_create_sealed_job_graphs.sql +000028_create_delphi_results.sql +000029_extend_delphi_graph_stages.sql +000030_create_delphi_writers.sql diff --git a/server/postgres/migrations/report/first-deploy.sql b/server/postgres/migrations/report/first-deploy.sql index 3398618f8d..0ac018aecf 100644 --- a/server/postgres/migrations/report/first-deploy.sql +++ b/server/postgres/migrations/report/first-deploy.sql @@ -2,7 +2,7 @@ -- First deployment only: no existing ledger or queue. Other states refuse. -- This is a catalog forecast, not DDL success or deployment-health proof. -- sha256 8fa1066d9fc29c1df528b300199c619a2f9edecf1432926ad6a8146ebc650593 adoption/helpers.sql --- sha256 491da3bcc32d863925f93a0cfcd5a5f777987089f696b796efc087f94819dcfc release.txt +-- sha256 29b2e187b4ed87eafd44e1e5634a021b1a4ddb0ca6c05bc86b7f7db693645fb0 release.txt -- sha256 8a8b24fa47a25c613a8329aa79f413ac7461a93b12134045ae1cf2457e2ee3b7 held.txt -- sha256 c6b5c71247d129964dd89ba560a4522caef1abbb9379ad1b3445d6ed96be805b adoption/000000_initial.sql -- sha256 28ef33976beb9c7c9248ac800f29aab048f4410c2170ad12e3ade71d72e5ecb5 adoption/000001_update_pwreset_table.sql @@ -49,6 +49,9 @@ -- sha256 97437ea57d90c51cc664385ebb6e8ec3d8684ac0f8e7b0c60e84df1a010531f7 000023_create_delphi_foundation.sql -- sha256 68261afb81f286bed45fdff6ab52e052392d1dd32abdb1e78579f72644116697 000024_create_polis_queue_large_class.sql -- sha256 fbbf948e4316010dd96344ae91542006c38299e52e0a0eb281371c039ab37775 000027_create_sealed_job_graphs.sql +-- sha256 0f90d8c7c6f9a440b2d9901dcf7a00f17222d01e48f38fb1c8c7d22be05156c8 000028_create_delphi_results.sql +-- sha256 b4dd41790639184c40418effd290ecd86d15cb8d9f46a1a77b616f37a63d5906 000029_extend_delphi_graph_stages.sql +-- sha256 14581a3a77c6e5cf73305a45f6635f8869a26a5674ea423b7d4fd2a146b653ce 000030_create_delphi_writers.sql BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY; SET LOCAL search_path=pg_catalog,public; SET LOCAL statement_timeout='30s'; @@ -1392,7 +1395,10 @@ SELECT (SELECT EXISTS(SELECT 1 FROM pg_index WHERE indexrelid=to_regclass('publi ELSE 'WOULD_APPLY' END FROM (VALUES ('000019_create_polis_queue.sql'),('000023_create_delphi_foundation.sql'), ('000024_create_polis_queue_large_class.sql'), - ('000027_create_sealed_job_graphs.sql')) p(migration) CROSS JOIN readiness + ('000027_create_sealed_job_graphs.sql'), + ('000028_create_delphi_results.sql'), + ('000029_extend_delphi_graph_stages.sql'), + ('000030_create_delphi_writers.sql')) p(migration) CROSS JOIN readiness UNION ALL SELECT migration, NULL::boolean, 'OUTSIDE_RELEASE' FROM (VALUES ('000020'),('000021'),('000025'),('000026')) p(migration) diff --git a/server/src/config.ts b/server/src/config.ts index 30167382fc..86da66fe92 100644 --- a/server/src/config.ts +++ b/server/src/config.ts @@ -30,6 +30,12 @@ import("source-map-support").then((sourceMapSupport) => { }); export default { + // Delphi readers remain on DynamoDB unless explicitly selected. Keep these + // reads at the shared configuration boundary, including test-time selection. + get delphiResultBackend(): string | undefined { return process.env.DELPHI_RESULT_BACKEND; }, + get delphiResultEnv(): string | undefined { return process.env.DELPHI_RESULT_ENV; }, + get delphiResultScope(): string | undefined { return process.env.DELPHI_RESULT_SCOPE; }, + get delphiWriterCodeSha(): string | undefined { return process.env.DELPHI_WRITER_CODE_SHA; }, domainOverride, isDevMode: devMode, reachableErrorHandler, diff --git a/server/src/db/pg-query.ts b/server/src/db/pg-query.ts index d982ef48a5..7252ae7916 100644 --- a/server/src/db/pg-query.ts +++ b/server/src/db/pg-query.ts @@ -257,6 +257,19 @@ function connectReadOnly() { return readPool.connect(); } +// Result requests hold a repeatable-read snapshot while other route work may +// borrow the primary pool. A separate bounded pool prevents pool starvation. +let resultSnapshotPool: Pool | undefined; +function connectResultSnapshot() { + if (!resultSnapshotPool) { + resultSnapshotPool = new Pool({ + ...pgConnection, max: 2, connectionTimeoutMillis: 30000, + } as unknown as PoolConfig); + resultSnapshotPool.on("error", error => logger.error("pg_result_snapshot_pool", error)); + } + return resultSnapshotPool.connect(); +} + // Session policy applied immediately after BEGIN, from // cost-reduction/04-plans/P-024-queue-substrate.md. These are declared initial // bounds for the queue substrate, not a general-purpose transaction profile; @@ -372,5 +385,6 @@ export default { stream_queryP_readOnly, connect, connectReadOnly, + connectResultSnapshot, withTransaction, }; diff --git a/server/src/nextComment.ts b/server/src/nextComment.ts index 50b1b5df4f..bbdc169bf8 100644 --- a/server/src/nextComment.ts +++ b/server/src/nextComment.ts @@ -1,5 +1,6 @@ import _ from "underscore"; import LruCache from "lru-cache"; +import { resultClient } from "./utils/delphiResults"; import { DynamoDBClient, DynamoDBClientConfig } from "@aws-sdk/client-dynamodb"; import { DynamoDBDocumentClient, QueryCommand } from "@aws-sdk/lib-dynamodb"; @@ -38,13 +39,12 @@ if (Config.dynamoDbEndpoint) { }; } -const dynamoClient = new DynamoDBClient(dynamoDBConfig); -const dynamoDocClient = DynamoDBDocumentClient.from(dynamoClient, { +const dynamoDocClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); const DELPHI_TOPIC_NAMES_TABLE = "Delphi_CommentClustersLLMTopicNames"; // This very much follows the outline of the random selection above, but factors out the probabilistic logic diff --git a/server/src/ops/delphiTopicNames.ts b/server/src/ops/delphiTopicNames.ts index 2d1f8840a2..2e6ce3b9d4 100644 --- a/server/src/ops/delphiTopicNames.ts +++ b/server/src/ops/delphiTopicNames.ts @@ -12,6 +12,7 @@ // instance role. The instance role already has dynamodb:Query on Delphi_*. import { DynamoDBDocumentClient, QueryCommand } from "@aws-sdk/lib-dynamodb"; +import { resultClient } from "../utils/delphiResults"; import { makeDynamoClient } from "../utils/dynamoClient"; export const TOPIC_NAMES_TABLE = "Delphi_CommentClustersLLMTopicNames"; @@ -92,7 +93,7 @@ export function makeDelphiTopicNameReader(): TopicNameReader { const out: TopicNames = new Map(); if (zids.length === 0) return out; try { - doc = doc || DynamoDBDocumentClient.from(makeDynamoClient()); + doc = doc || resultClient(() => DynamoDBDocumentClient.from(makeDynamoClient())); } catch { for (const zid of zids) out.set(zid, null); return out; diff --git a/server/src/routes/api/v3/feeds.ts b/server/src/routes/api/v3/feeds.ts index 2915e291bc..81d72e9dd9 100644 --- a/server/src/routes/api/v3/feeds.ts +++ b/server/src/routes/api/v3/feeds.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../../../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../../../utils/logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -26,13 +27,12 @@ if (Config.dynamoDbEndpoint) { logger.info(`Using default AWS credential provider chain`); } } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Handler for feeds directory listing - shows available feeds for a report diff --git a/server/src/routes/collectiveStatement.ts b/server/src/routes/collectiveStatement.ts index d88f7e9056..5314584178 100644 --- a/server/src/routes/collectiveStatement.ts +++ b/server/src/routes/collectiveStatement.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../utils/logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -32,13 +33,12 @@ if (Config.dynamoDbEndpoint) { }; } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); const anthropic = Config.anthropicApiKey ? new Anthropic({ diff --git a/server/src/routes/delphi.ts b/server/src/routes/delphi.ts index bc439779c8..89a53443e5 100644 --- a/server/src/routes/delphi.ts +++ b/server/src/routes/delphi.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../utils/logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -28,13 +29,12 @@ if (Config.dynamoDbEndpoint) { }; } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Handler for Delphi API route that retrieves LLM topic names from DynamoDB diff --git a/server/src/routes/delphi/jobGuard.ts b/server/src/routes/delphi/jobGuard.ts index 632676d129..6b19d29fdd 100644 --- a/server/src/routes/delphi/jobGuard.ts +++ b/server/src/routes/delphi/jobGuard.ts @@ -1,3 +1,4 @@ +import Config from "../../config"; /** * Server-side active-work deduplication for Delphi job submission (P-003 S3). * @@ -46,6 +47,8 @@ import { createHash, randomUUID } from "crypto"; import { DynamoDB, DynamoDBClientConfig } from "@aws-sdk/client-dynamodb"; import { DynamoDBDocument } from "@aws-sdk/lib-dynamodb"; import logger from "../../utils/logger"; +import pg from "../../db/pg-query"; +import { postgresResults } from "../../utils/delphiResults"; import { AwsCredentialsConfigurationError, buildDynamoClientConfig, @@ -968,6 +971,9 @@ export async function assessConversationLiveness( /** Rows as the first sweep read them, for callers needing more than liveness. */ rowsByJobId: Map; }> { + if (postgresResults()) { + return assessPostgresConversationLiveness(conversationId); + } const liveByJobId = new Map(); let rowsByJobId = new Map(); @@ -1047,6 +1053,58 @@ export async function assessConversationLiveness( return { complete: true, liveByJobId, rowsByJobId }; } +/** One primary-database snapshot covers graph peers, descendants, provider work + * and process-exit receipts. Archived metadata never becomes executable work. */ +async function assessPostgresConversationLiveness(conversationId: string) { + const liveByJobId = new Map(); + const rowsByJobId = new Map(); + const env = Config.delphiResultEnv; + if (!env) throw new Error("DELPHI_RESULT_ENV is required for Postgres results"); + const scope = Config.delphiResultScope || null; + try { + const rows = await pg.queryP(` + WITH RECURSIVE selected AS ( + SELECT j.*, n.graph_id FROM public.delphi_jobs j + LEFT JOIN public.delphi_graph_nodes n USING (env,job_id) + LEFT JOIN public.delphi_graphs g ON g.env=n.env AND g.graph_id=n.graph_id + WHERE j.env=$1 AND j.zid::text=$2 AND ($3::text IS NULL OR g.scope_key=$3) + ), related(root_id,job_id) AS ( + SELECT s.job_id,s.job_id FROM selected s + UNION + SELECT r.root_id,c.job_id FROM related r + JOIN public.delphi_jobs c ON c.env=$1 AND c.parent_job_id=r.job_id + ), work AS ( + SELECT root_id,job_id FROM related + UNION + SELECT s.job_id,n.job_id FROM selected s + JOIN public.delphi_graph_nodes n ON n.env=s.env AND n.graph_id=s.graph_id + ) + SELECT v.*, EXISTS ( + SELECT 1 FROM work w + JOIN public.delphi_jobs j ON j.env=$1 AND j.job_id=w.job_id + LEFT JOIN public.polis_queue_jobs q ON q.env=j.env AND q.job_id=j.job_id + WHERE w.root_id=s.job_id AND ( + COALESCE(q.state,j.status) NOT IN ('succeeded','dead','cancelled') + OR EXISTS (SELECT 1 FROM public.polis_queue_attempts a + WHERE a.env=j.env AND a.job_id=j.job_id AND a.process_exit_confirmed_at IS NULL) + OR EXISTS (SELECT 1 FROM public.delphi_provider_requests p + WHERE p.env=j.env AND p.job_id=j.job_id + AND p.state IN ('intent','submission_unknown','submitted')) + ) + ) AS work_live + FROM selected s JOIN public.delphi_result_jobs v ON v.env=s.env AND v.job_id=s.job_id::text + ORDER BY v.job_id`, [env, conversationId, scope]) as any[]; + for (const row of rows) { + liveByJobId.set(row.job_id, row.work_live !== false); + rowsByJobId.set(row.job_id, row); + } + return { complete: true, liveByJobId, rowsByJobId }; + } catch (error: any) { + logger.warn(`Postgres Delphi liveness unavailable: ${error?.message || error}`); + return { complete: false, liveByJobId, rowsByJobId }; + } +} + /** * A cheap fingerprint of everything about a row that would change the answer. * Two reads that produce the same anchor were not separated by a write. @@ -1291,6 +1349,9 @@ export async function admitDelphiJob( request: AdmissionRequest, store: JobAdmissionStore = dynamoJobAdmissionStore ): Promise { + if (postgresResults() && store === dynamoJobAdmissionStore) { + return admitPostgresWriter(request); + } const { scope } = request; const scopeKey = scopeGuardKey(scope); const configHash = configFingerprint(scope.jobConfig); @@ -1608,3 +1669,43 @@ export async function admitDelphiJob( `Delphi job admission did not settle for scope ${logScope(scopeKey)}` ); } + +/** Queue and guard admission remain one SQL transaction under the same switch. */ +async function admitPostgresWriter(request: AdmissionRequest): Promise { + const {scope, jobItem} = request; + const env = Config.delphiResultEnv; + const resultScope = Config.delphiResultScope; + const code = Config.delphiWriterCodeSha; + if (!env || !resultScope || !code || !/^[0-9a-f]{40,64}$/.test(code)) { + throw new JobAdmissionUnavailableError("Postgres writer env, scope and code pin are required"); + } + const stages: Record = {FULL_PIPELINE:"delphi_full_pipeline",CREATE_NARRATIVE_BATCH:"delphi_narrative"}; + const stage = stages[scope.jobType]; + if (!stage) throw new JobAdmissionUnavailableError("Checker attempts belong to their existing queued narrative job"); + const zid = Number(scope.conversationId); + if (!Number.isSafeInteger(zid) || zid < 1) throw new JobAdmissionUnavailableError("Invalid conversation"); + const rawConfig = typeof scope.jobConfig === "string" ? JSON.parse(scope.jobConfig) : scope.jobConfig || {}; + const nested = rawConfig.stages?.[0]?.config || {}; + const config = {...rawConfig,...nested,include_moderation:true, + batch_size:nested.max_batch_size ?? rawConfig.max_batch_size ?? 20}; + // Keep the existing moderation pin from job_poller.pipeline_moderation_flags. + const payload = canonicalise({jobType:scope.jobType,zid,report:scope.reportId || null,config:rawConfig}); + const sha = createHash("sha256").update(JSON.stringify(payload)).digest("hex"); + const key = request.idempotencyKey || String(jobItem.job_id); + try { + const rows = await pg.queryP<{value:any}>(`SELECT public.pd_writer_admit( + $1::text,$2::integer,$3::text,$4::text,$5::text,$6::text,$7::uuid,$8::uuid, + $9::text,$10::text,$11::jsonb,$12::text) AS value`, + [env,zid,resultScope,"server-delphi",key,sha,randomUUID(),randomUUID(),stage, + scope.reportId || null,JSON.stringify(config),code]); + const value = rows[0]?.value; + if (value?.outcome === "conflict") return {outcome:"idempotency_conflict",jobId:value.job_id}; + if (!["enqueued","existing"].includes(value?.outcome)) throw new Error("Admission unavailable"); + const status: Record = {succeeded:"COMPLETED",dead:"FAILED",cancelled:"CANCELLED",running:"PROCESSING",queued:"PENDING",parked:"AWAITING_RECHECK"}; + if (value.outcome === "enqueued") return {outcome:"created",jobId:value.job_id,jobStatus:"PENDING",workLive:true}; + return {outcome:"deduplicated",jobId:value.job_id,jobStatus:status[value.state] || value.state, + workLive:!["succeeded","dead","cancelled"].includes(value.state)}; + } catch (error) { + throw new JobAdmissionUnavailableError("Postgres writer admission failed; no Dynamo fallback"); + } +} diff --git a/server/src/routes/delphi/reports.ts b/server/src/routes/delphi/reports.ts index f3fd0d95b4..119d4d7c20 100644 --- a/server/src/routes/delphi/reports.ts +++ b/server/src/routes/delphi/reports.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../../utils/logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -26,13 +27,12 @@ if (Config.dynamoDbEndpoint) { logger.info(`Using default AWS credential provider chain`); } } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Handler for Delphi API route that retrieves LLM-generated reports from DynamoDB diff --git a/server/src/routes/delphi/topicAgenda.ts b/server/src/routes/delphi/topicAgenda.ts index aa91b51cef..69b45e9878 100644 --- a/server/src/routes/delphi/topicAgenda.ts +++ b/server/src/routes/delphi/topicAgenda.ts @@ -1,3 +1,5 @@ +import { resultClient, postgresResults } from "../../utils/delphiResults"; +import { resultQuery } from "../../utils/delphiResultSnapshot"; import _ from "underscore"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; import { @@ -30,13 +32,12 @@ if (Config.dynamoDbEndpoint) { }; } -const dynamoClient = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(dynamoClient, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); // Bound the work one participant save can trigger. Without a cap the query is // bounded only by the conversation's partition size; with COALESCE in place, @@ -47,8 +48,17 @@ const JOB_QUERY_PAGE_SIZE = 25; /** * Get the newest completed Delphi job ID for a conversation. */ -async function getCurrentDelphiJobId(zid: string): Promise { +export async function getCurrentDelphiJobId(zid: string): Promise { try { + if (postgresResults()) { + const env=Config.delphiResultEnv; + if (!env) throw new Error("DELPHI_RESULT_ENV is required for Postgres results"); + const scope=Config.delphiResultScope; + const rows=await resultQuery<{job_id:string}>(`SELECT job_id FROM public.delphi_result_publications + WHERE env=$1 AND zid=$2 ${scope ? "AND scope_key=$3" : ""}`,scope ? [env,Number(zid),scope] : [env,Number(zid)]); + if (rows.length>1) throw new Error("Ambiguous published job; configure DELPHI_RESULT_SCOPE"); + return rows[0]?.job_id || null; + } // Query the ConversationIndex GSI to find completed jobs for this conversation const queryParams: QueryCommandInput = { TableName: "Delphi_JobQueue", @@ -83,7 +93,7 @@ async function getCurrentDelphiJobId(zid: string): Promise { return null; } catch (error: any) { - logger.error("Error getting current Delphi job ID from DynamoDB", error); + logger.error("Error getting current published Delphi job ID", error); // Degrade instead of failing the request: a null here cannot erase an // existing attribution (the writes below COALESCE it), while throwing // would 500 the handler and drop the participant's selections entirely. diff --git a/server/src/routes/delphi/topicMod.ts b/server/src/routes/delphi/topicMod.ts index faa7cf0276..755e5b1784 100644 --- a/server/src/routes/delphi/topicMod.ts +++ b/server/src/routes/delphi/topicMod.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../../utils/logger"; import { DynamoDBClient, DynamoDBClientConfig } from "@aws-sdk/client-dynamodb"; @@ -32,13 +33,12 @@ if (Config.dynamoDbEndpoint) { } } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Two of the tables this file reads have never existed in any environment. diff --git a/server/src/routes/delphi/topics.ts b/server/src/routes/delphi/topics.ts index 690b6608b9..2b5c9a9820 100644 --- a/server/src/routes/delphi/topics.ts +++ b/server/src/routes/delphi/topics.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../../utils/logger"; import { DynamoDBClient, ListTablesCommand } from "@aws-sdk/client-dynamodb"; @@ -50,13 +51,13 @@ logger.info(`DynamoDB Config: `); // Create DynamoDB clients -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const client = resultClient(() => new DynamoDBClient(dynamoDBConfig)); +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Handler for Delphi API route that retrieves LLM topic names from DynamoDB diff --git a/server/src/routes/delphi/visualizations.ts b/server/src/routes/delphi/visualizations.ts index 512cc65cbf..b5d1d7c6c8 100644 --- a/server/src/routes/delphi/visualizations.ts +++ b/server/src/routes/delphi/visualizations.ts @@ -1,3 +1,4 @@ +import { resultClient, postgresResults } from "../../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../../utils/logger"; import { getZidFromReport } from "../../utils/parameter"; @@ -27,13 +28,12 @@ if (Config.dynamoDbEndpoint) { }; } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); /** * Handler for Delphi API route that retrieves visualization information @@ -82,6 +82,14 @@ export async function handle_GET_delphi_visualizations( `Fetching visualizations for report_id: ${report_id}, conversation_id: ${conversation_id}` ); + // Queue graph results have no S3 static visualization artifact contract. + // Return actual job metadata without issuing a cloud listing. + if (postgresResults()) { + const metadata=await fetchJobMetadata(conversation_id); + return res.json({status:"success",report_id,visualizations:[], + jobs:Object.values(metadata).filter((job:any)=>!jobId || job.jobId===jobId)}); + } + // Configure S3 client const s3Config: any = { region: Config.AWS_REGION || "us-east-1", @@ -372,7 +380,8 @@ function processJobItems( // Additive: lets a reloaded client tell "finished" from "terminal row, // work still outstanding underneath" without a second request. A false // here only ever comes from the authoritative sweep. - workLive: livenessComplete ? liveByJobId.get(job_id) !== false : true, + ...(item.archived === true ? {archived:true} : {}), + workLive: item.archived === true ? false : livenessComplete ? liveByJobId.get(job_id) !== false : true, // Set when this server withdrew the job after losing a race: it names the // job that actually carries the work. The client follows it rather than // dropping the id it was acknowledged with. diff --git a/server/src/routes/topicStats.ts b/server/src/routes/topicStats.ts index cb06743276..23ce0e4a78 100644 --- a/server/src/routes/topicStats.ts +++ b/server/src/routes/topicStats.ts @@ -1,3 +1,4 @@ +import { resultClient } from "../utils/delphiResults"; import { Request, Response } from "express"; import logger from "../utils/logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -23,13 +24,12 @@ if (Config.dynamoDbEndpoint) { }; } -const client = new DynamoDBClient(dynamoDBConfig); -const docClient = DynamoDBDocumentClient.from(client, { +const docClient = resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, -}); +})); interface TopicMetrics { comment_count: number; diff --git a/server/src/utils/commentClusters.ts b/server/src/utils/commentClusters.ts index 0ead7b4337..da68b47d4c 100644 --- a/server/src/utils/commentClusters.ts +++ b/server/src/utils/commentClusters.ts @@ -1,3 +1,4 @@ +import { resultClient, postgresResults } from "./delphiResults"; import pg from "../db/pg-query"; import logger from "./logger"; import { DynamoDBClient } from "@aws-sdk/client-dynamodb"; @@ -89,13 +90,12 @@ function createDynamoDBClient(): DynamoDBDocumentClient { }; } - const client = new DynamoDBClient(dynamoDBConfig); - return DynamoDBDocumentClient.from(client, { + return resultClient(() => DynamoDBDocumentClient.from(new DynamoDBClient(dynamoDBConfig), { marshallOptions: { convertEmptyValues: true, removeUndefinedValues: true, }, - }); + })); } /** @@ -115,6 +115,8 @@ export async function getClusterAssignments( zid: number, useCache = false ): Promise> { + // A request snapshot must not reuse assignments from an older publication. + if (postgresResults()) useCache = false; // Check cache if enabled if (useCache) { const cached = clusterAssignmentsCache.get(zid); diff --git a/server/src/utils/delphiResultSnapshot.ts b/server/src/utils/delphiResultSnapshot.ts new file mode 100644 index 0000000000..a831b319f9 --- /dev/null +++ b/server/src/utils/delphiResultSnapshot.ts @@ -0,0 +1,56 @@ +import Config from "../config"; +/** One publication snapshot for every PostgreSQL result read in an HTTP request. */ +import { AsyncLocalStorage } from "async_hooks"; +import { RequestHandler } from "express"; +import { PoolClient } from "pg"; +import pg from "../db/pg-query"; +import logger from "./logger"; + +type Snapshot = { client?: Promise; closed: boolean; failure?: Error; onError?: (error: Error) => void }; +const snapshots = new AsyncLocalStorage(); + +export const delphiResultSnapshot: RequestHandler = (_req, res, next) => { + if (Config.delphiResultBackend !== "postgres") return next(); + const snapshot: Snapshot = { closed: false }; + const close = () => { + if (snapshot.closed) return; + snapshot.closed = true; + if (!snapshot.client) return; + void snapshot.client.then(async client => { + let failure: Error | undefined = snapshot.failure; + try { await client.query("ROLLBACK"); } + catch (error) { failure = error as Error; logger.error("delphi_result_snapshot_close", error); } + finally { + client.release(failure); + if (!failure && snapshot.onError) client.removeListener("error", snapshot.onError); + } + }, () => { /* Failed setup already released its client. */ }); + }; + res.once("finish", close); + res.once("close", close); + snapshots.run(snapshot, next); +}; + +/** Primary connection: replicas may lag a newly published result pointer. */ +export async function resultQuery(sql: string, params?: any[]): Promise { + const snapshot = snapshots.getStore(); + if (!snapshot) return pg.queryP(sql, params) as Promise; + if (snapshot.closed) throw new Error("Delphi result request has closed"); + if (snapshot.failure) throw snapshot.failure; + if (!snapshot.client) snapshot.client = pg.connectResultSnapshot().then(async client => { + snapshot.onError = error => { snapshot.failure = error; }; + client.on("error", snapshot.onError); + try { + await client.query("BEGIN ISOLATION LEVEL REPEATABLE READ READ ONLY"); + await client.query("SET LOCAL statement_timeout = '30s'; SET LOCAL idle_in_transaction_session_timeout = '30s'"); + return client; + } catch (error) { + client.release(error as Error); + // Discarded sockets may still emit an asynchronous error. + throw error; + } + }); + const client = await snapshot.client; + if (snapshot.failure) throw snapshot.failure; + return (await client.query(sql, params)).rows as T[]; +} diff --git a/server/src/utils/delphiResultWriter.ts b/server/src/utils/delphiResultWriter.ts new file mode 100644 index 0000000000..51eea33d0d --- /dev/null +++ b/server/src/utils/delphiResultWriter.ts @@ -0,0 +1,35 @@ +import {randomUUID} from "crypto"; +import Config from "../config"; +import pg from "../db/pg-query"; +import {canonicalNumber} from "./delphiStorageCodec"; + +function tag(value:any):any { + if (value === null) return {NULL:true}; + if (typeof value === "string") return {S:value}; + if (typeof value === "boolean") return {BOOL:value}; + if (typeof value === "number" && Number.isFinite(value)) { + if (Number.isInteger(value) && !Number.isSafeInteger(value)) throw new Error("Unsafe result integer"); + return {N:canonicalNumber(String(value))}; + } + if (Buffer.isBuffer(value)) return {B:value.toString("base64")}; + if (Array.isArray(value)) return {L:value.map(tag)}; + if (value && typeof value === "object" && !(value instanceof Set)) return {M:tagItem(value)}; + throw new Error("Unsupported result value"); +} +function tagItem(value:any):any { + return Object.fromEntries(Object.entries(value).filter(([,v]) => v !== undefined).map(([k,v])=>[k,tag(v)])); +} +/** Use a primary autocommit connection: request read snapshots are READ ONLY. */ +export async function mutatePostgresResult(command:any):Promise { + const operation = command.constructor.name; + const input = command.input; + if (!Config.delphiResultEnv || !Config.delphiResultScope) throw new Error("Postgres writer env and scope required"); + if (input.ConditionExpression || input.UpdateExpression) throw new Error("Unsupported synchronous result condition"); + const item = operation === "DeleteItemCommand" ? input.Key : tagItem(input.Item || input.Key); + const rows = await pg.queryP<{value:any}>(`SELECT public.pd_result_mutate( + $1::text,$2::text,$3::text,$4::uuid,$5::text,$6::text,$7::jsonb) AS value`, + [Config.delphiResultEnv,Config.delphiResultScope,"server-result-edit",randomUUID(),input.TableName, + operation === "PutCommand" ? "put" : "delete",JSON.stringify(item)]); + if (rows[0]?.value?.outcome !== "succeeded") throw new Error("Result mutation was not committed"); + return {producer_job_id:rows[0].value.job_id}; +} diff --git a/server/src/utils/delphiResults.ts b/server/src/utils/delphiResults.ts new file mode 100644 index 0000000000..a8137820e2 --- /dev/null +++ b/server/src/utils/delphiResults.ts @@ -0,0 +1,218 @@ +import Config from "../config"; +import { mutatePostgresResult } from "./delphiResultWriter"; +/** Published Delphi results. Backend choice is explicit and never falls back. */ +import { resultQuery } from "./delphiResultSnapshot"; +import { FAMILIES, decodeFamily, canonicalNumber } from "./delphiStorageCodec"; + +export function postgresResults(): boolean { + const value = Config.delphiResultBackend || "dynamodb"; + if (!["postgres", "dynamodb"].includes(value)) throw new Error("invalid DELPHI_RESULT_BACKEND"); + return value === "postgres"; +} +export function decodeAttribute(value: any): any { + if ("S" in value) return value.S; + if ("N" in value) { + const n = Number(value.N); + if (!Number.isFinite(n) || (Number.isInteger(n) && !Number.isSafeInteger(n))) { + throw new Error("result numeric value exceeds reader precision"); + } + return n; + } + if ("BOOL" in value) return value.BOOL; + if ("NULL" in value) return null; + if ("L" in value) return value.L.map(decodeAttribute); + if ("M" in value) return decodeItem(value.M); + if ("SS" in value) return new Set(value.SS); + if ("NS" in value) return new Set(value.NS.map((n: string) => decodeAttribute({N:n}))); + if ("B" in value) return Buffer.from(value.B, "base64"); + if ("BS" in value) return new Set(value.BS.map((b: string) => Buffer.from(b,"base64"))); + throw new Error("invalid stored AttributeValue"); +} +export function decodeItem(item: any): Record { + return Object.fromEntries(Object.entries(item).map(([k,v]) => [k, decodeAttribute(v)])); +} + +/** Historical controls are display-only. Preserve exact values even when JS + * cannot represent a legacy decimal; the raw tagged item remains available. */ +export function decodeArchivedItem(tagged: any): Record { + let reason: string | undefined; + const value = (av:any):any => { + if ("N" in av) { + const n=Number(av.N); + if (!Number.isFinite(n) || (Number.isInteger(n) && !Number.isSafeInteger(n)) || canonicalNumber(String(n)) !== av.N) { + reason="legacy numeric metadata exceeds JavaScript precision; exact decimal retained as text"; + return av.N; + } + } + if ("L" in av) return av.L.map(value); + if ("M" in av) return Object.fromEntries(Object.entries(av.M).map(([k,v])=>[k,value(v)])); + if ("NS" in av) return new Set(av.NS.map((n:string)=>value({N:n}))); + return decodeAttribute(av); + }; + const item=Object.fromEntries(Object.entries(tagged).map(([k,v])=>[k,value(v)])); + return {...item,archived:true,...(reason ? {unreadable_metadata_reason:reason,legacy_control_item:tagged} : {})}; +} + +/** Only the expression grammar used by the audited Delphi readers is accepted. */ +export function predicate(expression: string | undefined, input: any): (item: any) => boolean { + if (!expression) return () => true; + const resolve = (token: string, item: any) => token.startsWith(":") + ? input.ExpressionAttributeValues?.[token] + : item[input.ExpressionAttributeNames?.[token] || token]; + const clauses = expression.split(/\s+AND\s+/i).map(part => { + const equal = part.trim().match(/^([#\w]+)\s*=\s*([:#\w]+)$/); + if (equal) return (item: any) => resolve(equal[1],item) === resolve(equal[2],item); + const prefix = part.trim().match(/^begins_with\(\s*([#\w]+)\s*,\s*(:\w+)\s*\)$/); + if (prefix) return (item: any) => String(resolve(prefix[1],item) ?? "").startsWith(String(resolve(prefix[2],item))); + throw new Error(`unsupported result expression: ${part}`); + }); + return item => clauses.every(clause => clause(item)); +} + +/** Bound parameters only: fields and values never become SQL identifiers. */ +export function sqlPredicate(expression: string | undefined, input: any, values: any[]): string { + if (!expression) return "TRUE"; + const bind = (value: any) => { values.push(value); return `$${values.length}`; }; + const field = (name: string) => `item->${bind(input.ExpressionAttributeNames?.[name] || name)}::text`; + const operand = (token: string): string => { + if (!token.startsWith(":")) return field(token); + const value = input.ExpressionAttributeValues?.[token]; + const tagged = typeof value === "string" ? {S:value} : typeof value === "number" && Number.isFinite(value) + ? {N:String(value)} : typeof value === "boolean" ? {BOOL:value} : undefined; + if (!tagged) throw new Error("unsupported result expression value"); + return `${bind(JSON.stringify(tagged))}::jsonb`; + }; + return expression.split(/\s+AND\s+/i).map(part => { + const equal = part.trim().match(/^([#\w]+)\s*=\s*([:#\w]+)$/); + if (equal) return `${field(equal[1])} = ${operand(equal[2])}`; + const prefix = part.trim().match(/^begins_with\(\s*([#\w]+)\s*,\s*(:\w+)\s*\)$/); + if (prefix) { + const value = input.ExpressionAttributeValues?.[prefix[2]]; + if (typeof value !== "string") throw new Error("begins_with requires a string"); + return `starts_with(${field(prefix[1])}->>'S',${bind(value)})`; + } + throw new Error(`unsupported result expression: ${part}`); + }).join(" AND "); +} + +export async function sendPostgresResult(command: any): Promise { + const input = command.input; + const operation = command.constructor.name; + const env = Config.delphiResultEnv; + if (!env) throw new Error("DELPHI_RESULT_ENV is required for Postgres results"); + if (operation === "ListTablesCommand") return {TableNames: Object.keys(FAMILIES)}; + const family = input.TableName; + if (!(family in FAMILIES)) { + const error = new Error(`Unknown Delphi result family: ${family}`); + error.name = "ResourceNotFoundException"; + throw error; + } + if (operation === "DescribeTableCommand") { + await resultQuery("SELECT 1 FROM public.delphi_result_current_rows LIMIT 0"); + return {Table:{TableName:family,TableStatus:"ACTIVE"}}; + } + if (["PutCommand","DeleteCommand","DeleteItemCommand"].includes(operation)) return mutatePostgresResult(command); + if (!["QueryCommand","ScanCommand","GetCommand"].includes(operation)) { + throw new Error("Published Delphi results are immutable; submit a new run"); + } + let items: any[]; + let generations: Record = {}; + if (family === "Delphi_JobQueue") { + const scope = Config.delphiResultScope; + const rows = await resultQuery(`SELECT * FROM public.delphi_result_jobs WHERE env=$1 + ${scope ? "AND scope_key=$2" : ""}`,scope ? [env,scope] : [env]) as any[]; + const archives = await resultQuery(`SELECT zid,scope_key,generation::text,codec_wire + FROM public.delphi_result_legacy_controls WHERE env=$1 AND family='Delphi_JobQueue' + ${scope ? "AND scope_key=$2" : ""} ORDER BY zid,scope_key`,scope ? [env,scope] : [env]) as any[]; + const activeIds = new Set(rows.map(row => row.job_id)); + const archived = new Map(); + for (const archive of archives) { + generations[`${archive.zid}:${archive.scope_key}`] = archive.generation; + const decoded = decodeFamily(Buffer.from(archive.codec_wire,"utf8")); + if (decoded.family !== family) throw new Error("legacy control family mismatch"); + for (const tagged of decoded.items) { + const item = decodeArchivedItem(tagged); + if (activeIds.has(item.job_id)) continue; + const prior = archived.get(item.job_id); + if (prior && JSON.stringify(prior) !== JSON.stringify(item)) throw new Error("Ambiguous archived job scopes; configure DELPHI_RESULT_SCOPE"); + archived.set(item.job_id,item); + } + } + items = [...archived.values(),...rows]; + } else { + const values: any[] = [env,family]; + let where = "env=$1 AND family=$2"; + const scope = Config.delphiResultScope; + if (scope) { values.push(scope); where += ` AND scope_key=$${values.length}`; } + where += " AND (" + sqlPredicate(input.KeyConditionExpression,input,values) + ")"; + for (const [name,value] of Object.entries(input.Key || {})) { + where += " AND (" + sqlPredicate("#key = :value", {ExpressionAttributeNames:{"#key":name},ExpressionAttributeValues:{":value":value}}, values) + ")"; + } + const rows = await resultQuery(`SELECT zid,scope_key,generation::text,item FROM public.delphi_result_current_rows + WHERE ${where} ORDER BY zid,scope_key,item_key::text`,values) as any[]; + const seen = new Map(); + items = []; + for (const row of rows) { + generations[`${row.zid}:${row.scope_key}`] = row.generation; + const item = decodeItem(row.item); + const key = JSON.stringify(FAMILIES[family].key.map(([name]) => item[name])); + if (seen.has(key)) { + if (JSON.stringify(seen.get(key)) !== JSON.stringify(row.item)) throw new Error("Ambiguous result scopes; configure DELPHI_RESULT_SCOPE"); + continue; + } + seen.set(key,row.item); items.push(item); + } + } + const cursorGenerations = input.ExclusiveStartKey?._polisPgGenerations; + if (input.ExclusiveStartKey && JSON.stringify(cursorGenerations) !== JSON.stringify(generations)) { + throw new Error("result generation changed; restart pagination"); + } + // Existing HTTP handlers perform authorization before reaching this adapter. + const matches = predicate(input.KeyConditionExpression,input); + items = items.filter(matches); + if (input.Key) items = items.filter(item => Object.entries(input.Key).every(([k,v]) => item[k] === v)); + const keys = FAMILIES[family].key.map(([k]) => k); + const key = (item: any) => Object.fromEntries(keys.map(k => [k,item[k]])); + const indexOrder: Record = {ConversationIndex:"created_at",StatusCreatedIndex:"created_at",ReportIdTimestampIndex:"timestamp","zid-created_at-index":"created_at"}; + if (input.IndexName && !indexOrder[input.IndexName]) throw new Error("unsupported result index"); + const order = input.IndexName ? indexOrder[input.IndexName] : keys[keys.length-1]; + if (input.IndexName) items = items.filter(item => item[order] !== undefined && item[order] !== null); + const compare = (a:any,b:any,fields:string[]) => { + for (const field of fields) { if(a[field]b[field])return 1; } + return 0; + }; + items.sort((a,b) => compare(a,b,[order,...keys])); + if (input.ScanIndexForward === false) items.reverse(); + if (input.ExclusiveStartKey) { + const offset = items.findIndex(item => Object.entries(input.ExclusiveStartKey).filter(([k]) => k !== "_polisPgGenerations").every(([k,v]) => item[k] === v)); + if (offset < 0) throw new Error("stale result cursor"); + items = items.slice(offset+1); + } + const limit = input.Limit ?? 1000; + if (!Number.isSafeInteger(limit) || limit < 1) throw new Error("invalid result limit"); + const page = items.slice(0,limit); + const filtered = page.filter(predicate(input.FilterExpression,input)); + if (operation === "GetCommand") { + const item=filtered[0]; + if (family === "Delphi_JobQueue" && item && item.archived !== true) { + const status=await resultQuery<{value:any}>("SELECT public.pq_job_status($1::text,$2::uuid) AS value",[env,item.job_id]); + const attempt=status[0]?.value?.attempt_id; + const entries=attempt ? await resultQuery(`SELECT + to_char(ts AT TIME ZONE 'UTC','YYYY-MM-DD"T"HH24:MI:SS.US"Z"') AS timestamp, + CASE stream WHEN 'stderr' THEN 'ERROR' ELSE 'INFO' END AS level,line AS message + FROM public.pq_attempt_logs($1::text,$2::uuid,NULL,1000) + WHERE stream IN ('stdout','stderr')`,[env,attempt]) : []; + return {Item:{...item,logs:{entries},log_attempt_id:attempt || null}}; + } + return {Item:item}; + } + return {Items:filtered, Count:filtered.length, ScannedCount:page.length, + ...(items.length > limit ? {LastEvaluatedKey:{...key(page[page.length-1]),_polisPgGenerations:generations}} : {})}; +} + +export function resultClient any}>(factory: () => T): T { + let legacy: T | undefined; + return {send: (command: any) => postgresResults() + ? sendPostgresResult(command) + : (legacy ||= factory()).send(command)} as T; +} diff --git a/server/src/utils/storage.ts b/server/src/utils/storage.ts index 297b0c81c6..9184d2134f 100644 --- a/server/src/utils/storage.ts +++ b/server/src/utils/storage.ts @@ -1,3 +1,4 @@ +import { resultClient } from "./delphiResults"; import { DeleteItemCommand, DescribeTableCommand, @@ -33,7 +34,7 @@ export default class DynamoStorageService { constructor(tableName: string, disableCache?: boolean) { // Shared credential precedence: local endpoint -> real configured keys -> // default AWS credential provider chain (the EC2 instance role in prod). - this.client = makeDynamoClient(); + this.client = resultClient(() => makeDynamoClient()); this.tableName = tableName; this.cacheDisabled = disableCache || false; } @@ -321,13 +322,20 @@ export default class DynamoStorageService { }, }; - const scanCommand = new ScanCommand(scanParams); - try { - const scanResponse = await this.client.send(scanCommand); - const items = scanResponse.Items; + const items: any[] = []; + let lastEvaluatedKey: Record | undefined; + do { + const scanResponse = await this.client.send(new ScanCommand({ + ...scanParams, + ExclusiveStartKey: lastEvaluatedKey, + })); + items.push(...(scanResponse.Items || [])); + lastEvaluatedKey = scanResponse.LastEvaluatedKey; + // Filtering can leave a page empty while later pages still match. + } while (lastEvaluatedKey); - if (!items || items.length === 0) { + if (items.length === 0) { logger.debug(`No items found with report ID prefix: ${reportIdPrefix}`); return { success: true, data: [] }; }