diff --git a/.changeset/acknowledged-desktop-document-writes.md b/.changeset/acknowledged-desktop-document-writes.md new file mode 100644 index 000000000..6826286b1 --- /dev/null +++ b/.changeset/acknowledged-desktop-document-writes.md @@ -0,0 +1,6 @@ +--- +'@xnetjs/react': patch +'@xnetjs/sqlite': patch +--- + +Serialize document saves and retain failed writes for retry when closing a desktop workspace. Keep newer edits dirty until their write finishes. Electron SQLite now syncs the WAL on commit before acknowledging writes. diff --git a/.changeset/drain-workspace-watcher.md b/.changeset/drain-workspace-watcher.md new file mode 100644 index 000000000..9da75a2a0 --- /dev/null +++ b/.changeset/drain-workspace-watcher.md @@ -0,0 +1,9 @@ +--- +'@xnetjs/plugins': minor +'@xnetjs/cli': patch +--- + +Add `AiWorkspaceWatcher.waitForIdle()` to finish active scans and asynchronous +callbacks after closing watch handles. Background failures stop the watch and +reject the drain. The agent daemon now drains before disposing storage on Ctrl-C, +so pending review writes and auto-applied edits cannot race database shutdown. diff --git a/.changeset/observe-shared-storage-writes.md b/.changeset/observe-shared-storage-writes.md new file mode 100644 index 000000000..4c6b9aaee --- /dev/null +++ b/.changeset/observe-shared-storage-writes.md @@ -0,0 +1,5 @@ +--- +'@xnetjs/data': minor +--- + +Local writes now advance from the persisted change clock, so editing a node after another local store imports it does not silently retain the earlier value. `NodeStore.refreshPersistedNodes` refreshes subscribers from already-committed shared storage without applying or broadcasting those changes again. diff --git a/.changeset/private-library-source-pages.md b/.changeset/private-library-source-pages.md new file mode 100644 index 000000000..4ce25647a --- /dev/null +++ b/.changeset/private-library-source-pages.md @@ -0,0 +1,5 @@ +--- +'@xnetjs/data': minor +--- + +Pages can link to source resources through the optional `sourceResources` relation. Personal notes and guides can cite imported material while retaining their own text and visibility. diff --git a/.changeset/safe-electron-schema-inspection.md b/.changeset/safe-electron-schema-inspection.md new file mode 100644 index 000000000..f4088123b --- /dev/null +++ b/.changeset/safe-electron-schema-inspection.md @@ -0,0 +1,5 @@ +--- +'@xnetjs/sqlite': patch +--- + +Report Electron schema inspection failures instead of treating unreadable version records as a new database. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c81ffc01a..6dbc2e4c6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -402,8 +402,8 @@ jobs: - name: Rebuild native modules for Electron (unpacked app) run: pnpm --filter xnet-desktop run deps:electron - - name: Install xvfb - run: sudo apt-get update && sudo apt-get install -y xvfb + - name: Install desktop test services + run: sudo apt-get update && sudo apt-get install -y xvfb dbus-x11 gnome-keyring libsecret-1-0 - name: Cache Playwright browser id: playwright-cache @@ -437,7 +437,8 @@ jobs: - name: Run convergence matrix + Electron app smoke (xvfb) run: | - xvfb-run --auto-servernum pnpm --filter @xnetjs/e2e-tests exec \ + bash scripts/with-linux-test-keyring.sh \ + xvfb-run --auto-servernum pnpm --filter @xnetjs/e2e-tests exec \ playwright test src/sync-matrix.spec.ts src/electron-smoke.spec.ts \ --project=chromium --project=electron --fail-on-flaky-tests env: diff --git a/.github/workflows/electron-release.yml b/.github/workflows/electron-release.yml index d4e3ec5a3..8f0e18516 100644 --- a/.github/workflows/electron-release.yml +++ b/.github/workflows/electron-release.yml @@ -372,7 +372,7 @@ jobs: - name: Smoke-test packaged app if: matrix.arch == 'x64' run: | - sudo apt-get update && sudo apt-get install -y xvfb + sudo apt-get update && sudo apt-get install -y xvfb dbus-x11 gnome-keyring libsecret-1-0 BIN=$(find "$PWD/apps/electron/dist/linux-unpacked" -maxdepth 1 -type f -executable \ ! -name 'chrome*' ! -name '*.so*' ! -name '*.bin' | head -1) if [ -z "$BIN" ]; then echo "no Electron binary found"; ls -la apps/electron/dist/linux-unpacked; exit 1; fi @@ -385,6 +385,7 @@ jobs: rm -f "$XNET_BOOT_TRACE" set +e XNET_PACKAGED_BINARY="$BIN" XNET_ELECTRON_NO_SANDBOX=1 XNET_DEBUG=1 E2E_DEBUG=1 \ + bash scripts/with-linux-test-keyring.sh \ xvfb-run --auto-servernum pnpm --filter @xnetjs/e2e-tests exec \ playwright test src/packaged-smoke.spec.ts --project=electron --fail-on-flaky-tests rc=$? diff --git a/apps/electron/README.md b/apps/electron/README.md index 9b1cfb78a..051d586ab 100644 --- a/apps/electron/README.md +++ b/apps/electron/README.md @@ -4,6 +4,12 @@ Electron desktop app for macOS, Windows, and Linux -- the primary development ta ## Development +Source launches use a `dev-` profile namespace, including the main checkout and +direct Electron launches. For example, the requested `default` profile stores +data in `xnet-desktop-dev-default`. Packaged apps keep their existing data path. +Existing folders are preserved; development no longer opens a daily workspace +implicitly. The app's runtime profile and window title show the resolved name. + ```bash pnpm dev # Start hub + app concurrently pnpm dev:both # Two instances for sync testing diff --git a/apps/electron/package.json b/apps/electron/package.json index 12d9da521..ec4963012 100644 --- a/apps/electron/package.json +++ b/apps/electron/package.json @@ -34,11 +34,11 @@ "@xnetjs/core": "workspace:*", "@xnetjs/data": "workspace:*", "@xnetjs/devkit": "workspace:*", - "@xnetjs/dictation": "workspace:*", - "@xnetjs/meetings": "workspace:*", "@xnetjs/devtools": "workspace:*", + "@xnetjs/dictation": "workspace:*", "@xnetjs/editor": "workspace:*", "@xnetjs/identity": "workspace:*", + "@xnetjs/meetings": "workspace:*", "@xnetjs/network": "workspace:*", "@xnetjs/plugins": "workspace:*", "@xnetjs/react": "workspace:*", @@ -52,9 +52,13 @@ "@xnetjs/views": "workspace:*", "@xnetjs/workbench": "workspace:*", "better-sqlite3": "^11.0.0", + "d3-force-3d": "3.0.6", "electron-updater": "^6.3.0", "lucide-react": "^0.400.0", "mermaid": "^11.4.0", + "parse5": "7.3.0", + "sharp": "^0.35.5", + "three": "0.186.1", "ws": "^8.18.0", "y-protocols": "^1.0.6", "yjs": "^13.6.24" @@ -66,6 +70,7 @@ "@types/node": "^20.0.0", "@types/react": "^18.2.0", "@types/react-dom": "^18.2.0", + "@types/three": "0.186.0", "@types/ws": "^8.5.10", "@vitejs/plugin-react": "^4.3.0", "autoprefixer": "^10.4.20", diff --git a/apps/electron/scripts/dev-launch.mjs b/apps/electron/scripts/dev-launch.mjs index 3f93365e1..e0959451f 100644 --- a/apps/electron/scripts/dev-launch.mjs +++ b/apps/electron/scripts/dev-launch.mjs @@ -28,7 +28,9 @@ import { spawn } from 'node:child_process' import { resolveDevScope, scopeEnv } from './dev-scope.mjs' -const PROBE_TIMEOUT_MS = Number(process.env.XNET_DEV_PROBE_TIMEOUT_MS || 90_000) +// Large libraries need several minutes for storage inspection before data-process +// initialization. Keep the launcher from terminating those checks mid-startup. +const PROBE_TIMEOUT_MS = Number(process.env.XNET_DEV_PROBE_TIMEOUT_MS || 900_000) const PROBE_INTERVAL_MS = 500 const argv = process.argv.slice(2) diff --git a/apps/electron/src/__tests__/ipc-node-batch.test.ts b/apps/electron/src/__tests__/ipc-node-batch.test.ts new file mode 100644 index 000000000..6d1821c63 --- /dev/null +++ b/apps/electron/src/__tests__/ipc-node-batch.test.ts @@ -0,0 +1,191 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { NodeStore, PageSchema, SQLiteNodeStorageAdapter } from '@xnetjs/data' +import { identityFromPrivateKey } from '@xnetjs/identity' +import Database from 'better-sqlite3' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { createDataService, type DataService } from '../data-process/data-service' +import { IPCNodeStorageAdapter } from '../renderer/lib/ipc-node-storage' + +const { sendEvent } = vi.hoisted(() => ({ sendEvent: vi.fn() })) +vi.mock('../data-process/events', () => ({ sendEvent })) + +let root: string +let service: DataService +let database: Database.Database +let store: NodeStore +const signingKey = new Uint8Array(32).fill(73) +const authorDID = identityFromPrivateKey(signingKey).did + +async function open(): Promise { + service = createDataService({ dbPath: join(root, 'data.db') }) + await service.initialize() + const api = { + applyNodeBatch: vi.fn(service.applyNodeBatch.bind(service)), + getNode: service.getNode.bind(service), + getLastChange: service.getLastChange.bind(service), + getChanges: service.getChanges.bind(service), + getLastLamportTime: service.getLastLamportTime.bind(service) + } + vi.stubGlobal('window', { xnetNodes: api }) + store = new NodeStore({ storage: new IPCNodeStorageAdapter(), authorDID, signingKey }) + await store.initialize() + database = new Database(join(root, 'data.db')) +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-ipc-batch-')) + await open() + sendEvent.mockClear() +}) +afterEach(async () => { + database?.close() + await service?.shutdown() + vi.unstubAllGlobals() + await rm(root, { recursive: true, force: true }) +}) + +it('commits desktop edits atomically and reads their signed history after reopening', async () => { + const node = await store.create({ + id: 'daily-note', + schemaId: PageSchema._schemaId, + properties: { title: 'Before' } + }) + await service.setDocumentContent(node.id, [0, 127, 255]) + await store.update(node.id, { properties: { title: 'After' } }) + const changes = await service.getChanges(node.id) + expect(changes).toHaveLength(2) + expect(changes[1].parentHash).toBe(changes[0].hash) + expect(changes[1].id).not.toBe(changes[1].hash) + expect(changes[1].payload.properties.title).toBe('After') + expect(await service.getChangeByHash(changes[1].hash)).toEqual(changes[1]) + expect(await service.getChangesSince(changes[0].lamport)).toEqual([changes[1]]) + expect(window.xnetNodes.applyNodeBatch).toHaveBeenCalledTimes(2) + expect(sendEvent).toHaveBeenCalledTimes(2) + expect(await service.getLastLamportTime()).toBe(changes[1].lamport) + + database.close() + await service.shutdown() + await open() + expect((await store.get(node.id))?.properties.title).toBe('After') + expect(await service.getChanges(node.id)).toEqual(changes) + expect(await service.getAllChanges()).toEqual(changes) + expect(await service.getDocumentContent(node.id)).toEqual([0, 127, 255]) +}) + +it('rolls back nodes, history, indexes, and clock when a later write fails; then retries', async () => { + await store.create({ + id: 'existing', + schemaId: PageSchema._schemaId, + properties: { title: 'Keep' } + }) + const clock = await service.getLastLamportTime() + const changes = await service.getAllChanges() + database.exec(`CREATE TRIGGER reject_batch BEFORE INSERT ON changes + WHEN NEW.node_id = 'rejected' + BEGIN SELECT RAISE(ABORT, 'Injected disk failure'); END`) + sendEvent.mockClear() + const operations = [ + { type: 'update' as const, nodeId: 'existing', options: { properties: { title: 'Replace' } } }, + { + type: 'create' as const, + options: { id: 'rejected', schemaId: PageSchema._schemaId, properties: { title: 'New' } } + } + ] + await expect(store.transaction(operations)).rejects.toThrow('Injected disk failure') + expect((await service.getNode('existing'))?.properties.title).toBe('Keep') + expect(await service.getNode('rejected')).toBeNull() + expect(await service.getAllChanges()).toEqual(changes) + expect(await service.getLastLamportTime()).toBe(clock) + expect( + database + .prepare('SELECT COUNT(*) AS count FROM node_property_scalars WHERE node_id = ?') + .get('rejected') + ).toEqual({ count: 0 }) + expect(sendEvent).not.toHaveBeenCalled() + + database.exec('DROP TRIGGER reject_batch') + const result = await store.transaction(operations) + expect((await service.getNode('existing'))?.properties.title).toBe('Replace') + expect((await service.getNode('rejected'))?.properties.title).toBe('New') + const saved = (await service.getAllChanges()).filter( + (change) => change.batchId === result.batchId + ) + expect(saved.map((change) => [change.batchIndex, change.batchSize])).toEqual([ + [0, 2], + [1, 2] + ]) + expect(saved.map((change) => change.id)).toEqual(result.changes.map((change) => change.id)) + expect(sendEvent).toHaveBeenCalledTimes(1) +}) + +it('preserves a deleted record and restores it after restart', async () => { + await store.create({ + id: 'deleted', + schemaId: PageSchema._schemaId, + properties: { title: 'Return' } + }) + await store.delete('deleted') + const deleted = await service.getNode('deleted') + expect(deleted?.deleted).toBe(true) + expect((await service.getLastChange('deleted'))?.authorDID).toBe(authorDID) + database.close() + await service.shutdown() + await open() + await store.restore('deleted') + expect((await store.get('deleted'))?.properties.title).toBe('Return') +}) + +it('rejects access after shutdown instead of acknowledging an empty workspace', async () => { + await service.shutdown() + await expect(service.getNode('missing')).rejects.toThrow('Database not initialized') + await expect(service.getAllChanges()).rejects.toThrow('Database not initialized') +}) + +it('edits an imported record without losing its original signed change identity', async () => { + await service.importDeterministicNodes({ + drafts: [ + { id: 'imported', schemaId: PageSchema._schemaId, properties: { title: 'Source title' } } + ], + authorDID, + signingKey: Array.from(signingKey) + }) + const original = await service.getLastChange('imported') + expect(original?.payload.nodeId).toBe('imported') + expect(original?.payload.properties.title).toBe('Source title') + await store.update('imported', { properties: { title: 'My title' } }) + const changes = await service.getChanges('imported') + expect(changes).toHaveLength(2) + expect(changes[0]).toEqual(original) + expect(changes[1].parentHash).toBe(original?.hash) + expect(changes[1].lamport).toBeGreaterThan(original!.lamport) + expect((await service.getNode('imported'))?.properties.title).toBe('My title') +}) + +it('hydrates a query page in one batch while preserving order and document bytes', async () => { + for (const id of ['old', 'middle', 'new']) + await store.create({ id, schemaId: PageSchema._schemaId, properties: { title: id } }) + const time = database.prepare('UPDATE nodes SET created_at=? WHERE id=?') + time.run(1, 'old') + time.run(2, 'middle') + time.run(3, 'new') + await service.setDocumentContent('middle', [0, 127, 255]) + const expected = [await service.getNode('middle'), await service.getNode('old')] + const single = vi.spyOn(service, 'getNode') + const batch = vi.spyOn(SQLiteNodeStorageAdapter.prototype, 'getNodes') + try { + const rows = await service.listNodes({ + schemaId: PageSchema._schemaId, + orderBy: { createdAt: 'desc' }, + offset: 1, + limit: 2 + }) + expect(rows).toEqual(expected) + expect(single).not.toHaveBeenCalled() + expect(batch).toHaveBeenCalledExactlyOnceWith(['middle', 'old']) + } finally { + single.mockRestore() + batch.mockRestore() + } +}) diff --git a/apps/electron/src/data-process/data-service.ts b/apps/electron/src/data-process/data-service.ts index b8c08c8d3..316cb46f8 100644 --- a/apps/electron/src/data-process/data-service.ts +++ b/apps/electron/src/data-process/data-service.ts @@ -12,7 +12,7 @@ * Events are sent to main process for relay to the appropriate window. */ -import type { DID } from '@xnetjs/core' +import type { ContentId, DID } from '@xnetjs/core' import type { ApplyNodeBatchResult, DeterministicNodeImportDraft, @@ -22,7 +22,6 @@ import type { NodeChange } from '@xnetjs/data' import type { ElectronSQLiteDiagnostics } from '@xnetjs/sqlite' -import { existsSync, unlinkSync } from 'fs' import { hashContent, createContentId } from '@xnetjs/core' import { NodeStore, SQLiteNodeStorageAdapter } from '@xnetjs/data' import { createElectronSQLiteAdapter, ElectronSQLiteAdapter } from '@xnetjs/sqlite/electron' @@ -38,6 +37,8 @@ import { } from '@xnetjs/sync' import WebSocket from 'ws' import * as Y from 'yjs' +import { deserializeNodeBatch, type SerializedNodeBatch } from '../shared/node-batch' +import { requireCompatibleDatabase } from '../storage/compatibility' import { sendEvent } from './events' // ─── Types ────────────────────────────────────────────────────────────────── @@ -183,6 +184,7 @@ export interface DataService { announceBlobs(cids: string[]): void // Node storage operations (for IPCNodeStorageAdapter) + applyNodeBatch(input: SerializedNodeBatch): Promise appendChange(change: SerializedNodeChange): Promise getChanges(nodeId: string): Promise getAllChanges(): Promise @@ -437,6 +439,11 @@ async function syncScalarRowsForNode( export function createDataService(config: DataServiceConfig): DataService { let adapter: ElectronSQLiteAdapter | null = null + let nodeStorage: SQLiteNodeStorageAdapter | null = null + function requireNodeStorage(): SQLiteNodeStorageAdapter { + if (!nodeStorage) throw new Error('Database not initialized') + return nodeStorage + } let ws: WebSocket | null = null let status: ConnectionStatus = 'disconnected' let signalingUrl = '' @@ -579,6 +586,10 @@ export function createDataService(config: DataServiceConfig): DataService { } function connect(): void { + if (process.env.XNET_RECOVERY_OFFLINE === 'true') { + log('Recovery review: sync remains offline until explicitly resumed.') + return + } if (destroyed || !signalingUrl) return if (ws) return @@ -1099,57 +1110,33 @@ export function createDataService(config: DataServiceConfig): DataService { // ─── Public API ───────────────────────────────────────────────────────── + async function flushPooledDocuments(): Promise { + if (!adapter) return + for (const [nodeId, entry] of pool) { + if (!entry.dirty) continue + const stored = await adapter.queryOne<{ state: Buffer }>( + 'SELECT state FROM yjs_state WHERE node_id = ?', + [nodeId] + ) + const merged = new Y.Doc({ gc: false }) + try { + if (stored) Y.applyUpdate(merged, stored.state) + Y.applyUpdate(merged, Y.encodeStateAsUpdate(entry.doc)) + await adapter.run( + 'INSERT OR REPLACE INTO yjs_state (node_id, state, updated_at) VALUES (?, ?, ?)', + [nodeId, Y.encodeStateAsUpdate(merged), Date.now()] + ) + } finally { + merged.destroy() + } + } + } + return { async initialize(): Promise { log('Initializing database at:', config.dbPath) - // Check if database exists and has old schema (without version tracking) - if (existsSync(config.dbPath)) { - try { - const tempAdapter = new ElectronSQLiteAdapter() - await tempAdapter.open({ path: config.dbPath }) - const version = await tempAdapter.getSchemaVersion() - await tempAdapter.close() - - if (version === 0) { - // Old database without version tracking - delete it - log('Found old database without version tracking, removing...') - try { - unlinkSync(config.dbPath) - } catch { - // File may not exist, ignore - } - try { - unlinkSync(`${config.dbPath}-wal`) - } catch { - // File may not exist, ignore - } - try { - unlinkSync(`${config.dbPath}-shm`) - } catch { - // File may not exist, ignore - } - } - } catch { - // Corrupted database - delete it - log('Found corrupted database, removing...') - try { - unlinkSync(config.dbPath) - } catch { - // File may not exist, ignore - } - try { - unlinkSync(`${config.dbPath}-wal`) - } catch { - // File may not exist, ignore - } - try { - unlinkSync(`${config.dbPath}-shm`) - } catch { - // File may not exist, ignore - } - } - } + requireCompatibleDatabase(config.dbPath) // Create adapter with unified schema. // @@ -1169,12 +1156,16 @@ export function createDataService(config: DataServiceConfig): DataService { readerPoolSize: 'auto' }) + nodeStorage = new SQLiteNodeStorageAdapter(adapter) + await nodeStorage.open() + log('Database initialized with schema version:', await adapter.getSchemaVersion()) }, async shutdown(): Promise { log('Shutting down') disconnect() + await flushPooledDocuments() // Close all renderer ports for (const [, port] of rendererPorts) { @@ -1190,6 +1181,9 @@ export function createDataService(config: DataServiceConfig): DataService { subscribedRooms.clear() tracked.clear() + await nodeStorage?.close() + nodeStorage = null + // Close adapter (handles WAL checkpoint) if (adapter) { await adapter.close() @@ -1383,6 +1377,14 @@ export function createDataService(config: DataServiceConfig): DataService { // These methods implement the NodeStorageAdapter interface for the renderer. // Data is stored in SQLite and changes are emitted for real-time sync. + async applyNodeBatch(input: SerializedNodeBatch): Promise { + const batch = deserializeNodeBatch(input) + const result = await requireNodeStorage().applyNodeBatch(batch) + // Subscribers may read immediately; publish only after the whole commit succeeds. + sendEvent('nodes:change', { changes: batch.changes.map(serializeNodeChange) }) + return result + }, + async appendChange(change: SerializedNodeChange): Promise { if (!adapter) throw new Error('Database not initialized') @@ -1434,156 +1436,35 @@ export function createDataService(config: DataServiceConfig): DataService { }, async getChanges(nodeId: string): Promise { - if (!adapter) return [] - - const rows = await adapter.query<{ - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer - }>('SELECT * FROM changes WHERE node_id = ? ORDER BY lamport_time ASC', [nodeId]) - - return rows.map(rowToSerializedChange) + return (await requireNodeStorage().getChanges(nodeId)).map(serializeNodeChange) }, async getAllChanges(): Promise { - if (!adapter) return [] - - const rows = await adapter.query<{ - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer - }>('SELECT * FROM changes ORDER BY lamport_time ASC', []) - - return rows.map(rowToSerializedChange) + return (await requireNodeStorage().getAllChanges()).map(serializeNodeChange) }, async getChangesSince(sinceLamport: number): Promise { - if (!adapter) return [] - - const rows = await adapter.query<{ - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer - }>('SELECT * FROM changes WHERE lamport_time > ? ORDER BY lamport_time ASC', [sinceLamport]) - - return rows.map(rowToSerializedChange) + return (await requireNodeStorage().getChangesSince(sinceLamport)).map(serializeNodeChange) }, async getChangeByHash(hash: string): Promise { - if (!adapter) return null - - const row = await adapter.queryOne<{ - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer - }>('SELECT * FROM changes WHERE hash = ?', [hash]) - - return row ? rowToSerializedChange(row) : null + const change = await requireNodeStorage().getChangeByHash(hash as ContentId) + return change ? serializeNodeChange(change) : null }, async getLastChange(nodeId: string): Promise { - if (!adapter) return null - - const row = await adapter.queryOne<{ - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer - }>('SELECT * FROM changes WHERE node_id = ? ORDER BY lamport_time DESC LIMIT 1', [nodeId]) - - return row ? rowToSerializedChange(row) : null + const change = await requireNodeStorage().getLastChange(nodeId) + return change ? serializeNodeChange(change) : null }, async getNode(id: string): Promise { - if (!adapter) return null - - const row = await adapter.queryOne<{ - id: string - schema_id: string - created_at: number - updated_at: number - created_by: string - deleted_at: number | null - }>('SELECT * FROM nodes WHERE id = ?', [id]) - - if (!row) return null - - // Get properties - const propRows = await adapter.query<{ - property_key: string - value: string | null - lamport_time: number - updated_by: string - updated_at: number - }>('SELECT * FROM node_properties WHERE node_id = ?', [id]) - - const properties: Record = {} - const timestamps: Record = {} - - for (const prop of propRows) { - properties[prop.property_key] = prop.value ? JSON.parse(prop.value) : null - timestamps[prop.property_key] = { - lamport: prop.lamport_time, - author: prop.updated_by, - wallTime: prop.updated_at - } - } - - // Get document content if exists - const yjsRow = await adapter.queryOne<{ state: Buffer }>( - 'SELECT state FROM yjs_state WHERE node_id = ?', - [id] - ) - - return { - id: row.id, - schemaId: row.schema_id, - properties, - timestamps, - deleted: row.deleted_at !== null, - deletedAt: row.deleted_at - ? { lamport: 0, author: row.created_by, wallTime: row.deleted_at } - : undefined, - createdAt: row.created_at, - createdBy: row.created_by, - updatedAt: row.updated_at, - updatedBy: row.created_by, // TODO: Track updatedBy separately - documentContent: yjsRow ? Array.from(yjsRow.state) : undefined - } + const node = await requireNodeStorage().getNode(id) + return node + ? { + ...node, + documentContent: node.documentContent ? Array.from(node.documentContent) : undefined + } + : null }, async getExistingNodeIds(ids: string[]): Promise { @@ -1749,14 +1630,12 @@ export function createDataService(config: DataServiceConfig): DataService { deleted_at: number | null }>(sql, params) - // Fetch full node state for each - const nodes: SerializedNodeState[] = [] - for (const row of rows) { - const node = await this.getNode(row.id) - if (node) nodes.push(node) - } - - return nodes + // One hydration read per page avoids a worker round-trip for every imported node. + const nodes = await requireNodeStorage().getNodes(rows.map((row) => row.id)) + return nodes.map((node) => ({ + ...node, + documentContent: node.documentContent ? Array.from(node.documentContent) : undefined + })) }, async countNodes(options?: CountNodesOptions): Promise { @@ -1852,39 +1731,6 @@ function serializeNodeChange(change: NodeChange): SerializedNodeChange { } } -function rowToSerializedChange(row: { - hash: string - node_id: string - payload: string - lamport_time: number - lamport_peer: string - wall_time: number - author: string - parent_hash: string | null - batch_id: string | null - signature: Buffer -}): SerializedNodeChange { - const payload = JSON.parse(row.payload) as { - nodeId: string - schemaId?: string - properties: Record - deleted?: boolean - } - - return { - id: row.hash, // Use hash as ID for now - type: 'node-change', - hash: row.hash, - payload, - lamport: row.lamport_time, - wallTime: row.wall_time, - authorDID: row.author, - parentHash: row.parent_hash, - batchId: row.batch_id ?? undefined, - signature: Array.from(row.signature) - } -} - function chunkItems(items: readonly T[], size: number): T[][] { const chunks: T[][] = [] for (let index = 0; index < items.length; index += size) { diff --git a/apps/electron/src/data-process/index.ts b/apps/electron/src/data-process/index.ts index a996427d5..98568a3b2 100644 --- a/apps/electron/src/data-process/index.ts +++ b/apps/electron/src/data-process/index.ts @@ -22,6 +22,8 @@ import type { DeterministicNodeImportDraft, NodeBatchWritePolicy } from '@xnetjs/data' import type { SyncReplicationConfig } from '@xnetjs/sync' +import { dirname } from 'node:path' +import { LibraryService } from '../library/service' import { createDataService, type DataService } from './data-service' // Debug logging - controllable via message from main process @@ -33,6 +35,7 @@ function log(...args: unknown[]): void { } let dataService: DataService | null = null +let library: LibraryService | null = null // Handle messages from main process via parentPort process.parentPort?.on('message', async (event) => { @@ -45,6 +48,89 @@ process.parentPort?.on('message', async (event) => { log('Received message:', type, requestId ? `(${requestId})` : '') try { + if (type.startsWith('library:')) { + if (!library) throw new Error('Library is not ready.') + let result: unknown + switch (type) { + case 'library:configure': + library.configure(payload as { authorDID: string; signingKey: number[] }) + result = true + break + case 'library:capture': + result = await library.capture(payload.input as import('../library/capture').CaptureInput) + break + case 'library:recover-captures': + await library.recoverCaptures() + result = true + break + case 'library:scan': + result = await library.scan() + break + case 'library:helper-status': + result = await library.helperStatus() + break + case 'library:helper-install': + result = await library.installHelper() + break + case 'library:helper-cancel': + library.cancelHelper() + result = true + break + case 'library:status': + result = library.status() + break + case 'library:search': + result = library.store.search( + payload as { text?: string; platform?: string; offset?: number; limit?: number } + ) + break + case 'library:get': + result = library.store.get(String(payload.id)) + break + case 'library:cards': + result = library.store.cards(payload.ids) + break + case 'library:graph': + // A single string avoids contextBridge recursively freezing tens of + // thousands of objects on the renderer's main thread. + result = JSON.stringify(await library.graph()) + break + case 'library:graph-detail': + if (typeof payload.id !== 'string' || !payload.id || payload.id.length > 500) + throw new Error('A valid Library resource ID is required.') + result = await library.graphDetail(payload.id) + break + case 'library:lookup': + result = await library.lookup(String(payload.url)) + break + case 'library:pause': + await library.pause() + result = true + break + case 'library:resume': + library.resume() + result = true + break + case 'library:retry': + library.retry(typeof payload.id === 'string' ? payload.id : undefined) + result = true + break + case 'library:freeze': + await library.freeze() + result = true + break + case 'library:thaw': + library.thaw() + result = true + break + default: + throw new Error('Unknown library operation') + } + sendResponse(requestId, { value: result }) + return + } + if (!dataService && !['init', 'shutdown'].includes(type)) + throw new Error('Workspace storage is not ready; no write was acknowledged.') switch (type) { // ─── Lifecycle ─────────────────────────────────────────────────────── @@ -53,12 +139,17 @@ process.parentPort?.on('message', async (event) => { log('Initializing data service with dbPath:', dbPath) dataService = createDataService({ dbPath }) await dataService.initialize() + library = new LibraryService(dataService, dirname(dbPath)) sendResponse(requestId, { success: true }) break } case 'shutdown': { log('Shutting down data service') + if (library) { + await library.close() + library = null + } if (dataService) { await dataService.shutdown() dataService = null @@ -301,6 +392,14 @@ process.parentPort?.on('message', async (event) => { // These handlers implement the NodeStorageAdapter interface for the renderer. // See: docs/explorations/0074_ELECTRON_IPC_NODE_STORAGE.md + case 'nodes:applyNodeBatch': { + const result = await dataService!.applyNodeBatch( + payload.input as import('../shared/node-batch').SerializedNodeBatch + ) + sendResponse(requestId, { result }) + break + } + case 'nodes:appendChange': { const { change } = payload as { change: unknown } if (dataService) { diff --git a/apps/electron/src/library/capture.test.ts b/apps/electron/src/library/capture.test.ts new file mode 100644 index 000000000..206a33774 --- /dev/null +++ b/apps/electron/src/library/capture.test.ts @@ -0,0 +1,210 @@ +import type { DataService } from '../data-process/data-service' +import { randomUUID } from 'node:crypto' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import * as Y from 'yjs' +import { + captureDocument, + readPageText, + saveCapture, + validateCapture, + type CaptureInput +} from './capture' +import { LibraryStore } from './store' + +let root: string +let store: LibraryStore +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-capture-')) + store = new LibraryStore(join(root, 'library.db')) +}) +afterEach(async () => { + store.close() + await rm(root, { recursive: true, force: true }) +}) +const input = (): CaptureInput => ({ + requestId: randomUUID(), + url: 'https://youtu.be/abcdefghijk?t=20', + title: 'My source note', + note: 'Why this matters to me.', + excerpt: 'A cited passage.' +}) +const identity = { authorDID: 'did:key:fixture', signingKey: Array(32).fill(1) as number[] } +function memoryData() { + const nodes = new Map>>>() + const documents = new Map() + let rejectDocument = false + const data: Pick< + DataService, + 'getNode' | 'getDocumentContent' | 'setDocumentContent' | 'importDeterministicNodes' + > = { + getNode: async (id) => nodes.get(id) ?? null, + getDocumentContent: async (id) => documents.get(id) ?? null, + setDocumentContent: async (id, bytes) => { + if (rejectDocument) throw new Error('Disk unavailable') + documents.set(id, bytes) + }, + importDeterministicNodes: async (options) => { + for (const draft of options.drafts) + nodes.set(draft.id, { + id: draft.id, + schemaId: draft.schemaId, + properties: draft.properties, + timestamps: {}, + createdAt: 1, + updatedAt: 1, + createdBy: identity.authorDID, + updatedBy: identity.authorDID, + deleted: false + }) + return { + batchId: 'fixture', + created: options.drafts.length, + updated: 0, + timings: {} as never + } + } + } + return { + data, + nodes, + documents, + failDocument: () => { + rejectDocument = true + }, + allowDocument: () => { + rejectDocument = false + } + } +} +it('saves an editable Page and independent citation with idempotent request retries', async () => { + const state = memoryData(), + request = input() + const first = await saveCapture({ input: request, store, data: state.data, identity }) + expect(await saveCapture({ input: request, store, data: state.data, identity })).toEqual(first) + expect(state.nodes.size).toBe(2) + expect(state.nodes.get(first.pageId)?.properties.sourceResources).toEqual([first.resourceId]) + expect(store.pendingCaptures()).toEqual([]) + expect(store.search({ text: 'matters' })[0].id).toBe(first.resourceId) + const doc = new Y.Doc() + Y.applyUpdate(doc, new Uint8Array(state.documents.get(first.pageId)!)) + expect(doc.getXmlFragment('content-v4').toString()).toContain('Why this matters to me.') + doc.destroy() +}) +it('retains the complete intent after a failed body write and resumes after reopening storage', async () => { + const state = memoryData(), + request = input() + state.failDocument() + await expect(saveCapture({ input: request, store, data: state.data, identity })).rejects.toThrow( + 'Disk unavailable' + ) + store.close() + store = new LibraryStore(join(root, 'library.db')) + expect(store.pendingCaptures()[0].input.note).toBe(request.note) + expect(store.pendingCaptures()[0].completed).toBe(false) + state.allowDocument() + const result = await saveCapture({ + input: store.pendingCaptures()[0].input, + store, + data: state.data, + identity + }) + expect(state.nodes.size).toBe(2) + expect(state.documents.has(result.pageId)).toBe(true) + expect(store.pendingCaptures()).toEqual([]) +}) +it('reuses a source while preserving separately requested notes and later edits', async () => { + const state = memoryData(), + request = input() + const first = await saveCapture({ input: request, store, data: state.data, identity }) + const edited = captureDocument({ ...request, note: 'My later edit.' }) + state.documents.set(first.pageId, edited) + const second = await saveCapture({ + input: { ...request, requestId: randomUUID(), note: 'Another independent note.' }, + store, + data: state.data, + identity + }) + expect(second.resourceId).toBe(first.resourceId) + expect(second.pageId).not.toBe(first.pageId) + expect(second.reusedResource).toBe(true) + expect(state.nodes.size).toBe(3) + await saveCapture({ input: request, store, data: state.data, identity }) + expect(state.documents.get(first.pageId)).toEqual(edited) +}) +it('refuses a different payload or identity under an existing retry key', async () => { + const state = memoryData(), + request = input() + await saveCapture({ input: request, store, data: state.data, identity }) + await expect( + saveCapture({ input: { ...request, note: 'Different' }, store, data: state.data, identity }) + ).rejects.toThrow('different capture') + await expect( + saveCapture({ + input: request, + store, + data: state.data, + identity: { ...identity, authorDID: 'did:key:other' } + }) + ).rejects.toThrow('different capture') + expect(() => validateCapture({ ...request, url: 'file:///private/data' })).toThrow() + expect(() => validateCapture({ ...request, note: 'a'.repeat(100001) })).toThrow( + 'not been discarded' + ) +}) + +it('keeps imported source properties and earlier personal notes when adding a capture', async () => { + const state = memoryData() + const first = await saveCapture({ input: input(), store, data: state.data, identity }) + const source = state.nodes.get(first.resourceId)! + const imported = { + ...source, + properties: { ...source.properties, title: 'Original export title', rawData: 'source-evidence' } + } + state.nodes.set(source.id, imported) + const next = await saveCapture({ + input: { ...input(), url: 'https://www.youtube.com/watch?v=abcdefghijk' }, + store, + data: state.data, + identity + }) + expect(next.resourceId).toBe(source.id) + expect(next.reusedResource).toBe(true) + expect(state.nodes.get(source.id)).toEqual(imported) + expect(store.get(source.id)?.notes?.map((note) => note.pageId)).toEqual([ + first.pageId, + next.pageId + ]) +}) + +it('replays an interrupted completion without overwriting a subsequently edited body', async () => { + const state = memoryData(), + request = input() + const write = store.saveCaptureIntent.bind(store) + const failure = vi.spyOn(store, 'saveCaptureIntent').mockImplementation((intent) => { + if (intent.completed) throw new Error('Completion write interrupted') + write(intent) + }) + await expect(saveCapture({ input: request, store, data: state.data, identity })).rejects.toThrow( + 'interrupted' + ) + const pending = store.pendingCaptures()[0] + const edited = captureDocument({ ...request, note: 'Later edit survives the recovery retry.' }) + state.documents.set(pending.result.pageId, edited) + failure.mockRestore() + await saveCapture({ input: request, store, data: state.data, identity }) + expect(readPageText(state.documents.get(pending.result.pageId)!)).toContain('Later edit survives') + expect(store.pendingCaptures()).toEqual([]) + expect(store.search({ text: 'survives' })[0].id).toBe(pending.result.resourceId) +}) + +it('does not index an unsupported document format as an empty note', () => { + const doc = new Y.Doc() + doc.getText('legacy-body').insert(0, 'Text that must not disappear from the index silently.') + expect(() => readPageText(Array.from(Y.encodeStateAsUpdate(doc)))).toThrow( + 'unsupported document format' + ) + doc.destroy() +}) diff --git a/apps/electron/src/library/capture.ts b/apps/electron/src/library/capture.ts new file mode 100644 index 000000000..2e6c8419e --- /dev/null +++ b/apps/electron/src/library/capture.ts @@ -0,0 +1,223 @@ +import type { LibraryStore } from './store' +import type { DataService } from '../data-process/data-service' +import type { CaptureInput, CaptureResult } from '../shared/library' +import { PageSchema } from '@xnetjs/data' +import { resourceIdentityForUrl } from '@xnetjs/social/import/core' +import { SocialContentSchema } from '@xnetjs/social/schemas' +import * as Y from 'yjs' + +export type { CaptureInput, CaptureResult } from '../shared/library' +export type CaptureIntent = { + version: 1 + input: CaptureInput + authorDID: string + result: CaptureResult + resource: ReturnType + document: number[] + completed: boolean + createdAt: number +} + +export function validateCapture(input: CaptureInput): CaptureInput { + if (!input || typeof input.requestId !== 'string' || !/^[\da-f-]{36}$/i.test(input.requestId)) + throw new Error('Capture request is missing its retry identity.') + for (const field of ['url', 'title', 'note', 'excerpt'] as const) + if (typeof input[field] !== 'string') throw new Error(`Capture ${field} must be text.`) + if ( + input.title.length > 500 || + input.note.length > 100000 || + input.excerpt.length > 20000 || + input.url.length > 500 + ) + throw new Error( + 'This capture exceeds the supported field size. Your input has not been discarded.' + ) + resourceIdentityForUrl(input.url) + return { ...input, url: input.url.trim(), title: input.title.trim() } +} + +export function captureDocument(input: CaptureInput): number[] { + const doc = new Y.Doc() + const group = new Y.XmlElement('blockGroup') + const lines = [ + input.url, + ...(input.note ? input.note.split('\n') : []), + ...(input.excerpt ? ['Excerpt', ...input.excerpt.split('\n')] : []) + ] + group.insert( + 0, + lines.map((line, index) => { + const block = new Y.XmlElement('blockContainer') + block.setAttribute('id', `capture-${input.requestId}-${index}`) + const paragraph = new Y.XmlElement('paragraph') + const text = new Y.XmlText() + if (index === 0) + text.applyDelta([{ insert: line, attributes: { link: { href: input.url } } }]) + else text.insert(0, line) + paragraph.insert(0, [text]) + block.insert(0, [paragraph]) + return block + }) + ) + doc.getXmlFragment('content-v4').insert(0, [group]) + const bytes = Array.from(Y.encodeStateAsUpdate(doc)) + doc.destroy() + return bytes +} + +export function readPageText(bytes: number[]): string { + const doc = new Y.Doc() + try { + Y.applyUpdate(doc, new Uint8Array(bytes)) + if (!doc.share.has('content-v4')) + throw new Error( + 'The source note uses an unsupported document format; search was not marked complete.' + ) + const read = (node: unknown): string => { + if (node instanceof Y.XmlText) + return (node.toDelta() as { insert?: unknown }[]) + .map((part) => (typeof part.insert === 'string' ? part.insert : '')) + .join('') + if ( + node instanceof Y.XmlElement && + ['wikilink', 'mention', 'hashtag'].includes(node.nodeName) + ) { + const attrs = node.getAttributes() + return String(attrs.title ?? attrs.label ?? attrs.name ?? '') + } + if (node instanceof Y.XmlElement || node instanceof Y.XmlFragment) + return node.toArray().map(read).join('\n') + throw new Error('The source note contains an unsupported document node.') + } + return read(doc.getXmlFragment('content-v4')) + } finally { + doc.destroy() + } +} + +export async function saveCapture(options: { + input: CaptureInput + store: LibraryStore + data: Pick< + DataService, + 'getNode' | 'getDocumentContent' | 'setDocumentContent' | 'importDeterministicNodes' + > + identity: { authorDID: string; signingKey: number[] } +}): Promise { + const input = validateCapture(options.input) + const { store, data, identity } = options + let intent = store.captureIntent(input.requestId) + if ( + intent && + (intent.authorDID !== identity.authorDID || + JSON.stringify(intent.input) !== JSON.stringify(input)) + ) + throw new Error('This retry belongs to a different capture or workspace identity.') + if (!intent) { + const resource = resourceIdentityForUrl(input.url) + const cached = store.byUrl(resource.url) + const existing = await data.getNode(cached?.id ?? resource.id) + if (existing?.deleted) + throw new Error('This source was removed. Restore it before adding another note.') + intent = { + version: 1, + input, + authorDID: identity.authorDID, + resource, + result: { + resourceId: existing?.id ?? resource.id, + pageId: `capture:${input.requestId}`, + reusedResource: !!existing + }, + document: captureDocument(input), + completed: false, + createdAt: Date.now() + } + store.saveCaptureIntent(intent) + } + const { resourceId, pageId } = intent.result + if (intent.completed) { + const savedPage = await data.getNode(pageId) + if (!savedPage || savedPage.deleted) + throw new Error('The saved note was removed; start a new capture to save another.') + return intent.result + } + const resource = await data.getNode(resourceId) + const page = await data.getNode(pageId) + if (resource?.deleted || page?.deleted) + throw new Error( + 'A record from this interrupted capture was removed. Its recovery text was kept.' + ) + const drafts = [ + ...(!resource + ? [ + { + id: resourceId, + schemaId: SocialContentSchema._schemaId, + properties: { + platform: intent.resource.platform, + platformContentId: intent.resource.nativeId, + contentKind: intent.resource.kind, + canonicalUrl: intent.resource.url, + title: input.title || intent.resource.url, + privacyClass: 'private', + visibility: 'private' + } + } + ] + : []), + ...(!page + ? [ + { + id: pageId, + schemaId: PageSchema._schemaId, + properties: { + title: input.title || input.url, + visibility: 'private', + sourceResources: [resourceId] + } + } + ] + : []) + ] + if (drafts.length) await data.importDeterministicNodes({ ...identity, drafts }) + // A retry never replaces text the user has already opened and edited. + if (!(await data.getDocumentContent(pageId))) + await data.setDocumentContent(pageId, intent.document) + const properties = resource?.properties + const value = (key: string) => + typeof properties?.[key] === 'string' ? (properties[key] as string) : '' + store.seed({ + id: resourceId, + platform: value('platform') || intent.resource.platform, + platformContentId: value('platformContentId') || intent.resource.nativeId, + url: value('canonicalUrl') || intent.resource.url, + title: value('title') || input.title || input.url, + sourceText: value('searchText'), + actor: value('actorHandle'), + privacy: value('privacyClass') || 'private', + addedAt: resource?.createdAt ?? intent.createdAt + }) + const savedDocument = await data.getDocumentContent(pageId) + if (!savedDocument) + throw new Error('The captured Page has no saved text. The recovery record was retained.') + const saved = store.get(resourceId)! + const next = { + ...saved, + notes: [ + ...(saved.notes ?? []).filter((note) => note.id !== pageId), + { + id: pageId, + title: input.title || input.url, + text: readPageText(savedDocument), + url: input.url, + author: identity.authorDID, + pageId + } + ] + } + store.put(next) + store.index(next) + store.saveCaptureIntent({ ...intent, completed: true }) + return intent.result +} diff --git a/apps/electron/src/library/conversations.test.ts b/apps/electron/src/library/conversations.test.ts new file mode 100644 index 000000000..bd4ff9e9c --- /dev/null +++ b/apps/electron/src/library/conversations.test.ts @@ -0,0 +1,60 @@ +import type { DataService } from '../data-process/data-service' +import { SocialConversationSchema, SocialMessageSchema } from '@xnetjs/social/schemas' +import { expect, it, vi } from 'vitest' +import { conversationResources } from './conversations' + +type SourceNode = Awaited>[number] +const node = (id: string, properties: Record): SourceNode => ({ + id, + schemaId: '', + properties, + createdAt: 1, + updatedAt: 1, + createdBy: 'did:key:test', + deleted: false, + timestamps: {}, + updatedBy: 'did:key:test' +}) +it('indexes full AI conversation text in order while excluding unrelated private messages', async () => { + const listNodes = vi.fn(async ({ schemaId }: { schemaId?: string } = {}) => + schemaId === SocialConversationSchema._schemaId + ? [ + node('chat', { + platform: 'claude', + conversationKind: 'ai-chat', + title: 'A conversation', + privacyClass: 'private' + }), + node('dm', { platform: 'x', conversationKind: 'dm' }) + ] + : schemaId === SocialMessageSchema._schemaId + ? [ + node('late', { + conversation: 'chat', + senderHandle: 'assistant', + sentAt: '2026-10-02', + searchText: 'Long text '.repeat(4000) + 'tailmarker' + }), + node('early', { + conversation: 'chat', + senderHandle: 'user', + sentAt: '2026-10-01', + searchText: 'First question' + }), + node('excluded', { conversation: 'dm', searchText: 'Not in Library' }) + ] + : [] + ) + const results = [] + for await (const resource of conversationResources({ listNodes })) results.push(resource) + expect(results).toHaveLength(1) + expect(results[0]).toMatchObject({ + id: 'chat', + kind: 'conversation', + privacy: 'private', + url: '' + }) + expect(results[0].sourceText).toMatch(/^user.*\nFirst question/s) + expect(results[0].sourceText).toContain('tailmarker') + expect(results[0].sourceText).not.toContain('Not in Library') +}) diff --git a/apps/electron/src/library/conversations.ts b/apps/electron/src/library/conversations.ts new file mode 100644 index 000000000..22a3e30d5 --- /dev/null +++ b/apps/electron/src/library/conversations.ts @@ -0,0 +1,62 @@ +import type { LibraryResource } from './types' +import type { DataService } from '../data-process/data-service' +import { SocialConversationSchema, SocialMessageSchema } from '@xnetjs/social/schemas' + +const text = (value: unknown) => (typeof value === 'string' ? value : '') +type SourceNode = Awaited>[number] + +async function* nodes(data: Pick, schemaId: string) { + for (let offset = 0; ; offset += 500) { + const page = await data.listNodes({ + schemaId, + limit: 500, + offset, + orderBy: { createdAt: 'asc' } + }) + yield* page + if (page.length < 500) return + } +} + +/** Rebuildable local text projection. Private conversations never need a network fetch. */ +export async function* conversationResources( + data: Pick +): AsyncGenerator { + const conversations = new Map() + for await (const node of nodes(data, SocialConversationSchema._schemaId)) { + if (node.properties.conversationKind === 'ai-chat') + conversations.set(node.id, { node, messages: [] }) + } + if (!conversations.size) return + for await (const message of nodes(data, SocialMessageSchema._schemaId)) { + conversations.get(text(message.properties.conversation))?.messages.push(message) + } + for (const { node, messages } of conversations.values()) { + const props = node.properties + const platform = text(props.platform) + const sourceText = messages + .sort( + (a, b) => + text(a.properties.sentAt).localeCompare(text(b.properties.sentAt)) || + a.createdAt - b.createdAt || + a.id.localeCompare(b.id) + ) + .map( + ({ properties: message }) => + `${text(message.senderHandle) || 'Message'}${message.sentAt ? ` · ${text(message.sentAt)}` : ''}\n${text(message.searchText) || text(message.textPreview)}` + ) + .join('\n\n') + yield { + id: node.id, + kind: 'conversation', + platform, + platformContentId: text(props.platformConversationId) || node.id, + url: '', + title: text(props.title) || `Saved ${platform} conversation`, + sourceText, + actor: '', + privacy: text(props.privacyClass) || 'private', + addedAt: node.createdAt + } + } +} diff --git a/apps/electron/src/library/extractor-discovery.test.ts b/apps/electron/src/library/extractor-discovery.test.ts new file mode 100644 index 000000000..7b3a90424 --- /dev/null +++ b/apps/electron/src/library/extractor-discovery.test.ts @@ -0,0 +1,39 @@ +import { access } from 'node:fs/promises' +import { beforeEach, expect, it, vi } from 'vitest' +import { LibraryHelperError, LibraryHelperProbeError, verifyHelperVersion } from './managed-helper' +import { findExtractor } from './providers' + +vi.mock('node:fs/promises', async (original) => ({ + ...(await original()), + access: vi.fn() +})) +vi.mock('./managed-helper', async (original) => ({ + ...(await original()), + verifyHelperVersion: vi.fn() +})) +beforeEach(() => { + vi.mocked(access).mockReset().mockResolvedValue(undefined) + vi.mocked(verifyHelperVersion).mockReset() +}) +it('retries an installed helper whose version probe returned no output', async () => { + vi.mocked(verifyHelperVersion).mockRejectedValue(new LibraryHelperProbeError('Empty probe')) + await expect(findExtractor()).rejects.toMatchObject({ + disposition: 'retry', + message: 'Empty probe' + }) +}) +it('uses a verified fallback after an empty version probe', async () => { + vi.mocked(verifyHelperVersion) + .mockRejectedValueOnce(new LibraryHelperProbeError('Empty probe')) + .mockResolvedValueOnce(undefined) + await expect(findExtractor()).resolves.toBe('/opt/homebrew/bin/yt-dlp') +}) +it('still blocks installed helpers with a different version', async () => { + vi.mocked(verifyHelperVersion).mockRejectedValue(new LibraryHelperError('Wrong version')) + await expect(findExtractor()).rejects.toMatchObject({ disposition: 'blocked' }) +}) +it('still blocks when no helper is installed', async () => { + vi.mocked(access).mockRejectedValue(Object.assign(new Error('Missing'), { code: 'ENOENT' })) + await expect(findExtractor()).rejects.toMatchObject({ disposition: 'blocked' }) + expect(verifyHelperVersion).not.toHaveBeenCalled() +}) diff --git a/apps/electron/src/library/graph.test.ts b/apps/electron/src/library/graph.test.ts new file mode 100644 index 000000000..3bb3d1b63 --- /dev/null +++ b/apps/electron/src/library/graph.test.ts @@ -0,0 +1,78 @@ +import type { GraphResource } from './graph' +import { expect, it } from 'vitest' +import { createGraphBuilder, hashtagsIn } from './graph' + +const resource = (id: string, values: Partial = {}): GraphResource => ({ + id, + title: id, + url: `https://example.com/${id}`, + platform: 'youtube', + provider: 'youtube', + author: '', + hashtags: [], + ...values +}) + +it('keeps every link, including orphans, and deduplicates repeated memberships', () => { + const builder = createGraphBuilder( + Array.from({ length: 1200 }, (_, i) => resource(`${i}`)), + 1500 + ) + builder.collection('playlist', { title: 'Learning', platform: 'youtube' }) + builder.collection('empty-export', { title: 'A name without entries' }) + builder.membership({ collection: 'playlist', item: '1199' }) + builder.membership({ collection: 'playlist', item: '1199' }) + builder.membership({ collection: 'unknown', item: '0' }) + builder.membership({ collection: 'playlist', item: 'not-a-link' }) + const graph = builder.finish() + expect(graph.linkCount).toBe(1200) + expect(graph.resourceCount).toBe(1500) + expect(graph.nodes).toHaveLength(1201) + expect(graph.edges).toHaveLength(1) + expect(graph.nodes[graph.edges[0].source].id).toBe('1199') + expect(graph.nodes[graph.edges[0].target].label).toBe('Learning') +}) + +it('joins explicit tags across platforms but keeps creator names scoped to their provider', () => { + const builder = createGraphBuilder( + [ + resource('one', { author: 'Alex', hashtags: ['Learning'] }), + resource('two', { author: 'Alex', platform: 'github', provider: 'github' }), + resource('three', { author: 'alex' }) + ], + 3 + ) + builder.content('two', { + platform: 'github', + metadataJson: JSON.stringify({ topics: ['learning'], language: 'TypeScript' }) + }) + const graph = builder.finish() + expect(graph.nodes.filter((node) => node.kind === 'tag')).toHaveLength(1) + expect(graph.nodes.filter((node) => node.kind === 'creator')).toHaveLength(2) + expect(graph.nodes.filter((node) => node.kind === 'category').map((node) => node.label)).toEqual([ + 'Language: TypeScript' + ]) + expect(graph.edges.filter((edge) => edge.kind === 'tag')).toHaveLength(2) +}) + +it('attaches garden tags to the link and reports unreadable source metadata', () => { + const builder = createGraphBuilder([resource('saved')], 1) + builder.content('note', { + platformContentKind: 'garden-commentary', + parentContent: 'saved', + metadataJson: JSON.stringify({ tags: ['meditation'] }) + }) + builder.content('saved', { metadataJson: '{broken' }) + const graph = builder.finish() + expect(graph.nodes.map((node) => node.label)).toEqual(['saved', '#meditation']) + expect(graph.warnings).toHaveLength(1) + expect(graph.warnings[0]).toContain('incomplete') +}) + +it('extracts actual hashtags without interpreting headings, URL fragments, or ordinary words as tags', () => { + expect( + hashtagsIn( + 'Learning #Mindfulness #éducation! #Mindfulness\n# A heading\nhttps://example.com/#fragment and plain words' + ) + ).toEqual(['Mindfulness', 'éducation']) +}) diff --git a/apps/electron/src/library/graph.ts b/apps/electron/src/library/graph.ts new file mode 100644 index 000000000..c4575ee6f --- /dev/null +++ b/apps/electron/src/library/graph.ts @@ -0,0 +1,204 @@ +import type { LibraryStore } from './store' +import type { DataService } from '../data-process/data-service' +import type { LibraryGraph, LibraryGraphKind, LibraryGraphRelation } from '../shared/library-graph' +import { + SocialCollectionSchema, + SocialCollectionItemSchema, + SocialContentSchema +} from '@xnetjs/social/schemas' + +export type GraphResource = { + id: string + url: string + title: string + platform: string + provider: string + author: string + hashtags: string[] +} + +const text = (value: unknown): string => (typeof value === 'string' ? value.trim() : '') +const normalized = (value: string) => value.normalize('NFKC').toLocaleLowerCase('en-US') +const labels = (value: unknown): string[] => + Array.isArray(value) + ? value + .map((item: unknown) => + typeof item === 'string' + ? item + : item && typeof item === 'object' && 'text' in item + ? text(item.text) + : '' + ) + .filter(Boolean) + : [] + +export const hashtagsIn = (value: string): string[] => [ + ...new Set( + Array.from( + value.matchAll(/(?:^|\s)#([\p{L}\p{N}_][\p{L}\p{N}_-]{0,79})(?=\s|[.,!?;:]|$)/gu), + (match) => match[1] + ) + ) +] + +/** Hubs preserve source relationships without expanding a playlist into a quadratic clique. */ +export function createGraphBuilder(resources: GraphResource[], resourceCount: number) { + const graph: LibraryGraph = { + nodes: resources.map((resource) => ({ + id: resource.id, + label: resource.title || resource.url, + url: resource.url, + platform: resource.platform, + kind: 'link' + })), + edges: [], + linkCount: resources.length, + resourceCount, + warnings: [], + builtAt: Date.now() + } + const index = new Map(graph.nodes.map((node, i) => [node.id, i])) + const edges = new Set() + let unreadable = 0 + const hub = (kind: LibraryGraphKind, key: string, label: string, platform = '') => { + const id = `${kind}:${key}` + const previous = index.get(id) + if (previous !== undefined) return previous + const position = graph.nodes.length + index.set(id, position) + graph.nodes.push({ id, kind, label, platform }) + return position + } + const connect = ( + id: string, + target: number, + kind: LibraryGraphRelation, + evidence: 'import' | 'metadata' | 'hashtag' + ) => { + const source = index.get(id) + if (source === undefined || graph.nodes[source].kind !== 'link') return + const key = `${source}:${target}` + if (edges.has(key)) return + edges.add(key) + graph.edges.push({ source, target, kind, evidence }) + } + const tag = (id: string, label: string, evidence: 'import' | 'hashtag') => { + const clean = label.trim().replace(/^#/, '') + if (clean) connect(id, hub('tag', normalized(clean), `#${clean}`), 'tag', evidence) + } + for (const resource of resources) { + if (resource.author.trim()) + connect( + resource.id, + hub( + 'creator', + `${resource.provider}:${normalized(resource.author)}`, + resource.author, + resource.provider + ), + 'creator', + 'metadata' + ) + resource.hashtags.forEach((label) => tag(resource.id, label, 'hashtag')) + } + return { + collection(id: string, properties: Record) { + const label = text(properties.title) || 'Untitled collection' + hub('collection', id, label, text(properties.platform)) + }, + membership(properties: Record) { + const target = index.get(`collection:${text(properties.collection)}`) + if (target !== undefined) connect(text(properties.item), target, 'collection', 'import') + }, + content(id: string, properties: Record) { + // Garden commentary belongs to the saved link, not to a duplicate link node. + const resourceId = + properties.platformContentKind === 'garden-commentary' ? text(properties.parentContent) : id + if (!index.has(resourceId)) return + const raw = text(properties.metadataJson) + if (!raw) return + let metadata: Record + try { + const parsed: unknown = JSON.parse(raw) + if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) + throw new Error('Invalid metadata') + metadata = parsed as Record + } catch { + unreadable++ + return + } + for (const label of [ + ...labels(metadata.tags), + ...labels(metadata.topics), + ...labels(metadata.hashtags) + ]) + tag(resourceId, label, 'import') + const categories = [ + ...labels(metadata.categories), + text(metadata.category), + text(metadata.subreddit) ? `r/${text(metadata.subreddit).replace(/^r\//, '')}` : '', + properties.platform === 'github' && text(metadata.language) + ? `Language: ${text(metadata.language)}` + : '' + ] + for (const label of categories.filter(Boolean)) + connect(resourceId, hub('category', normalized(label), label), 'category', 'import') + }, + finish(): LibraryGraph { + // An empty exported collection makes no claim about any link's membership. + const used = new Set(graph.edges.flatMap((edge) => [edge.source, edge.target])) + const remap = new Map() + const nodes = graph.nodes.filter((node, i) => { + if (node.kind !== 'link' && !used.has(i)) return false + remap.set(i, remap.size) + return true + }) + if (unreadable) + graph.warnings.push( + `${unreadable.toLocaleString()} source metadata records could not be read; their tag relationships are incomplete.` + ) + return { + ...graph, + nodes, + edges: graph.edges.map((edge) => ({ + ...edge, + source: remap.get(edge.source)!, + target: remap.get(edge.target)! + })) + } + } + } +} + +export async function readLibraryGraph( + data: Pick, + store: LibraryStore +): Promise { + const builder = createGraphBuilder(store.graphResources(), store.status().resources) + for (const [schemaId, consume] of [ + [ + SocialCollectionSchema._schemaId, + (id: string, props: Record) => builder.collection(id, props) + ], + [ + SocialCollectionItemSchema._schemaId, + (_id: string, props: Record) => builder.membership(props) + ], + [ + SocialContentSchema._schemaId, + (id: string, props: Record) => builder.content(id, props) + ] + ] as const) { + for (let offset = 0; ; offset += 500) { + const nodes = await data.listNodes({ + schemaId, + orderBy: { createdAt: 'asc' }, + limit: 500, + offset + }) + nodes.forEach((node) => consume(node.id, node.properties)) + if (nodes.length < 500) break + } + } + return builder.finish() +} diff --git a/apps/electron/src/library/instagram.test.ts b/apps/electron/src/library/instagram.test.ts new file mode 100644 index 000000000..ef9f5d8f8 --- /dev/null +++ b/apps/electron/src/library/instagram.test.ts @@ -0,0 +1,79 @@ +import { expect, it } from 'vitest' +import { parseInstagramPage } from './instagram' + +const embed = (body: string, kind = 'reel') => ` + + + + example.creator + + ${body} +` + +it('keeps full written captions, line breaks, entities and hashtags without comments or scripts', () => { + const tail = 'late phrase '.repeat(1000) + const result = parseInstagramPage( + embed(`
+ example.creator

+ A "quote" & 🌿 'note'
Second line + #example${tail} +
View all 123 comments
+
`), + 'abc123' + ) + expect(result.provider).toBe('instagram-embed/1') + expect(result.author).toBe('example.creator') + expect(result.thumbnailUrl).toBe('https://images.example/post.jpg?x=1&y=2') + expect(result.description).toContain('A "quote" & 🌿 \'note\'\nSecond line') + expect(result.description).toContain('#example' + tail.trim()) + expect(result.description).not.toContain('View all') + expect(result.description).not.toContain('never execute') + expect(result.description).not.toContain('example.creator') + expect(result.fields.description.state).toBe('complete') + expect(result.fields.captions.state).toBe('unavailable') + expect(result.tracks).toBeUndefined() +}) + +it('uses the post poster for photos and carousels and preserves an explicitly empty caption', () => { + const result = parseInstagramPage( + embed('', 'p'), + 'abc123' + ) + expect(result.thumbnailUrl).toContain('post.jpg') + expect(result.description).toBe('') + expect(result.title).toBe('Instagram post by example.creator') + expect(result.fields.description.state).toBe('complete') +}) + +it('marks an absent caption as unavailable and never substitutes an avatar', () => { + const result = parseInstagramPage(embed(''), 'abc123') + expect(result.fields.description.state).toBe('unavailable') + expect(() => + parseInstagramPage(embed('').replace(/]*>/, ''), 'abc123') + ).toThrow('no post text or thumbnail') +}) + +it('keeps Open Graph fallback coverage partial and handles quotes in attributes', () => { + const result = parseInstagramPage( + ` + + + + + `, + 'abc123' + ) + expect(result.author).toBe('Creator') + expect(result.description).toBe("It's a public preview & more") + expect(result.fields.description.state).toBe('partial') + expect(result.provider).toBe('instagram-page/1') +}) + +it('rejects login pages, unrelated posts and an external canonical host', () => { + for (const html of [ + 'Log in', + embed('
Another post
').replace('abc123', 'another'), + '' + ]) + expect(() => parseInstagramPage(html, 'abc123')).toThrow() +}) diff --git a/apps/electron/src/library/instagram.ts b/apps/electron/src/library/instagram.ts new file mode 100644 index 000000000..ebd82ee9a --- /dev/null +++ b/apps/electron/src/library/instagram.ts @@ -0,0 +1,129 @@ +import type { LibraryMetadata } from './types' +import type { DefaultTreeAdapterTypes } from 'parse5' +import { parse, serializeOuter } from 'parse5' +import { LibraryProviderError } from './provider-error' + +type Element = DefaultTreeAdapterTypes.Element +type Node = DefaultTreeAdapterTypes.Node +const attribute = (node: Element, name: string) => + node.attrs.find((entry) => entry.name === name)?.value +const hasClass = (node: Element, name: string) => + (attribute(node, 'class') ?? '').split(/\s+/).includes(name) + +function elements(root: Node): Element[] { + const found: Element[] = [] + const pending = [root] + while (pending.length) { + const node = pending.pop()! + if ('tagName' in node) found.push(node) + if ('childNodes' in node) pending.push(...[...node.childNodes].reverse()) + } + return found +} + +function readableText(root: Node): string { + const parts: string[] = [] + const pending = [root] + while (pending.length) { + const node = pending.pop()! + if ('tagName' in node) { + if ( + ['script', 'style', 'template'].includes(node.tagName) || + hasClass(node, 'CaptionUsername') || + hasClass(node, 'CaptionComments') + ) + continue + if (node.tagName === 'br') parts.push('\n') + } + if ('value' in node && node.nodeName === '#text') parts.push(node.value) + if ('childNodes' in node) pending.push(...[...node.childNodes].reverse()) + } + return parts.join('').trim() +} + +function identifiesPost(value: string | undefined, shortcode: string): boolean { + if (!value) return false + try { + const url = new URL(value, 'https://www.instagram.com') + return ( + ['instagram.com', 'www.instagram.com'].includes(url.hostname) && + url.pathname.match(/^\/(?:[^/]+\/)?(?:p|reels?|tv)\/([A-Za-z0-9_-]+)\/?$/)?.[1] === shortcode + ) + } catch { + return false + } +} + +/** Read public post markup as data, without executing scripts or loading embedded media. */ +export function parseInstagramPage(html: string, shortcode: string): LibraryMetadata { + const nodes = elements(parse(html)) + const meta = (name: string) => + nodes + .filter((node) => node.tagName === 'meta') + .find((node) => attribute(node, 'property') === name) + const metaValue = (name: string) => { + const node = meta(name) + return node ? attribute(node, 'content') : undefined + } + const media = nodes.find((node) => hasClass(node, 'EmbeddedMedia')) + const embedded = !!media && identifiesPost(attribute(media, 'href'), shortcode) + if (media && !embedded) + throw new LibraryProviderError('Instagram returned a different post in its embed.', 'blocked') + if (!embedded && !identifiesPost(metaValue('og:url'), shortcode)) + throw new LibraryProviderError( + 'Instagram did not return this post. It may require sign-in, be private, or have been removed.', + 'blocked' + ) + + const caption = embedded ? nodes.find((node) => hasClass(node, 'Caption')) : undefined + const username = embedded ? nodes.find((node) => hasClass(node, 'UsernameText')) : undefined + const author = username + ? readableText(username) + : metaValue('og:title')?.match(/^(.*?) on Instagram:/)?.[1] + const image = embedded + ? nodes.find((node) => node.tagName === 'img' && hasClass(node, 'EmbeddedMediaImage')) + : undefined + const thumbnailUrl = image ? attribute(image, 'src') : metaValue('og:image') + const description = caption ? readableText(caption) : metaValue('og:description') + if (!thumbnailUrl && !description) + throw new LibraryProviderError('Instagram returned no post text or thumbnail.', 'blocked') + const title = + description?.split('\n')[0].slice(0, 180) || + (author ? `Instagram post by ${author}` : `Instagram post ${shortcode}`) + return { + title, + description, + author, + thumbnailUrl, + fields: { + title: { + state: 'partial', + reason: 'Display label from the written post; Instagram does not expose a separate title.' + }, + description: caption + ? { state: 'complete' } + : description + ? { + state: 'partial', + reason: 'Public page preview; the full written caption was not returned.' + } + : { state: 'unavailable', reason: 'The public embed returned no written caption.' }, + captions: { + state: 'unavailable', + reason: + 'The public Instagram post does not expose spoken caption tracks. Its written caption is saved separately; local transcription has not run.' + } + }, + provider: embedded ? 'instagram-embed/1' : 'instagram-page/1', + fetchedAt: Date.now(), + evidence: { + shortcode, + sourceUrl: embedded ? attribute(media!, 'href') : metaValue('og:url'), + captionHtml: caption ? serializeOuter(caption) : undefined, + pageTitle: metaValue('og:title'), + pageDescription: metaValue('og:description'), + author, + thumbnailUrl + } + } +} diff --git a/apps/electron/src/library/managed-helper.test.ts b/apps/electron/src/library/managed-helper.test.ts new file mode 100644 index 000000000..45640eab0 --- /dev/null +++ b/apps/electron/src/library/managed-helper.test.ts @@ -0,0 +1,151 @@ +import { createHash } from 'node:crypto' +import { mkdtemp, readFile, readdir, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { + inspectManagedHelper, + installManagedHelper, + verifyHelperVersion, + LibraryHelperError, + LibraryHelperProbeError +} from './managed-helper' + +const bytes = Buffer.from('a fixture helper executable') +const artifact = { + version: 'fixture', + filename: 'fixture-helper', + url: 'https://example.invalid/helper', + size: bytes.length, + sha256: createHash('sha256').update(bytes).digest('hex') +} +let root: string +let directory: string +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-helper-')) + directory = join(root, 'helpers') +}) +afterEach(() => rm(root, { recursive: true, force: true })) +const install = (overrides: Partial[0]> = {}) => + installManagedHelper({ + directory, + artifact, + signal: new AbortController().signal, + download: async () => new Response(bytes), + verifyVersion: async () => {}, + ...overrides + }) + +it('promotes only verified bytes and avoids downloading an already ready helper', async () => { + expect((await inspectManagedHelper(directory, artifact)).state).toBe('missing') + const verifyVersion = vi.fn(async (path: string) => { + expect(await readFile(path)).toEqual(bytes) + expect(await readdir(directory)).not.toContain(artifact.filename) + }) + expect((await install({ verifyVersion })).state).toBe('ready') + expect(verifyVersion).toHaveBeenCalledOnce() + const download = vi.fn(async () => new Response(bytes)) + expect((await install({ download })).state).toBe('ready') + expect(download).not.toHaveBeenCalled() + expect(await readdir(directory)).toEqual([artifact.filename]) +}) + +it('preserves the complete binary when the download arrives in several chunks', async () => { + await install({ + download: async () => + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(bytes.subarray(0, 4)) + controller.enqueue(bytes.subarray(4, 13)) + controller.enqueue(bytes.subarray(13)) + controller.close() + } + }) + ) + }) + expect(await readFile(join(directory, artifact.filename))).toEqual(bytes) +}) + +it.each(['short', 'oversized', 'checksum', 'version', 'http'])( + 'never publishes a helper after a %s failure', + async (failure) => { + const payload = + failure === 'short' + ? bytes.subarray(1) + : failure === 'oversized' + ? Buffer.concat([bytes, bytes]) + : failure === 'checksum' + ? Buffer.alloc(bytes.length) + : bytes + const verifyVersion = vi.fn(async () => { + if (failure === 'version') throw new Error('Unexpected version') + }) + await expect( + install({ + verifyVersion, + download: async () => new Response(payload, { status: failure === 'http' ? 503 : 200 }) + }) + ).rejects.toThrow() + expect(await readdir(directory)).toEqual([]) + if (failure !== 'version') expect(verifyVersion).not.toHaveBeenCalled() + } +) + +it('detects modified installed bytes and preserves them when a repair download fails', async () => { + await install() + const path = join(directory, artifact.filename) + // chmod represents an external modification by the owning user. + const { chmod } = await import('node:fs/promises') + await chmod(path, 0o700) + const damaged = Buffer.alloc(bytes.length, 120) + await writeFile(path, damaged) + expect((await inspectManagedHelper(directory, artifact)).state).toBe('damaged') + await expect( + install({ + download: async () => { + throw new Error('offline') + } + }) + ).rejects.toThrow('offline') + expect(await readFile(path)).toEqual(damaged) + expect((await install()).state).toBe('ready') +}) + +it('cancels before publication and removes partial bytes', async () => { + const controller = new AbortController() + await expect( + install({ + signal: controller.signal, + verifyVersion: async (_path, _version, signal) => { + expect(signal).toBe(controller.signal) + controller.abort() + signal.throwIfAborted() + } + }) + ).rejects.toThrow() + expect(await readdir(directory)).toEqual([]) +}) + +it('rejects a helper directory symlink before downloading', async () => { + await symlink(root, directory) + const download = vi.fn(async () => new Response(bytes)) + await expect(install({ download })).rejects.toThrow('regular directory') + expect(download).not.toHaveBeenCalled() +}) + +// A real child process distinguishes empty success output from a version mismatch. +it.skipIf(process.platform === 'win32')( + 'retries an empty helper probe without accepting it as a verified version', + async () => { + const path = join(root, 'probe') + await writeFile(path, '#!/bin/sh\nexit 0\n', { mode: 0o700 }) + await expect(verifyHelperVersion(path, 'fixture')).rejects.toBeInstanceOf( + LibraryHelperProbeError + ) + await writeFile(path, '#!/bin/sh\nprintf "wrong\\n"\n', { mode: 0o700 }) + await expect(verifyHelperVersion(path, 'fixture')).rejects.toBeInstanceOf(LibraryHelperError) + await writeFile(path, '#!/bin/sh\nprintf "fixture\\n"\n', { mode: 0o700 }) + await expect(verifyHelperVersion(path, 'fixture')).resolves.toBeUndefined() + } +) diff --git a/apps/electron/src/library/managed-helper.ts b/apps/electron/src/library/managed-helper.ts new file mode 100644 index 000000000..0f5d74933 --- /dev/null +++ b/apps/electron/src/library/managed-helper.ts @@ -0,0 +1,165 @@ +import type { LibraryHelperStatus } from '../shared/library' +import { execFile } from 'node:child_process' +import { createHash, randomUUID } from 'node:crypto' +import { lstat, mkdir, open, readFile, rename, rm } from 'node:fs/promises' +import { join } from 'node:path' +import { promisify } from 'node:util' +import { TaggedError } from '@xnetjs/core' + +const exec = promisify(execFile) +export const TESTED_EXTRACTOR_VERSION = '2026.07.04' +// Official immutable release asset; the digest is pinned in source, never trusted from a download. +export const MAC_VIDEO_HELPER = { + version: TESTED_EXTRACTOR_VERSION, + filename: `yt-dlp-${TESTED_EXTRACTOR_VERSION}-macos`, + url: `https://github.com/yt-dlp/yt-dlp/releases/download/${TESTED_EXTRACTOR_VERSION}/yt-dlp_macos`, + size: 38_256_544, + sha256: '498bd0dae17855c599d371d68ec5bafc439a9d8640e838be25c765a9792f261b' +} as const + +type HelperArtifact = { + version: string + filename: string + url: string + size: number + sha256: string +} +export class LibraryHelperError extends TaggedError { + readonly _tag = 'LibraryHelperError' +} +export class LibraryHelperProbeError extends TaggedError { + readonly _tag = 'LibraryHelperProbeError' +} +const verified = new Map() +const message = (error: unknown) => (error instanceof Error ? error.message : String(error)) +const digest = (bytes: Uint8Array) => createHash('sha256').update(bytes).digest('hex') +const pathFor = (directory: string, artifact: HelperArtifact) => { + if (!/^[a-zA-Z0-9._-]+$/.test(artifact.filename) || artifact.filename.startsWith('.')) + throw new LibraryHelperError('Invalid helper filename') + return join(directory, artifact.filename) +} + +export async function inspectManagedHelper( + directory: string, + artifact: HelperArtifact = MAC_VIDEO_HELPER +): Promise { + const base = { version: artifact.version, bytes: artifact.size } + try { + const root = await lstat(directory) + if (!root.isDirectory() || root.isSymbolicLink()) + throw new LibraryHelperError('Helper storage is not a regular directory.') + const path = pathFor(directory, artifact) + const stat = await lstat(path, { bigint: true }) + if (!stat.isFile() || stat.isSymbolicLink()) + throw new LibraryHelperError('The managed helper is not a regular file.') + if (stat.size !== BigInt(artifact.size) || (stat.mode & 0o100n) === 0n) + throw new LibraryHelperError('The managed helper is incomplete or not executable.') + const stamp = `${artifact.sha256}:${stat.ino}:${stat.size}:${stat.mtimeNs}:${stat.ctimeNs}` + if (verified.get(path) !== stamp) { + if (digest(await readFile(path)) !== artifact.sha256) + throw new LibraryHelperError('The managed helper checksum does not match this build.') + verified.set(path, stamp) + } + return { ...base, state: 'ready' } + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return { ...base, state: 'missing' } + return { ...base, state: 'damaged', reason: message(error) } + } +} + +export async function managedHelperPath(directory: string): Promise { + const status = await inspectManagedHelper(directory) + if (status.state === 'missing') return null + if (status.state !== 'ready') + throw new LibraryHelperError( + status.reason ?? 'Repair the video helper in Library → Coverage & gaps.' + ) + return pathFor(directory, MAC_VIDEO_HELPER) +} + +export async function verifyHelperVersion( + path: string, + version = TESTED_EXTRACTOR_VERSION, + signal?: AbortSignal +): Promise { + const result = await exec(path, ['--ignore-config', '--no-plugin-dirs', '--version'], { + timeout: 15_000, + maxBuffer: 65536, + signal + }) + if (!result.stdout.trim()) + throw new LibraryHelperProbeError( + 'The video helper returned no version output. Try again shortly.' + ) + if (result.stdout.trim() !== version) + throw new LibraryHelperError(`Expected yt-dlp ${version}; found ${result.stdout.trim()}.`) +} + +/** Only the native caller chooses the artifact and download transport; neither comes from IPC. */ +export async function installManagedHelper(options: { + directory: string + signal: AbortSignal + download: (url: string, signal: AbortSignal, limit: number) => Promise + artifact?: HelperArtifact + verifyVersion?: (path: string, version: string, signal: AbortSignal) => Promise +}): Promise { + const artifact = options.artifact ?? MAC_VIDEO_HELPER + const destination = pathFor(options.directory, artifact) + const existing = await inspectManagedHelper(options.directory, artifact) + if (existing.state === 'ready') return existing + options.signal.throwIfAborted() + await mkdir(options.directory, { recursive: true, mode: 0o700 }) + const root = await lstat(options.directory) + if (!root.isDirectory() || root.isSymbolicLink()) + throw new LibraryHelperError('Helper storage is not a regular directory.') + const temporary = join(options.directory, `.incomplete-${randomUUID()}`) + try { + const response = await options.download(artifact.url, options.signal, artifact.size) + if (!response.ok || !response.body) + throw new LibraryHelperError(`Helper download failed (HTTP ${response.status}).`) + const file = await open(temporary, 'wx', 0o600) + let size = 0 + const hash = createHash('sha256') + const reader = response.body.getReader() + try { + for (;;) { + const { value: chunk, done } = await reader.read() + if (done) break + options.signal.throwIfAborted() + size += chunk.byteLength + if (size > artifact.size) + throw new LibraryHelperError('Helper download exceeds the expected size.') + hash.update(chunk) + await file.writeFile(chunk) + } + if (size !== artifact.size || hash.digest('hex') !== artifact.sha256) + throw new LibraryHelperError('Helper download failed size or checksum verification.') + await file.chmod(0o500) + await file.sync() + } finally { + reader.releaseLock() + await file.close() + } + options.signal.throwIfAborted() + await (options.verifyVersion ?? verifyHelperVersion)( + temporary, + artifact.version, + options.signal + ) + options.signal.throwIfAborted() + await rename(temporary, destination) + const parent = await open(options.directory, 'r') + try { + await parent.sync() + } finally { + await parent.close() + } + verified.delete(destination) + const status = await inspectManagedHelper(options.directory, artifact) + if (status.state !== 'ready') + throw new LibraryHelperError(status.reason ?? 'Helper could not be verified.') + return status + } finally { + await rm(temporary, { force: true }) + } +} diff --git a/apps/electron/src/library/provider-error.ts b/apps/electron/src/library/provider-error.ts new file mode 100644 index 000000000..85fece84e --- /dev/null +++ b/apps/electron/src/library/provider-error.ts @@ -0,0 +1,14 @@ +import { TaggedError } from '@xnetjs/core' + +export class LibraryProviderError extends TaggedError { + readonly _tag = 'LibraryProviderError' + constructor( + message: string, + readonly disposition: 'retry' | 'blocked' | 'unavailable', + readonly retryAt?: number, + readonly scope: 'resource' | 'provider' = 'resource', + readonly host?: string + ) { + super(message) + } +} diff --git a/apps/electron/src/library/providers.test.ts b/apps/electron/src/library/providers.test.ts new file mode 100644 index 000000000..f143e3f5d --- /dev/null +++ b/apps/electron/src/library/providers.test.ts @@ -0,0 +1,82 @@ +import { expect, it } from 'vitest' +import { chooseTrack, extractorUrl, parseCaptionBody, parseExtractorMetadata } from './providers' +import { imageContentType } from './service' + +it('discovers real caption languages and prefers authored tracks within a language', () => { + const parsed = parseExtractorMetadata({ + title: 'Full title', + description: 'Full description', + subtitles: { de: [{ ext: 'vtt', url: 'https://example.com/de.vtt' }] }, + automatic_captions: { en: [{ ext: 'json3', url: 'https://example.com/en.json3' }] } + }) + expect(chooseTrack(parsed.tracks!, 'de')?.autoGenerated).toBe(false) + expect(chooseTrack(parsed.tracks!, 'en')?.language).toBe('en') + expect(chooseTrack(parsed.tracks!, 'fr')?.language).toBe('de') +}) +it('retains full description text and records missing fields explicitly', () => { + const description = 'a'.repeat(50000) + expect(parseExtractorMetadata({ title: 'Title', description }).description).toBe(description) + expect(parseExtractorMetadata({ title: 'Title' }).fields.description.state).toBe('unavailable') +}) +it('rejects malformed caption bodies and timings instead of declaring no captions', () => { + expect(() => parseCaptionBody('Sign in', 'json3')).toThrow() + expect(() => parseCaptionBody('{}', 'json3')).toThrow() + expect(() => parseCaptionBody('Sign in', 'vtt')).toThrow() + expect(() => parseCaptionBody('WEBVTT\n\n00:01 --> bad\nPartial words', 'vtt')).toThrow( + 'not truncated' + ) + expect(() => + parseCaptionBody( + JSON.stringify({ events: [{ tStartMs: -1, segs: [{ utf8: 'words' }] }] }), + 'json3' + ) + ).toThrow() +}) +it('preserves JSON3 word fragments and VTT timing', () => { + expect( + parseCaptionBody( + JSON.stringify({ + events: [ + { tStartMs: 2500, dDurationMs: 500, segs: [{ utf8: 'hel' }, { utf8: 'lo world' }] } + ] + }), + 'json3' + ) + ).toEqual([{ startMs: 2500, durationMs: 500, text: 'hello world' }]) + expect( + parseCaptionBody('WEBVTT\n\n00:01.000 --> 00:02.500\nHello & welcome\n', 'vtt') + ).toEqual([{ startMs: 1000, durationMs: 1500, text: 'Hello & welcome' }]) +}) +it('rejects local paths and provider identifiers that could change helper arguments', () => { + expect(() => + extractorUrl({ platform: 'youtube', platformContentId: '--exec bad' } as never) + ).toThrow() + expect(() => + extractorUrl({ platform: 'generic', url: 'file:///private/example' } as never) + ).toThrow() +}) +it('resolves Instagram export fbids through the saved post URL without changing their identity', () => { + for (const path of ['p/abc123', 'reel/abc123', 'creator/reel/abc123', 'tv/abc123']) { + expect( + extractorUrl({ + platform: 'instagram', + platformContentId: '17866765571880000', + url: `https://www.instagram.com/${path}/?utm_source=export` + } as never) + ).toBe('https://www.instagram.com/p/abc123/') + } + for (const url of [ + 'https://evil.example/p/abc123/', + 'https://www.instagram.com/reels/audio/123/', + 'https://www.instagram.com/creator/' + ]) { + expect(() => + extractorUrl({ platform: 'instagram', platformContentId: '17866765571880000', url } as never) + ).toThrow() + } +}) +it('does not accept HTML or SVG as a cached thumbnail', () => { + expect(imageContentType(Buffer.from('Login'))).toBeNull() + expect(imageContentType(Buffer.from(''))).toBeNull() + expect(imageContentType(Buffer.from([137, 80, 78, 71, 13, 10, 26, 10, 0, 0]))).toBe('image/png') +}) diff --git a/apps/electron/src/library/providers.ts b/apps/electron/src/library/providers.ts new file mode 100644 index 000000000..8748189b2 --- /dev/null +++ b/apps/electron/src/library/providers.ts @@ -0,0 +1,693 @@ +import type { CaptionTrack, Cue, LibraryMetadata, LibraryResource } from './types' +import { execFile } from 'node:child_process' +import { lookup } from 'node:dns/promises' +import { access } from 'node:fs/promises' +import { request as httpRequest } from 'node:http' +import { request as httpsRequest } from 'node:https' +import { homedir } from 'node:os' +import { join } from 'node:path' +import { promisify } from 'node:util' +import { assertPublicUrl } from '@xnetjs/core' +import { parseInstagramPage } from './instagram' +import { + LibraryHelperProbeError, + managedHelperPath, + TESTED_EXTRACTOR_VERSION, + verifyHelperVersion +} from './managed-helper' +import { LibraryProviderError } from './provider-error' +import { parsePostPreview, parsePublicPage, parseTikTokPage } from './public-pages' +import { providerResource } from './source' + +const exec = promisify(execFile) +export { TESTED_EXTRACTOR_VERSION } from './managed-helper' +export { LibraryProviderError } from './provider-error' +const record = (value: unknown): value is Record => + value !== null && typeof value === 'object' && !Array.isArray(value) +const text = (value: unknown): string | undefined => + typeof value === 'string' && value.trim() ? value.trim() : undefined +type PublicFetchOptions = { + signal?: AbortSignal + limit?: number + headers?: Record + timeoutMs?: number +} + +function untilAborted(promise: Promise, signal: AbortSignal): Promise { + return new Promise((resolve, reject) => { + const abort = () => reject(signal.reason) + signal.addEventListener('abort', abort, { once: true }) + if (signal.aborted) abort() + promise.then( + (value) => { + signal.removeEventListener('abort', abort) + resolve(value) + }, + (error: unknown) => { + signal.removeEventListener('abort', abort) + reject(error) + } + ) + }) +} + +function requestAtAddress( + parsed: URL, + address: { address: string; family: number }, + options: PublicFetchOptions, + signal: AbortSignal +): Promise { + return new Promise((resolve, reject) => { + const request = (parsed.protocol === 'https:' ? httpsRequest : httpRequest)( + parsed, + { + signal, + agent: false, + family: address.family, + // Pin a validated address while preserving the original Host and TLS name. + lookup: (_host, _options, callback) => callback(null, address.address, address.family), + headers: { + 'User-Agent': 'xNet-Personal-Library/1', + ...options.headers, + 'Accept-Encoding': 'identity' + } + }, + (incoming) => { + clearTimeout(connectTimer) + try { + const headers = new Headers() + for (let i = 0; i < incoming.rawHeaders.length; i += 2) + headers.append(incoming.rawHeaders[i], incoming.rawHeaders[i + 1]) + const status = incoming.statusCode ?? 502 + if (status < 200 || status >= 300) { + incoming.destroy() + resolve(new Response(null, { status, headers })) + return + } + if (headers.has('content-encoding') && headers.get('content-encoding') !== 'identity') { + incoming.destroy() + reject( + new LibraryProviderError('Source ignored the uncompressed response request.', 'retry') + ) + return + } + const chunks: Buffer[] = [] + let length = 0 + incoming.on('data', (chunk: Buffer) => { + length += chunk.length + if (length > (options.limit ?? 8 * 1024 * 1024)) { + reject( + new LibraryProviderError( + 'Response exceeds the byte limit; it was not truncated.', + 'blocked' + ) + ) + incoming.destroy() + return + } + chunks.push(chunk) + }) + incoming.on('error', reject) + incoming.on('aborted', () => + reject( + new LibraryProviderError('Source closed before the response completed.', 'retry') + ) + ) + incoming.on('end', () => { + try { + resolve( + new Response([204, 205].includes(status) ? null : Buffer.concat(chunks), { + status, + headers + }) + ) + } catch (error) { + reject(error) + } + }) + } catch (error) { + incoming.destroy() + reject(error) + } + } + ) + // A single unreachable CDN address must not consume the entire request deadline. + const connectTimer = setTimeout( + () => + request.destroy( + new LibraryProviderError( + 'Source did not respond in time. Check your connection or firewall settings, then retry.', + 'retry' + ) + ), + 8_000 + ) + request.once('close', () => clearTimeout(connectTimer)) + request.once('error', (error) => { + clearTimeout(connectTimer) + reject(error) + }) + request.end() + }) +} + +/** Validate all DNS answers and every redirect, bound the whole request, and send no cookies. */ +export async function fetchPublic( + url: string, + options: PublicFetchOptions = {} +): Promise { + const controller = new AbortController() + const abort = () => controller.abort(options.signal?.reason) + options.signal?.addEventListener('abort', abort, { once: true }) + if (options.signal?.aborted) abort() + // An owned timer stays live for the entire operation, including redirected requests. + const deadline = setTimeout( + () => + controller.abort( + new Error( + 'Source request timed out. Check your connection or firewall settings, then retry.' + ) + ), + options.timeoutMs ?? 30_000 + ) + const signal = controller.signal + try { + for (let redirects = 0; redirects <= 5; redirects++) { + signal.throwIfAborted() + assertPublicUrl(url) + const parsed = new URL(url) + if (parsed.username || parsed.password) + throw new LibraryProviderError('URLs with embedded credentials are not fetched.', 'blocked') + const addresses = await untilAborted( + lookup(parsed.hostname.replace(/^\[|\]$/g, ''), { all: true }), + signal + ) + signal.throwIfAborted() + for (const { address, family } of addresses) + assertPublicUrl(`https://${family === 6 ? `[${address}]` : address}/`) + const ordered = [ + ...addresses.filter((entry) => entry.family === 4), + ...addresses.filter((entry) => entry.family !== 4) + ] + let response: Response | undefined + let lastError: unknown = new LibraryProviderError('Source hostname has no address.', 'retry') + for (const address of ordered.slice(0, 8)) { + signal.throwIfAborted() + try { + response = await requestAtAddress(parsed, address, options, signal) + break + } catch (error) { + signal.throwIfAborted() + if (error instanceof LibraryProviderError && error.disposition === 'blocked') throw error + lastError = error + } + } + if (!response) throw lastError + if ([301, 302, 303, 307, 308].includes(response.status)) { + const location = response.headers.get('location') + await response.body?.cancel() + if (!location) throw new LibraryProviderError('Redirect has no destination.', 'retry') + url = new URL(location, url).href + continue + } + if ([401, 403, 429].includes(response.status)) { + const retry = response.headers.get('retry-after') + const delay = + retry && /^\d+$/.test(retry) + ? Number(retry) * 1000 + : retry + ? Date.parse(retry) - Date.now() + : 0 + await response.body?.cancel() + throw new LibraryProviderError( + `Provider refused this request (HTTP ${response.status}).`, + response.status === 429 ? 'retry' : 'blocked', + Date.now() + Math.max(60_000, Number.isFinite(delay) ? delay : 0), + response.status === 429 ? 'provider' : 'resource', + parsed.hostname.toLowerCase() + ) + } + if (response.status === 404 || response.status === 410) { + await response.body?.cancel() + throw new LibraryProviderError('Source is unavailable or no longer public.', 'unavailable') + } + if (!response.ok) { + await response.body?.cancel() + throw new LibraryProviderError(`Provider returned HTTP ${response.status}.`, 'retry') + } + return response + } + throw new LibraryProviderError('Too many source redirects.', 'blocked') + } finally { + clearTimeout(deadline) + options.signal?.removeEventListener('abort', abort) + } +} + +export async function findExtractor(directory?: string): Promise { + if (directory && process.platform === 'darwin') { + try { + const managed = await managedHelperPath(directory) + if (managed) return managed + } catch (error) { + throw new LibraryProviderError( + `${error instanceof Error ? error.message : String(error)} Repair the helper in Library → Coverage & gaps.`, + 'blocked' + ) + } + } + const failures: string[] = [] + let probeFailure: LibraryHelperProbeError | undefined + for (const path of [ + join(homedir(), '.local/bin/yt-dlp'), + '/opt/homebrew/bin/yt-dlp', + '/usr/local/bin/yt-dlp' + ]) { + try { + await access(path) + } catch { + continue + } + try { + await verifyHelperVersion(path) + return path + } catch (error) { + if (error instanceof LibraryHelperProbeError) probeFailure = error + failures.push(error instanceof Error ? error.message : String(error)) + } + } + if (probeFailure) throw new LibraryProviderError(probeFailure.message, 'retry') + throw new LibraryProviderError( + `Video extraction needs yt-dlp ${TESTED_EXTRACTOR_VERSION}. Open Library → Coverage & gaps to install the managed Mac helper.${failures.length ? ` ${failures.join(' ')}` : ''}`, + 'blocked' + ) +} + +export function extractorUrl(resource: LibraryResource): string { + // Only named providers reach the helper: a captured URL cannot select a local file or arbitrary extractor. + if (resource.platform === 'youtube' && /^[A-Za-z0-9_-]{11}$/.test(resource.platformContentId)) + return `https://www.youtube.com/watch?v=${resource.platformContentId}` + if (resource.platform === 'instagram') { + // Meta exports may identify a saved record by its numeric fbid. The saved + // URL carries the post's actual shortcode; the two are not interchangeable. + const url = new URL(resource.url) + const shortcode = url.pathname.match( + /^\/(?:[^/]+\/)?(?:p|reels?|tv)\/([A-Za-z0-9_-]+)\/?$/ + )?.[1] + if (['instagram.com', 'www.instagram.com'].includes(url.hostname) && shortcode) + return `https://www.instagram.com/p/${shortcode}/` + } + if ( + (resource.platform === 'x' || resource.platform === 'twitter') && + /^\d+$/.test(resource.platformContentId) + ) + return `https://x.com/i/status/${resource.platformContentId}` + throw new LibraryProviderError( + 'This resource has no supported native video identifier.', + 'unavailable' + ) +} + +export function parseExtractorMetadata( + value: unknown, + provider = `yt-dlp/${TESTED_EXTRACTOR_VERSION}` +): LibraryMetadata { + if (!record(value)) + throw new LibraryProviderError('Extractor returned an invalid metadata object.', 'retry') + const tracks: CaptionTrack[] = [] + for (const [field, autoGenerated] of [ + ['subtitles', false], + ['automatic_captions', true] + ] as const) { + const source = value[field] + if (source === undefined) continue + if (!record(source)) + throw new LibraryProviderError('Extractor caption inventory is malformed.', 'retry') + for (const [language, items] of Object.entries(source)) { + if (language === 'live_chat') continue + if (!Array.isArray(items)) + throw new LibraryProviderError('Extractor caption tracks are malformed.', 'retry') + for (const item of items) { + if (!record(item)) continue + if ((item.ext === 'json3' || item.ext === 'vtt') && typeof item.url === 'string') + tracks.push({ url: item.url, language, format: item.ext, autoGenerated }) + } + } + } + const title = text(value.title) + const description = text(value.description) + const thumbnailUrl = text(value.thumbnail) + if (!title && !description && !thumbnailUrl) + throw new LibraryProviderError('Extractor returned no usable metadata.', 'retry') + return { + title, + description, + thumbnailUrl, + author: text(value.channel) ?? text(value.uploader), + language: text(value.language), + ...(typeof value.duration === 'number' && Number.isFinite(value.duration) + ? { durationSeconds: value.duration } + : {}), + tracks, + fields: { + title: title + ? { state: 'complete' } + : { state: 'unavailable', reason: 'No source title returned.' }, + description: description + ? { state: 'complete' } + : { state: 'unavailable', reason: 'No written description returned.' }, + captions: + value.subtitles !== undefined || value.automatic_captions !== undefined + ? { state: 'complete' } + : { state: 'unavailable', reason: 'Extractor did not report caption availability.' } + }, + provider, + fetchedAt: Date.now(), + evidence: value + } +} + +/** Parse source JSON as data; never evaluate scripts from a fetched page. */ +export function parseYouTubePage(html: string, videoId: string): LibraryMetadata { + const match = + /(?:var\s+ytInitialPlayerResponse\s*=|(?:window\[)?["']ytInitialPlayerResponse["']\]?\s*[:=])\s*\{/.exec( + html + ) + if (!match) throw new LibraryProviderError('YouTube did not return video metadata.', 'retry') + const start = match.index + match[0].lastIndexOf('{') + let depth = 0, + quoted = false, + escaped = false, + end = start + for (; end < html.length; end++) { + const char = html[end] + if (quoted) { + if (escaped) escaped = false + else if (char === '\\') escaped = true + else if (char === '"') quoted = false + } else if (char === '"') quoted = true + else if (char === '{') depth++ + else if (char === '}' && --depth === 0) break + } + const value: unknown = JSON.parse(html.slice(start, end + 1)) + if (!record(value)) throw new LibraryProviderError('YouTube metadata is malformed.', 'retry') + const status = record(value.playabilityStatus) ? value.playabilityStatus : {} + const details = record(value.videoDetails) ? value.videoDetails : {} + if (details.videoId !== videoId || !text(details.title)) { + const reason = text(status.reason) ?? 'YouTube did not identify the requested video.' + throw new LibraryProviderError( + reason, + /private|removed|unavailable|deleted/i.test(reason) ? 'unavailable' : 'blocked' + ) + } + const images = + record(details.thumbnail) && Array.isArray(details.thumbnail.thumbnails) + ? details.thumbnail.thumbnails.filter(record).filter((image) => text(image.url)) + : [] + const thumbnail = images.sort((a, b) => Number(b.width ?? 0) - Number(a.width ?? 0))[0] + const captions = + record(value.captions) && record(value.captions.playerCaptionsTracklistRenderer) + ? value.captions.playerCaptionsTracklistRenderer.captionTracks + : undefined + if ( + captions !== undefined && + (!Array.isArray(captions) || + captions.some((track) => !record(track) || !text(track.baseUrl) || !text(track.languageCode))) + ) + throw new LibraryProviderError('YouTube caption inventory is malformed.', 'retry') + const tracks: CaptionTrack[] = Array.isArray(captions) + ? captions.filter(record).map((track) => { + const url = new URL(text(track.baseUrl)!) + url.searchParams.set('fmt', 'json3') + return { + url: url.href, + language: text(track.languageCode)!, + format: 'json3', + autoGenerated: track.kind === 'asr' + } + }) + : [] + return { + title: text(details.title), + description: + typeof details.shortDescription === 'string' ? details.shortDescription : undefined, + author: text(details.author), + thumbnailUrl: thumbnail ? text(thumbnail.url) : undefined, + ...(Number.isFinite(Number(details.lengthSeconds)) + ? { durationSeconds: Number(details.lengthSeconds) } + : {}), + tracks, + fields: { + title: { state: 'complete' }, + description: + typeof details.shortDescription === 'string' + ? { + state: 'complete', + ...(details.shortDescription ? {} : { reason: 'The author supplied no description.' }) + } + : { state: 'unavailable', reason: 'YouTube returned no description.' }, + captions: + status.status === 'OK' + ? { state: 'complete' } + : { state: 'partial', reason: text(status.reason) ?? 'Video playback is restricted.' } + }, + provider: 'youtube-page/1', + fetchedAt: Date.now(), + evidence: { videoDetails: details, playabilityStatus: status, captions: value.captions } + } +} + +async function fetchYouTubeMetadata( + resource: LibraryResource, + signal: AbortSignal +): Promise { + const url = extractorUrl(resource) + let pageError: unknown + try { + const response = await fetchPublic(url, { signal, limit: 8 * 1024 * 1024, timeoutMs: 20_000 }) + return parseYouTubePage(await response.text(), resource.platformContentId) + } catch (error) { + if (signal.aborted || (error instanceof LibraryProviderError && error.scope === 'provider')) + throw error + pageError = error + } + // Public oEmbed still supplies useful cards if the full page is unavailable. + const response = await fetchPublic( + `https://www.youtube.com/oembed?url=${encodeURIComponent(url)}&format=json`, + { signal, limit: 128 * 1024, timeoutMs: 15_000 } + ) + const value: unknown = await response.json() + if (!record(value) || !text(value.title)) + throw new LibraryProviderError('YouTube preview has no title.', 'retry') + const reason = + pageError instanceof Error ? pageError.message : 'Full video metadata is unavailable.' + return { + title: text(value.title), + author: text(value.author_name), + thumbnailUrl: text(value.thumbnail_url), + fields: { + title: { state: 'complete' }, + description: { state: 'partial', reason }, + captions: { state: 'partial', reason } + }, + provider: 'youtube-oembed/1', + fetchedAt: Date.now(), + evidence: value + } +} + +export async function fetchLibraryMetadata( + resource: LibraryResource, + signal: AbortSignal, + helperDirectory?: string +): Promise { + return fetchProviderMetadata(providerResource(resource), signal, helperDirectory) +} + +async function fetchProviderMetadata( + resource: LibraryResource, + signal: AbortSignal, + helperDirectory?: string +): Promise { + if (resource.platform === 'youtube') return fetchYouTubeMetadata(resource, signal) + if (resource.platform === 'instagram') { + const url = extractorUrl(resource) + const shortcode = new URL(url).pathname.split('/')[2] + try { + const response = await fetchPublic(`${url}embed/captioned/`, { + signal, + limit: 4 * 1024 * 1024, + timeoutMs: 20_000 + }) + return parseInstagramPage(await response.text(), shortcode) + } catch (error) { + if (signal.aborted || (error instanceof LibraryProviderError && error.scope === 'provider')) + throw error + } + const response = await fetchPublic(url, { signal, limit: 4 * 1024 * 1024, timeoutMs: 20_000 }) + return parseInstagramPage(await response.text(), shortcode) + } + if (['twitter', 'x'].includes(resource.platform)) + return fetchExtractorMetadata(resource, signal, helperDirectory) + if (resource.platform === 'tiktok') { + try { + const page = await fetchPublic(resource.url, { + signal, + limit: 8 * 1024 * 1024, + timeoutMs: 20_000 + }) + return parseTikTokPage(await page.text(), resource.platformContentId) + } catch (error) { + if (signal.aborted || (error instanceof LibraryProviderError && error.scope === 'provider')) + throw error + } + const response = await fetchPublic( + `https://www.tiktok.com/oembed?url=${encodeURIComponent(resource.url)}`, + { signal, limit: 256 * 1024 } + ) + return parsePostPreview(await response.json(), 'tiktok') + } + if (resource.platform === 'reddit') { + const response = await fetchPublic( + `https://www.reddit.com/oembed?url=${encodeURIComponent(resource.url)}`, + { signal, limit: 256 * 1024 } + ) + return parsePostPreview(await response.json(), 'reddit') + } + const response = await fetchPublic(resource.url, { signal, limit: 8 * 1024 * 1024 }) + return parsePublicPage(await response.text(), resource.url, resource.platform === 'github') +} + +export async function fetchExtractorMetadata( + resource: LibraryResource, + signal: AbortSignal, + helperDirectory?: string +): Promise { + const helper = await findExtractor(helperDirectory) + try { + const result = await exec( + helper, + [ + '--ignore-config', + '--no-plugin-dirs', + '--no-cache-dir', + '--no-playlist', + '--skip-download', + '--ignore-no-formats-error', + ...(resource.platform === 'youtube' + ? ['--extractor-args', 'youtube:skip=translated_subs'] + : []), + '--dump-single-json', + '--no-warnings', + '--no-progress', + '--socket-timeout', + '15', + '--retries', + '0', + '--extractor-retries', + '0', + '--', + extractorUrl(resource) + ], + { + signal, + timeout: 90_000, + maxBuffer: 16 * 1024 * 1024, + env: { ...process.env, PYTHONUNBUFFERED: '1' } + } + ) + const metadata = parseExtractorMetadata(JSON.parse(result.stdout)) + if (['x', 'twitter'].includes(resource.platform)) + metadata.fields.title = { + state: 'partial', + reason: 'Extractor display label; this platform may not provide a separate authored title.' + } + return metadata + } catch (error) { + if (signal.aborted) throw error + if (error instanceof LibraryProviderError) throw error + const message = error instanceof Error ? error.message : String(error) + const blocked = /sign in|login|log in|cookies|private|403|429|confirm.*bot/i.test(message) + throw new LibraryProviderError(message.slice(0, 1800), blocked ? 'blocked' : 'retry') + } +} + +export function chooseTrack( + tracks: readonly CaptionTrack[], + preferred = 'en' +): CaptionTrack | null { + const score = (track: CaptionTrack) => + (track.language === preferred + ? 0 + : track.language.split('-')[0] === preferred.split('-')[0] + ? 10 + : 20) + + (track.autoGenerated ? 2 : 0) + + (track.format === 'json3' ? 0 : 1) + return ( + [...tracks].sort((a, b) => score(a) - score(b) || a.language.localeCompare(b.language))[0] ?? + null + ) +} + +export function parseCaptionBody(body: string, format: 'json3' | 'vtt'): Cue[] { + if (format === 'json3') { + const value: unknown = JSON.parse(body) + if (!record(value) || !Array.isArray(value.events)) + throw new LibraryProviderError('Caption response is not a JSON3 track.', 'retry') + return value.events.flatMap((event): Cue[] => { + if (!record(event) || !Array.isArray(event.segs)) return [] + const parts = event.segs + .map((seg) => (record(seg) && typeof seg.utf8 === 'string' ? seg.utf8 : '')) + .filter(Boolean) + if (!parts.length) return [] + if ( + typeof event.tStartMs !== 'number' || + !Number.isFinite(event.tStartMs) || + event.tStartMs < 0 || + (event.dDurationMs !== undefined && + (typeof event.dDurationMs !== 'number' || + !Number.isFinite(event.dDurationMs) || + event.dDurationMs < 0)) + ) + throw new LibraryProviderError('Caption timing is malformed.', 'retry') + return [ + { + text: parts.join('').replace(/\s+/g, ' ').trim(), + startMs: event.tStartMs, + durationMs: typeof event.dDurationMs === 'number' ? event.dDurationMs : 0 + } + ] + }) + } + if (!body.trimStart().startsWith('WEBVTT')) + throw new LibraryProviderError('Caption response is not a WebVTT track.', 'retry') + const time = (raw: string) => { + const parts = raw.split(':').map(Number) + if (parts.some((part) => !Number.isFinite(part))) + throw new LibraryProviderError('Caption timing is malformed.', 'retry') + return Math.round(parts.reduce((total, part) => total * 60 + part, 0) * 1000) + } + return body.split(/\r?\n\s*\r?\n/).flatMap((block): Cue[] => { + const match = /(?:^|\n)([\d:.]+) --> ([\d:.]+)[^\n]*\n([\s\S]*)/.exec(block) + if (!match) { + if (block.includes('-->') && !/^(NOTE|STYLE|REGION)(?:\s|$)/.test(block.trimStart())) + throw new LibraryProviderError( + 'Caption cue is malformed; the track was not truncated.', + 'retry' + ) + return [] + } + const startMs = time(match[1]) + const end = time(match[2]) + if (end < startMs) throw new LibraryProviderError('Caption ends before it starts.', 'retry') + const words = match[3] + .replace(/<[^>]*>/g, '') + .replaceAll('&', '&') + .replaceAll('<', '<') + .replaceAll('>', '>') + .replace(/\s+/g, ' ') + .trim() + return words ? [{ startMs, durationMs: end - startMs, text: words }] : [] + }) +} diff --git a/apps/electron/src/library/public-http.test.ts b/apps/electron/src/library/public-http.test.ts new file mode 100644 index 000000000..489a45dc3 --- /dev/null +++ b/apps/electron/src/library/public-http.test.ts @@ -0,0 +1,249 @@ +import type { RequestOptions } from 'node:https' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { fetchLibraryMetadata, fetchPublic } from './providers' + +const stubs = vi.hoisted(() => ({ lookup: vi.fn(), request: vi.fn() })) +vi.mock('node:dns/promises', () => ({ lookup: stubs.lookup })) +vi.mock('node:https', () => ({ request: stubs.request })) +vi.mock('node:http', () => ({ request: stubs.request })) + +type Incoming = PassThrough & { rawHeaders: string[]; statusCode: number } +const addresses = [ + { address: '93.184.216.34', family: 4 }, + { address: '93.184.216.35', family: 4 } +] +let responses: ((request: EventEmitter, callback: (incoming: Incoming) => void) => void)[] +beforeEach(() => { + stubs.lookup.mockReset().mockResolvedValue(addresses) + stubs.request + .mockReset() + .mockImplementation( + (_url: URL, options: RequestOptions, callback: (incoming: Incoming) => void) => { + const request = new EventEmitter() as EventEmitter & { + end(): void + destroy(error: Error): void + } + request.destroy = (error) => { + request.emit('error', error) + request.emit('close') + } + const abort = () => request.destroy(new Error('aborted')) + options.signal?.addEventListener('abort', abort, { once: true }) + request.once('close', () => options.signal?.removeEventListener('abort', abort)) + request.end = () => queueMicrotask(() => responses.shift()!(request, callback)) + return request + } + ) + responses = [] +}) +afterEach(() => vi.useRealTimers()) +const respond = + (status: number, body: string, headers: string[] = []) => + (request: EventEmitter, callback: (incoming: Incoming) => void) => { + const stream = Object.assign(new PassThrough(), { statusCode: status, rawHeaders: headers }) + stream.once('close', () => request.emit('close')) + callback(stream) + if (!stream.destroyed) stream.end(body) + } + +const instagramResource = { + id: 'instagram-post', + platform: 'instagram', + platformContentId: '17866765571880000', + url: 'https://www.instagram.com/p/abc123/', + title: 'Imported post', + sourceText: '', + actor: '', + privacy: 'private', + addedAt: 0 +} + +it('fetches Instagram written captions and posters without requiring the video helper', async () => { + responses = [ + respond( + 200, + `creator
A written caption
` + ) + ] + const result = await fetchLibraryMetadata( + instagramResource, + new AbortController().signal, + '/missing-helper' + ) + expect(result.description).toBe('A written caption') + expect(result.thumbnailUrl).toBe('https://images.example/post.jpg') + expect(result.fields.captions.state).toBe('unavailable') + expect(stubs.request.mock.calls[0][0].pathname).toBe('/p/abc123/embed/captioned/') + expect(stubs.request).toHaveBeenCalledOnce() +}) + +it('keeps Instagram public page previews partial when its embed is unavailable', async () => { + responses = [ + respond(200, 'Log in'), + respond( + 200, + `` + ) + ] + const result = await fetchLibraryMetadata(instagramResource, new AbortController().signal) + expect(result.provider).toBe('instagram-page/1') + expect(result.fields.description.state).toBe('partial') + expect(stubs.request).toHaveBeenCalledTimes(2) +}) + +it('stops at an Instagram rate limit and reports login pages without saving them as post metadata', async () => { + responses = [respond(429, '', ['retry-after', '120'])] + await expect( + fetchLibraryMetadata(instagramResource, new AbortController().signal) + ).rejects.toMatchObject({ scope: 'provider', disposition: 'retry' }) + expect(stubs.request).toHaveBeenCalledOnce() + responses = [respond(200, 'Log in'), respond(200, 'InstagramLog in')] + await expect( + fetchLibraryMetadata(instagramResource, new AbortController().signal) + ).rejects.toThrow('sign-in') +}) + +it('falls back to another validated address when the first connection fails', async () => { + responses = [ + (request) => { + request.emit('error', new Error('unreachable')) + request.emit('close') + }, + respond(200, 'complete content') + ] + expect(await (await fetchPublic('https://example.com/page')).text()).toBe('complete content') + const selected: string[] = [] + for (const call of stubs.request.mock.calls) { + const options = call[1] as RequestOptions + options.lookup!('example.com', {}, (_error, address) => { + if (typeof address !== 'string') throw new Error('Expected one pinned address') + selected.push(address) + }) + } + expect(selected).toEqual(addresses.map((value) => value.address)) +}) + +it('rejects a private DNS answer before opening any connection', async () => { + stubs.lookup.mockResolvedValue([...addresses, { address: '127.0.0.1', family: 4 }]) + await expect(fetchPublic('https://example.com/page')).rejects.toThrow() + expect(stubs.request).not.toHaveBeenCalled() +}) + +it('leaves an unresponsive address before the whole request deadline', async () => { + vi.useFakeTimers() + responses = [() => {}, respond(200, 'next address')] + const result = fetchPublic('https://example.com/page', { timeoutMs: 20_000 }) + await vi.advanceTimersByTimeAsync(8_000) + expect(await (await result).text()).toBe('next address') + expect(stubs.request).toHaveBeenCalledTimes(2) + expect(vi.getTimerCount()).toBe(0) +}) + +it('validates the destination after a redirect', async () => { + responses = [respond(302, '', ['location', 'http://127.0.0.1/private'])] + await expect(fetchPublic('https://example.com/page')).rejects.toThrow() + expect(stubs.request).toHaveBeenCalledTimes(1) +}) + +it('fails a byte limit immediately without retrying other addresses', async () => { + responses = [respond(200, 'oversized')] + await expect(fetchPublic('https://example.com/page', { limit: 3 })).rejects.toMatchObject({ + disposition: 'blocked' + }) + expect(stubs.request).toHaveBeenCalledTimes(1) +}) + +it('bounds stalled DNS and never opens a late request', async () => { + vi.useFakeTimers() + let finish!: (value: typeof addresses) => void + stubs.lookup.mockImplementation( + () => + new Promise((resolve) => { + finish = resolve + }) + ) + const pending = expect( + fetchPublic('https://example.com/page', { timeoutMs: 100 }) + ).rejects.toThrow('timed out') + await vi.advanceTimersByTimeAsync(100) + await pending + finish(addresses) + await Promise.resolve() + expect(stubs.request).not.toHaveBeenCalled() +}) + +it('cancels an active request and clears its timers', async () => { + vi.useFakeTimers() + const controller = new AbortController() + responses = [() => {}] + const pending = expect( + fetchPublic('https://example.com/page', { signal: controller.signal }) + ).rejects.toThrow() + await vi.advanceTimersByTimeAsync(1) + controller.abort() + await pending + expect(vi.getTimerCount()).toBe(0) +}) + +it('distinguishes provider rate limits from one restricted resource', async () => { + responses = [respond(429, '', ['retry-after', '120']), respond(403, '')] + await expect(fetchPublic('https://example.com/rate')).rejects.toMatchObject({ + scope: 'provider', + disposition: 'retry', + retryAt: expect.any(Number), + host: 'example.com' + }) + await expect(fetchPublic('https://example.com/private')).rejects.toMatchObject({ + scope: 'resource', + disposition: 'blocked' + }) +}) + +it('keeps YouTube cards useful when the page cannot expose complete metadata', async () => { + responses = [ + respond(200, 'Consent needed'), + respond( + 200, + JSON.stringify({ + title: 'Public preview title', + author_name: 'Original author', + thumbnail_url: 'https://i.ytimg.com/poster.jpg' + }) + ) + ] + const result = await fetchLibraryMetadata( + { + id: 'fixture', + platform: 'youtube', + platformContentId: 'abcdefghijk', + url: 'https://www.youtube.com/watch?v=abcdefghijk', + title: 'Unresolved', + sourceText: '', + actor: '', + privacy: 'private', + addedAt: 1 + }, + new AbortController().signal, + '/nonexistent/helper' + ) + expect(result.title).toBe('Public preview title') + expect(result.thumbnailUrl).toBe('https://i.ytimg.com/poster.jpg') + expect(result.fields.description.state).toBe('partial') + expect(result.fields.captions.state).toBe('partial') + expect(stubs.request).toHaveBeenCalledTimes(2) +}) + +it('attributes throttling to the final response host after an image redirect', async () => { + responses = [ + respond(302, '', ['location', 'https://images.example/poster']), + respond(429, '', ['retry-after', '180']) + ] + await expect(fetchPublic('https://example.com/image')).rejects.toMatchObject({ + scope: 'provider', + disposition: 'retry', + host: 'images.example' + }) + expect(stubs.request).toHaveBeenCalledTimes(2) +}) diff --git a/apps/electron/src/library/public-pages.test.ts b/apps/electron/src/library/public-pages.test.ts new file mode 100644 index 000000000..83a47d752 --- /dev/null +++ b/apps/electron/src/library/public-pages.test.ts @@ -0,0 +1,170 @@ +import type { LibraryResource } from './types' +import { expect, it } from 'vitest' +import { parsePostPreview, parsePublicPage, parseTikTokPage } from './public-pages' +import { providerResource, queueProvider } from './source' + +const resource = (url: string): LibraryResource => ({ + id: 'citation', + url, + platform: 'openai', + platformContentId: url, + title: url, + sourceText: '', + actor: '', + privacy: 'private', + addedAt: 1 +}) +const page = (item: unknown) => + `` + +it('routes archived citations through the URL provider without changing their source identity', () => { + const input = resource('https://www.tiktokv.com/share/video/12345') + expect(providerResource(input)).toMatchObject({ + id: 'citation', + platform: 'tiktok', + platformContentId: '12345', + url: 'https://www.tiktok.com/@_/video/12345', + privacy: 'private' + }) + expect(input.platform).toBe('openai') + expect(providerResource(resource('https://youtu.be/abcdefghijk?t=5')).platform).toBe('youtube') + expect(providerResource(resource('https://github.com/owner/repo')).platform).toBe('github') + expect(providerResource(resource('https://notyoutube.com/watch?v=abcdefghijk')).platform).toBe( + 'openai' + ) +}) + +it('reads full TikTok captions, source posters and native spoken tracks without executing scripts', () => { + const result = parseTikTokPage( + page({ + id: '12345', + desc: 'A long written caption '.repeat(100), + author: { uniqueId: 'creator' }, + textLanguage: 'en', + video: { + originCover: 'https://images.example/post.jpg', + duration: 30, + claInfo: { + captionInfos: [ + { + url: 'https://captions.example/post.vtt', + languageCode: 'en', + captionFormat: 'webvtt', + isAutoGen: true + } + ] + }, + subtitleInfos: [ + { Url: 'https://captions.example/post.vtt', LanguageCodeName: 'eng-US', Format: 'webvtt' } + ] + } + }), + '12345' + ) + expect(result.description).toHaveLength(2300) + expect(result.tracks).toEqual([ + { url: 'https://captions.example/post.vtt', language: 'en', format: 'vtt', autoGenerated: true } + ]) + expect(result.thumbnailUrl).toContain('post.jpg') + expect(result.fields.description.state).toBe('complete') + expect(result.fields.title.state).toBe('partial') +}) + +it('keeps photo posts usable and refuses mismatched TikTok pages and malformed inventories', () => { + const photo = { + id: '12345', + desc: 'Photos', + imagePost: { images: [] }, + video: { cover: 'https://images.example/photo.jpg' } + } + expect(parseTikTokPage(page(photo), '12345').tracks).toEqual([]) + expect(() => parseTikTokPage(page(photo), '999')).toThrow('this post') + expect(() => parseTikTokPage('Log in', '12345')).toThrow('public post data') + expect(() => parseTikTokPage(page({ ...photo, video: { subtitleInfos: {} } }), '12345')).toThrow( + 'inventory' + ) +}) + +it('extracts an available web subtitle track and a rendered GitHub README', () => { + const result = parsePublicPage( + `Repository & notes

Readme

${'Late searchable text '.repeat(1000)}

`, + 'https://github.com/example/repo', + true + ) + expect(result.description?.length).toBeGreaterThan(19000) + expect(result.description).not.toContain('not content') + expect(result.fields.readme.state).toBe('complete') + expect(result.tracks?.[0]).toEqual({ + url: 'https://github.com/subs.vtt', + language: 'fr', + format: 'vtt', + autoGenerated: false + }) + expect(result.thumbnailUrl).toBe('https://github.com/poster.jpg') +}) + +it('labels Reddit previews as partial and never treats them as transcripts', () => { + const result = parsePostPreview({ title: 'A saved comment', author_name: 'creator' }, 'reddit') + expect(result.fields.description.state).toBe('partial') + expect(result.fields.captions.state).toBe('unavailable') + expect(() => parsePostPreview({ error: 'not found' }, 'reddit')).toThrow('no public post') +}) + +it('keeps unrelated web hosts independent of the archive that imported them', () => { + expect(queueProvider(resource('https://example.org/a'))).toBe('web:example.org') + expect(queueProvider(resource('https://www.example.org/b'))).toBe('web:example.org') + expect(queueProvider(resource('https://another.org/a'))).toBe('web:another.org') +}) + +it('reads GitHub embedded READMEs and repository facts without retaining viewer data', () => { + const data = { + payload: { + csrf_tokens: { secret: 'must-not-retain' }, + codeViewRepoRoute: { + overview: { + overviewFiles: [ + { preferredFileType: 'license', richText: '

License text

' }, + { + preferredFileType: 'readme', + path: 'README.md', + richText: '

Example

Searchable readme body

' + } + ] + } + }, + sidebarAbout: { + description: 'Repository purpose', + topics: [{ name: 'learning' }, { name: 'graphs' }], + website: 'https://example.org', + stargazerCount: 123, + forksCount: 4, + repo: { license: { spdxId: 'MIT' } }, + viewer: { secret: 'must-not-retain' } + } + } + } + const result = parsePublicPage( + `Example repository`, + 'https://github.com/example/repo', + true + ) + expect(result.description).toContain('Repository purpose') + expect(result.description).toContain('Searchable readme body') + expect(result.description).toContain('learning, graphs') + expect(result.fields.readme.state).toBe('complete') + expect(result.fields.description.state).toBe('complete') + expect(result.evidence).toMatchObject({ + repository: { stars: 123, forks: 4, license: 'MIT', readmePath: 'README.md' } + }) + expect(JSON.stringify(result)).not.toContain('must-not-retain') + expect(result.description).not.toContain('License text') +}) + +it('indexes visible article text without claiming hidden content was retrieved', () => { + const result = parsePublicPage( + 'Essay

Late searchable passage

', + 'https://example.org/essay' + ) + expect(result.description).toBe('Late searchable passage') + expect(result.fields.description.state).toBe('partial') +}) diff --git a/apps/electron/src/library/public-pages.ts b/apps/electron/src/library/public-pages.ts new file mode 100644 index 000000000..41ffcc1a3 --- /dev/null +++ b/apps/electron/src/library/public-pages.ts @@ -0,0 +1,304 @@ +import type { CaptionTrack, LibraryMetadata } from './types' +import type { DefaultTreeAdapterTypes } from 'parse5' +import { parse } from 'parse5' +import { LibraryProviderError } from './provider-error' + +type Element = DefaultTreeAdapterTypes.Element +type Node = DefaultTreeAdapterTypes.Node +const record = (value: unknown): value is Record => + value !== null && typeof value === 'object' && !Array.isArray(value) +const text = (value: unknown) => (typeof value === 'string' ? value : undefined) +const attr = (node: Element, key: string) => node.attrs.find((a) => a.name === key)?.value +function elements(html: string): Element[] { + const pending: Node[] = [parse(html)], + result: Element[] = [] + while (pending.length) { + const node = pending.pop()! + if ('tagName' in node) result.push(node) + if ('childNodes' in node) + for (let i = node.childNodes.length - 1; i >= 0; i--) pending.push(node.childNodes[i]) + } + return result +} +function content(root: Node): string { + const pending: Node[] = [root], + result: string[] = [] + while (pending.length) { + const node = pending.pop()! + if ('tagName' in node && ['script', 'style', 'template'].includes(node.tagName)) continue + if ('value' in node && node.nodeName === '#text') result.push(node.value) + if ('tagName' in node && ['p', 'br', 'div', 'li', 'h1', 'h2', 'h3'].includes(node.tagName)) + result.push('\n') + if ('childNodes' in node) + for (let i = node.childNodes.length - 1; i >= 0; i--) pending.push(node.childNodes[i]) + } + return result + .join('') + .replace(/[ \t]+/g, ' ') + .replace(/\n\s*\n/g, '\n\n') + .trim() +} +function scriptText(node: Element): string { + return node.childNodes + .map((child) => (child.nodeName === '#text' && 'value' in child ? child.value : '')) + .join('') +} + +/** Read only the public repository fields, never the page's tokens or viewer state. */ +function githubOverview(nodes: Element[]) { + const script = nodes.find( + (node) => node.tagName === 'script' && attr(node, 'data-target') === 'react-app.embeddedData' + ) + if (!script) return undefined + const value: unknown = JSON.parse(scriptText(script)) + const payload = record(value) && record(value.payload) ? value.payload : undefined + if (!payload) return undefined + const route = record(payload.codeViewRepoRoute) ? payload.codeViewRepoRoute : {} + const overview = record(route.overview) ? route.overview : {} + const files = Array.isArray(overview.overviewFiles) ? overview.overviewFiles : [] + const readme = files.find((file: unknown) => record(file) && file.preferredFileType === 'readme') + const about = record(payload.sidebarAbout) ? payload.sidebarAbout : {} + const repo = record(about.repo) ? about.repo : {} + const license = record(repo.license) ? repo.license : {} + return { + description: text(about.description), + website: text(about.website), + topics: Array.isArray(about.topics) + ? about.topics.flatMap((topic: unknown) => { + const name = + typeof topic === 'string' ? topic : record(topic) ? text(topic.name) : undefined + return name ? [name] : [] + }) + : [], + stars: typeof about.stargazerCount === 'number' ? about.stargazerCount : undefined, + forks: typeof about.forksCount === 'number' ? about.forksCount : undefined, + license: text(license.spdxId) ?? text(license.name), + readmePath: record(readme) ? text(readme.path) : undefined, + readmeHtml: record(readme) ? text(readme.richText) : undefined + } +} +const labelCoverage = { + state: 'partial' as const, + reason: 'Display label derived from the written post, not a separate authored title.' +} + +export function parseTikTokPage(html: string, videoId: string): LibraryMetadata { + const script = elements(html).find( + (node) => attr(node, 'id') === '__UNIVERSAL_DATA_FOR_REHYDRATION__' + ) + if (!script) throw new LibraryProviderError('TikTok did not return public post data.', 'blocked') + const data: unknown = JSON.parse(scriptText(script)) + const scope = record(data) && data.__DEFAULT_SCOPE__ + const detail = record(scope) && scope['webapp.video-detail'] + const info = record(detail) && detail.itemInfo + const item = record(info) && info.itemStruct + if (!record(item) || item.id !== videoId) + throw new LibraryProviderError( + 'TikTok did not return this post; it may be private or removed.', + 'blocked' + ) + const video = record(item.video) ? item.video : {} + const author = record(item.author) ? item.author : {} + const cla = record(video.claInfo) ? video.claInfo : {} + const tracks: CaptionTrack[] = [] + let unsupportedTracks = false + for (const [items, modern] of [ + [cla.captionInfos, true], + [video.subtitleInfos, false] + ] as const) { + if (items !== undefined && !Array.isArray(items)) + throw new LibraryProviderError('TikTok caption inventory is malformed.', 'retry') + for (const entry of Array.isArray(items) ? items : []) { + if (!record(entry)) + throw new LibraryProviderError('TikTok caption entry is malformed.', 'retry') + const url = text(modern ? entry.url : entry.Url) + const format = text(modern ? entry.captionFormat : entry.Format) + const language = text( + modern ? (entry.languageCode ?? entry.language) : entry.LanguageCodeName + ) + if (!url || !language || !format) + throw new LibraryProviderError('TikTok caption entry is incomplete.', 'retry') + if (!['webvtt', 'vtt'].includes(format)) { + unsupportedTracks = true + continue + } + if (tracks.some((track) => track.url === url)) continue + tracks.push({ + url, + language: language.replace(/^eng(?=-|$)/, 'en'), + format: 'vtt', + autoGenerated: modern ? entry.isAutoGen === true : entry.Source === 'ASR' + }) + } + } + const description = text(item.desc) + const thumbnailUrl = text(video.originCover) ?? text(video.cover) + if (!description && !thumbnailUrl) + throw new LibraryProviderError('TikTok returned no caption or poster.', 'blocked') + const imagePost = record(item.imagePost) + return { + title: + description?.split('\n')[0].slice(0, 180) || + `TikTok post by ${text(author.uniqueId) ?? videoId}`, + description, + thumbnailUrl, + author: text(author.uniqueId) ?? text(author.nickname), + ...(typeof video.duration === 'number' ? { durationSeconds: video.duration } : {}), + language: text(item.textLanguage), + tracks, + fields: { + title: labelCoverage, + description: + description !== undefined + ? { state: 'complete' } + : { state: 'unavailable', reason: 'TikTok returned no written caption.' }, + captions: unsupportedTracks + ? { state: 'partial', reason: 'Some TikTok subtitle formats are not supported.' } + : tracks.length || + imagePost || + Array.isArray(video.subtitleInfos) || + Array.isArray(cla.captionInfos) + ? { state: 'complete' } + : { state: 'unavailable', reason: 'TikTok did not report spoken caption availability.' } + }, + provider: 'tiktok-page/1', + fetchedAt: Date.now(), + evidence: { + id: item.id, + desc: description, + author: text(author.uniqueId), + createTime: item.createTime, + thumbnailUrl, + tracks, + imagePost + } + } +} + +export function parsePublicPage(html: string, url: string, github = false): LibraryMetadata { + const nodes = elements(html) + const repository = github ? githubOverview(nodes) : undefined + const meta = (key: string) => { + const node = nodes.find( + (n) => n.tagName === 'meta' && (attr(n, 'property') === key || attr(n, 'name') === key) + ) + return node ? attr(node, 'content') : undefined + } + const titleNode = nodes.find((n) => n.tagName === 'title') + const title = meta('og:title') ?? (titleNode ? content(titleNode) : undefined) + const preview = repository?.description ?? meta('og:description') ?? meta('description') + const readme = github + ? nodes.find( + (n) => + n.tagName === 'article' && (attr(n, 'class') ?? '').split(/\s+/).includes('markdown-body') + ) + : undefined + const readmeText = repository?.readmeHtml + ? content(parse(repository.readmeHtml)) + : readme + ? content(readme) + : undefined + const article = !github ? nodes.find((node) => node.tagName === 'article') : undefined + const articleText = article ? content(article) : undefined + const topics = repository?.topics.length ? `Topics: ${repository.topics.join(', ')}` : undefined + const description = + [preview, readmeText ?? articleText, topics].filter(Boolean).join('\n\n') || undefined + const image = meta('og:image') ?? meta('twitter:image') + const tracks = nodes + .filter( + (n) => n.tagName === 'track' && ['captions', 'subtitles'].includes(attr(n, 'kind') ?? '') + ) + .flatMap((node): CaptionTrack[] => { + const src = attr(node, 'src') + if (!src) return [] + return [ + { + url: new URL(src, url).href, + format: 'vtt', + language: attr(node, 'srclang') ?? 'und', + autoGenerated: false + } + ] + }) + if (!title && !description && !image) + throw new LibraryProviderError('Page returned no usable metadata.', 'unavailable') + return { + title, + description, + thumbnailUrl: image ? new URL(image, url).href : undefined, + author: meta('author') ?? (github ? new URL(url).pathname.split('/')[1] : undefined), + tracks, + fields: { + title: title ? { state: 'complete' } : { state: 'unavailable', reason: 'Page has no title.' }, + description: description + ? github && readmeText + ? { state: 'complete' } + : { + state: 'partial', + reason: articleText + ? 'Visible article text; the source may omit content from its public page.' + : github + ? 'Repository description; no rendered README was returned.' + : 'Page preview; full article text has not been extracted.' + } + : { state: 'unavailable', reason: 'Page has no description.' }, + ...(github + ? { + readme: readmeText + ? { state: 'complete' as const } + : { state: 'unavailable' as const, reason: 'No rendered README found.' } + } + : {}), + ...(tracks.length ? { captions: { state: 'complete' as const } } : {}) + }, + provider: github ? 'github-page/2' : 'public-page/2', + fetchedAt: Date.now(), + evidence: { + url, + title, + preview, + readme: readmeText, + article: articleText, + image, + tracks, + ...(repository + ? { + repository: { + website: repository.website, + topics: repository.topics, + stars: repository.stars, + forks: repository.forks, + license: repository.license, + readmePath: repository.readmePath + } + } + : {}) + } + } +} + +export function parsePostPreview(value: unknown, platform: 'reddit' | 'tiktok'): LibraryMetadata { + if (!record(value) || (!text(value.title) && !text(value.thumbnail_url))) + throw new LibraryProviderError(`${platform} returned no public post preview.`, 'blocked') + const description = text(value.title) + return { + title: description?.slice(0, 180), + description, + author: text(value.author_name), + thumbnailUrl: text(value.thumbnail_url), + fields: { + title: labelCoverage, + description: { + state: 'partial', + reason: 'Public embed preview; full post text was not returned.' + }, + captions: { + state: 'unavailable', + reason: 'The public preview does not expose spoken caption tracks.' + } + }, + provider: `${platform}-oembed/1`, + fetchedAt: Date.now(), + evidence: value + } +} diff --git a/apps/electron/src/library/service.test.ts b/apps/electron/src/library/service.test.ts new file mode 100644 index 000000000..54f97450c --- /dev/null +++ b/apps/electron/src/library/service.test.ts @@ -0,0 +1,433 @@ +import type { LibraryMetadata, LibraryResource } from './types' +import type { DataService } from '../data-process/data-service' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { LibraryProviderError } from './providers' +import { LibraryService } from './service' + +const stubs = vi.hoisted(() => ({ + metadata: vi.fn(), + extractor: vi.fn(), + fetch: vi.fn(), + listNodes: vi.fn(), + putBlob: vi.fn() +})) +vi.mock('./providers', async (original) => ({ + ...(await original()), + fetchLibraryMetadata: stubs.metadata, + fetchExtractorMetadata: stubs.extractor, + fetchPublic: stubs.fetch +})) + +const metadata: LibraryMetadata = { + title: 'Fetched title', + description: 'Full description', + tracks: [], + fields: { + title: { state: 'complete' }, + description: { state: 'complete' }, + captions: { state: 'complete' } + }, + provider: 'fixture', + fetchedAt: 1, + evidence: {} +} +const resource = (id: string): LibraryResource => ({ + id, + platform: 'youtube', + platformContentId: 'abcdefghijk', + url: 'https://www.youtube.com/watch?v=abcdefghijk', + title: 'Unresolved title', + sourceText: '', + actor: '', + privacy: 'private', + addedAt: 1 +}) +let root: string +let service: LibraryService +const open = (path: string) => { + const result = new LibraryService( + { + importDeterministicNodes: vi.fn(async () => undefined), + listNodes: stubs.listNodes, + putBlob: stubs.putBlob + } as unknown as DataService, + path + ) + result.configure({ authorDID: 'did:key:fixture', signingKey: Array(32).fill(1) }) + return result +} +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-enrichment-')) + vi.useFakeTimers() + stubs.metadata.mockReset().mockResolvedValue(metadata) + stubs.extractor.mockReset().mockResolvedValue(metadata) + stubs.fetch.mockReset() + stubs.listNodes.mockReset().mockResolvedValue([]) + stubs.putBlob.mockReset() + service = open(root) +}) +afterEach(async () => { + await service.close() + vi.useRealTimers() + await rm(root, { recursive: true, force: true }) +}) +const count = (capability: string, state: string) => + service.status().counts.find((row) => row.capability === capability && row.state === state) + ?.count ?? 0 + +it('starts metadata before a large local index backlog finishes', async () => { + for (let i = 0; i < 500; i++) service.store.seed(resource(`source-${i}`)) + service.resume() + await vi.advanceTimersByTimeAsync(1) + expect(count('metadata', 'complete')).toBe(1) + expect(count('index', 'queued')).toBeGreaterThan(0) + expect(stubs.metadata).toHaveBeenCalledOnce() +}) + +it('lets another video progress while one request hangs, and drains cancellation before pausing', async () => { + service.store.seed(resource('a')) + service.store.seed(resource('b')) + stubs.metadata.mockImplementation((item: LibraryResource, signal: AbortSignal) => + item.id === 'a' + ? new Promise((_resolve, reject) => + signal.addEventListener('abort', () => reject(new Error('cancelled')), { once: true }) + ) + : Promise.resolve(metadata) + ) + service.resume() + await vi.advanceTimersByTimeAsync(1500) + expect(service.store.get('b')?.metadata?.title).toBe('Fetched title') + expect(service.status().running.some((job) => job.resourceId === 'a')).toBe(true) + await service.pause() + expect(service.status().running).toEqual([]) + expect(count('metadata', 'queued')).toBe(1) + stubs.metadata.mockResolvedValue(metadata) + service.resume() + await vi.advanceTimersByTimeAsync(1000) + expect(count('metadata', 'complete')).toBe(2) +}) + +it('records a restricted video without putting the whole provider into backoff', async () => { + service.store.seed(resource('a')) + service.store.seed(resource('b')) + stubs.metadata.mockRejectedValueOnce(new LibraryProviderError('Private video', 'blocked')) + service.resume() + await vi.advanceTimersByTimeAsync(1500) + expect(count('metadata', 'blocked')).toBe(1) + expect(service.store.get('b')?.metadata?.title).toBe('Fetched title') + expect(service.status().error).toBeNull() +}) + +it('retains both fetched captions and a thumbnail when their requests finish out of order', async () => { + service.store.seed(resource('a')) + stubs.metadata.mockResolvedValue({ + ...metadata, + thumbnailUrl: 'https://example.com/thumbnail.png', + tracks: [ + { url: 'https://example.com/captions', format: 'json3', language: 'en', autoGenerated: true } + ] + }) + stubs.fetch.mockImplementation( + async (url: string) => + new Response( + url.endsWith('.png') + ? Buffer.from( + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAACXBIWXMAAAPoAAAD6AG1e1JrAAAADElEQVQImWP4//8/AAX+Av5Y8msOAAAAAElFTkSuQmCC', + 'base64' + ) + : JSON.stringify({ + events: [ + { + tStartMs: 0, + dDurationMs: 1000, + segs: [{ utf8: 'Caption survives the image request.' }] + } + ] + }) + ) + ) + let finishBlob!: (cid: string) => void + stubs.putBlob.mockImplementation( + () => + new Promise((resolve) => { + finishBlob = resolve + }) + ) + service.resume() + await vi.waitFor(() => { + expect(stubs.putBlob).toHaveBeenCalledOnce() + expect(count('transcript', 'complete')).toBe(1) + }) + finishBlob('fixture-image-cid') + await vi.advanceTimersByTimeAsync(1) + expect(count('thumbnail', 'complete')).toBe(1) + const saved = service.store.get('a')! + expect(saved.thumbnail?.cid).toBe('fixture-image-cid') + expect(saved.transcript?.cues[0].text).toBe('Caption survives the image request.') + expect(saved.metadata?.title).toBe('Fetched title') +}) + +it('honors a rate limit after retry and restart, then processes every remaining video', async () => { + for (let i = 0; i < 250; i++) service.store.seed(resource(`video-${i}`)) + stubs.metadata.mockRejectedValueOnce( + new LibraryProviderError('HTTP 429', 'retry', Date.now() + 60_000, 'provider') + ) + service.resume() + await vi.advanceTimersByTimeAsync(2000) + service.retry() + await service.close() + service = open(root) + service.resume() + await vi.advanceTimersByTimeAsync(30_000) + expect(stubs.metadata).toHaveBeenCalledOnce() + expect(service.status().nextAt).toBeGreaterThan(Date.now()) + await vi.advanceTimersByTimeAsync(300_000) + expect(count('metadata', 'complete')).toBe(250) + expect(count('metadata', 'queued')).toBe(0) + expect(count('metadata', 'retry')).toBe(0) + expect(service.status().running).toEqual([]) + expect(service.status().nextAt).toBeNull() + for (let i = 0; i < 250; i++) + expect(service.store.get(`video-${i}`)?.metadata?.title).toBe('Fetched title') +}) + +it('reports an empty caption response as an access gap without retrying or blocking metadata', async () => { + service.store.seed(resource('a')) + service.store.seed(resource('b')) + stubs.metadata.mockResolvedValue({ + ...metadata, + tracks: [ + { url: 'https://example.com/captions', format: 'json3', language: 'en', autoGenerated: true } + ] + }) + stubs.fetch.mockImplementation(async () => new Response('')) + service.resume() + await vi.advanceTimersByTimeAsync(3000) + expect(count('metadata', 'complete')).toBe(2) + expect(count('transcript', 'blocked')).toBe(2) + expect(count('transcript', 'retry')).toBe(0) + expect( + service + .status() + .recent.filter((job) => job.capability === 'transcript') + .every((job) => job.reason?.includes('empty caption track')) + ).toBe(true) +}) + +it('never fetches private conversation text or fabricates remote media jobs', async () => { + service.store.seed({ + ...resource('conversation'), + kind: 'conversation', + url: '', + platform: 'claude', + sourceText: 'Private conversation contents' + }) + service.resume() + await vi.advanceTimersByTimeAsync(2000) + expect(stubs.metadata).not.toHaveBeenCalled() + expect(stubs.fetch).not.toHaveBeenCalled() + expect(stubs.extractor).not.toHaveBeenCalled() + expect(service.store.search({ text: 'contents' })[0].id).toBe('conversation') + expect(count('metadata', 'not-applicable')).toBe(1) + expect(count('transcript', 'not-applicable')).toBe(1) + expect(count('thumbnail', 'not-applicable')).toBe(1) +}) + +it('indexes saved comments and local export text without turning garden commentary into a second card', async () => { + stubs.listNodes.mockImplementation(async (options: { schemaId: string }) => + options.schemaId.includes('/SocialContent@') + ? [ + { + id: 'comment', + createdAt: 1, + properties: { + platform: 'reddit', + parentContent: 'post', + contentKind: 'comment', + canonicalUrl: 'https://www.reddit.com/r/example/comments/post/comment', + searchText: 'Saved comment' + } + }, + { + id: 'local', + createdAt: 1, + properties: { + platform: 'tiktok', + contentKind: 'comment', + searchText: 'Exported local comment' + } + }, + { + id: 'note', + createdAt: 1, + properties: { + platform: 'generic', + parentContent: 'comment', + platformContentKind: 'garden-commentary', + searchText: 'My garden note' + } + } + ] + : [] + ) + expect(await service.scan()).toBe(2) + expect(service.store.search({ text: 'Saved comment' })[0].id).toBe('comment') + expect(service.store.get('comment')?.notes?.[0].text).toBe('My garden note') + expect(service.store.get('local')?.kind).toBe('archive-text') + expect(service.store.get('note')).toBeNull() +}) + +it('reads graph memberships beyond the first page and shares concurrent graph reads', async () => { + for (let i = 0; i < 1001; i++) service.store.put(resource(`graph-${i}`)) + const { SocialCollectionSchema, SocialCollectionItemSchema } = + await import('@xnetjs/social/schemas') + const memberships = Array.from({ length: 1001 }, (_, i) => ({ + id: `membership-${i}`, + properties: { collection: 'playlist', item: `graph-${i}` } + })) + stubs.listNodes.mockImplementation( + async (options: { schemaId: string; offset: number; limit: number }) => { + const rows = + options.schemaId === SocialCollectionSchema._schemaId + ? [{ id: 'playlist', properties: { title: 'Full playlist' } }] + : options.schemaId === SocialCollectionItemSchema._schemaId + ? memberships + : [] + return rows.slice(options.offset, options.offset + options.limit) + } + ) + const first = service.graph() + expect(service.graph()).toBe(first) + const graph = await first + expect(graph.linkCount).toBe(1001) + expect(graph.edges.filter((edge) => edge.kind === 'collection')).toHaveLength(1001) + expect( + stubs.listNodes.mock.calls + .filter(([options]) => options.schemaId === SocialCollectionItemSchema._schemaId) + .map(([options]) => options.offset) + ).toEqual([0, 500, 1000]) +}) + +it('does not start new graph reads after the recovery write barrier closes', async () => { + await service.freeze() + await expect(service.graph()).rejects.toThrow('being copied or imported') + await expect(service.graphDetail('example')).rejects.toThrow('being copied or imported') + service.thaw() + expect((await service.graph()).nodes).toEqual([]) +}) + +it('keeps other providers moving while several requests from one source hang', async () => { + const hosts = ['youtube', 'instagram', 'github', 'reddit'] + for (const host of hosts) + for (let i = 0; i < 4; i++) + service.store.seed({ + ...resource(`${host}-${i}`), + platform: host, + url: + host === 'youtube' + ? resource('').url + : host === 'instagram' + ? `https://www.instagram.com/p/post${i}/` + : host === 'github' + ? `https://github.com/owner/repo${i}` + : `https://www.reddit.com/comments/${i}` + }) + stubs.metadata.mockImplementation( + (_item: LibraryResource, signal: AbortSignal) => + new Promise((_resolve, reject) => + signal.addEventListener('abort', () => reject(new Error('cancelled')), { once: true }) + ) + ) + service.resume() + await vi.advanceTimersByTimeAsync(9000) + const called = new Set( + stubs.metadata.mock.calls.map(([item]) => (item as LibraryResource).platform) + ) + expect([...called].sort()).toEqual(hosts.sort()) + expect(service.status().running.length).toBeLessThanOrEqual(10) + await service.pause() + expect(service.status().running).toEqual([]) +}) + +it('backs off repeated provider throttling without exhausting the per-resource retry limit', async () => { + service.store.seed(resource('throttled')) + stubs.metadata.mockImplementation(() => + Promise.reject( + new LibraryProviderError('Rate limited', 'retry', Date.now() + 60_000, 'provider') + ) + ) + service.resume() + for (let attempt = 1; attempt <= 6; attempt++) { + const startedAt = Date.now() + await vi.advanceTimersByTimeAsync(1000) + expect(stubs.metadata).toHaveBeenCalledTimes(attempt) + expect(count('metadata', 'blocked')).toBe(0) + expect(count('metadata', 'retry')).toBe(1) + const nextAt = service.status().nextAt! + expect(nextAt).toBeGreaterThanOrEqual(startedAt + 30_000 * 2 ** attempt) + vi.setSystemTime(nextAt) + } + stubs.metadata.mockResolvedValue(metadata) + await vi.advanceTimersByTimeAsync(1000) + expect(count('metadata', 'complete')).toBe(1) +}) + +it('preserves a longer provider Retry-After across restart while another provider progresses', async () => { + service.store.seed(resource('throttled')) + service.store.seed({ + ...resource('other-provider'), + platform: 'github', + url: 'https://github.com/example/repo' + }) + const retryAt = Date.now() + 2 * 60 * 60 * 1000 + stubs.metadata.mockImplementation((item: LibraryResource) => + item.id === 'throttled' + ? Promise.reject(new LibraryProviderError('Rate limited', 'retry', retryAt, 'provider')) + : Promise.resolve(metadata) + ) + service.resume() + await vi.advanceTimersByTimeAsync(1500) + expect(service.store.get('other-provider')?.metadata?.title).toBe('Fetched title') + expect(count('metadata', 'retry')).toBe(1) + expect(service.status().nextAt).toBe(retryAt) + await service.close() + service = open(root) + service.resume() + stubs.metadata.mockResolvedValue(metadata) + vi.setSystemTime(retryAt - 2000) + await vi.advanceTimersByTimeAsync(1000) + expect(stubs.metadata).toHaveBeenCalledTimes(2) + expect(count('metadata', 'retry')).toBe(1) + await vi.advanceTimersByTimeAsync(2000) + expect(count('metadata', 'complete')).toBe(2) +}) + +it.each([ + ['images.example', 2], + ['www.youtube.com', 1], + [undefined, 1] +])('keeps image throttling scoped to its source host (%s)', async (host, expectedMetadata) => { + service.store.seed(resource('a')) + service.store.seed(resource('b')) + stubs.metadata.mockResolvedValue({ + ...metadata, + thumbnailUrl: 'https://images.example/poster.png' + }) + stubs.fetch.mockRejectedValue( + new LibraryProviderError('Image rate limit', 'retry', Date.now() + 60_000, 'provider', host) + ) + service.resume() + await vi.advanceTimersByTimeAsync(1500) + expect(count('metadata', 'complete')).toBe(expectedMetadata) + expect(count('thumbnail', 'retry')).toBe(1) + expect(stubs.fetch).toHaveBeenCalledOnce() + await service.close() + service = open(root) + service.resume() + await vi.advanceTimersByTimeAsync(5000) + expect(stubs.fetch).toHaveBeenCalledOnce() + expect(count('metadata', 'complete')).toBe(expectedMetadata) +}) diff --git a/apps/electron/src/library/service.ts b/apps/electron/src/library/service.ts new file mode 100644 index 000000000..10079bf2d --- /dev/null +++ b/apps/electron/src/library/service.ts @@ -0,0 +1,622 @@ +import type { CaptureInput, CaptureResult } from './capture' +import type { LibraryJob, LibraryResource, LibraryStatus } from './types' +import type { DataService } from '../data-process/data-service' +import type { LibraryHelperStatus } from '../shared/library' +import type { LibraryGraph, LibraryGraphDetail } from '../shared/library-graph' +import type { DeterministicNodeImportDraft } from '@xnetjs/data' +import { createHash } from 'node:crypto' +import { dirname, join } from 'node:path' +import { PageSchema } from '@xnetjs/data' +import { resourceIdentityForUrl } from '@xnetjs/social/import/core' +import { + SocialContentSchema, + SocialEnrichmentSchema, + createSocialEnrichmentId +} from '@xnetjs/social/schemas' +import { createTranscriptContentDrafts } from '@xnetjs/social/transcripts' +import sharp from 'sharp' +import { readPageText, saveCapture } from './capture' +import { conversationResources } from './conversations' +import { readLibraryGraph } from './graph' +import { inspectManagedHelper, installManagedHelper, MAC_VIDEO_HELPER } from './managed-helper' +import { fetchLibraryMetadata, fetchPublic, LibraryProviderError } from './providers' +import { providerResource, queueProvider } from './source' +import { LibraryStore } from './store' +import { fetchLibraryTranscript } from './transcripts' + +const string = (value: unknown): string => (typeof value === 'string' ? value : '') +const errorMessage = (error: unknown) => (error instanceof Error ? error.message : String(error)) + +export class LibraryService { + readonly store: LibraryStore + private identity: { authorDID: string; signingKey: number[] } | null = null + private active = new Map< + string, + { job: LibraryJob; controller: AbortController; done: Promise } + >() + private scanning: Promise | null = null + private graphRead: Promise | null = null + private frozen = false + private fatal: string | null = null + private projection: Promise = Promise.resolve() + private timer: ReturnType + private readonly helperDirectory: string + private helperController: AbortController | null = null + private helperInstall: Promise | null = null + constructor( + private readonly data: DataService, + dataPath: string + ) { + this.store = new LibraryStore(join(dataPath, 'library.db')) + this.helperDirectory = join(dirname(dataPath), 'library-helpers') + this.timer = setInterval(() => { + void this.tick().catch((error: unknown) => { + this.fatal = errorMessage(error) + }) + }, 250) + this.timer.unref() + } + async helperStatus(): Promise { + const base = { version: MAC_VIDEO_HELPER.version, bytes: MAC_VIDEO_HELPER.size } + if (process.platform !== 'darwin') + return { + ...base, + state: 'unsupported', + reason: 'Managed video-helper installation currently supports macOS.' + } + if (this.helperInstall) return { ...base, state: 'installing' } + return inspectManagedHelper(this.helperDirectory) + } + cancelHelper(): void { + this.helperController?.abort() + } + installHelper(): Promise { + if (process.platform !== 'darwin') + throw new Error('Managed video-helper installation currently supports macOS.') + if (process.env.XNET_RECOVERY_OFFLINE === 'true') + throw new Error( + 'Review this recovered workspace and reconnect before downloading the helper.' + ) + this.requireWritable() + if (this.helperInstall) throw new Error('Video helper installation is already running.') + const controller = new AbortController() + this.helperController = controller + this.helperInstall = installManagedHelper({ + directory: this.helperDirectory, + signal: controller.signal, + download: (url, signal, limit) => fetchPublic(url, { signal, limit, timeoutMs: 120_000 }) + }).finally(() => { + this.helperInstall = null + this.helperController = null + }) + return this.helperInstall + } + configure(identity: { authorDID: string; signingKey: number[] }): void { + if (!identity.authorDID || identity.signingKey.length !== 32) + throw new Error('Library identity is not ready.') + this.identity = identity + } + status(): LibraryStatus & { error: string | null } { + return { ...this.store.status(), error: this.fatal } + } + graph(): Promise { + if (this.frozen) + return Promise.reject( + new Error('Graph is unavailable while workspace storage is being copied or imported.') + ) + if (!this.graphRead) + this.graphRead = (async () => { + await this.scanning + return readLibraryGraph(this.data, this.store) + })().finally(() => { + this.graphRead = null + }) + return this.graphRead + } + async graphDetail(id: string): Promise { + if (this.frozen) + throw new Error( + 'Graph details are unavailable while workspace storage is being copied or imported.' + ) + return { + resource: this.store.get(id), + source: (await this.data.getNode(id))?.properties ?? null + } + } + async capture(input: CaptureInput): Promise { + if (!this.frozen || !this.identity) + throw new Error('Capture requires the workspace write barrier and identity.') + return saveCapture({ input, store: this.store, data: this.data, identity: this.identity }) + } + async recoverCaptures(): Promise { + for (const intent of this.store.pendingCaptures()) await this.capture(intent.input) + } + async lookup( + url: string + ): Promise<{ id: string; title: string; notes: { pageId: string; title: string }[] } | null> { + const identity = resourceIdentityForUrl(url) + const cached = this.store.byUrl(identity.url) ?? this.store.get(identity.id) + const node = cached ? null : await this.data.getNode(identity.id) + if (!cached && !node) return null + return { + id: cached?.id ?? node!.id, + title: cached?.metadata?.title || cached?.title || string(node?.properties.title) || url, + notes: (cached?.notes ?? []).flatMap((note) => + note.pageId ? [{ pageId: note.pageId, title: note.title }] : [] + ) + } + } + async pause(): Promise { + this.requireWritable() + this.store.setPaused(true) + for (const task of this.active.values()) task.controller.abort() + await Promise.all([...this.active.values()].map((task) => task.done)) + } + resume(): void { + this.requireWritable() + if (process.env.XNET_RECOVERY_OFFLINE === 'true') + throw new Error('Review this recovered workspace and reconnect before fetching sources.') + if (!this.identity) throw new Error('Library identity is not ready.') + this.fatal = null + this.store.setPaused(false) + void this.tick().catch((error: unknown) => { + this.fatal = errorMessage(error) + }) + } + private requireWritable(): void { + if (this.frozen) + throw new Error('Library is paused while workspace storage is being copied or imported.') + } + retry(id?: string): void { + this.requireWritable() + this.store.retry(id) + } + async freeze(): Promise { + this.frozen = true + for (const task of this.active.values()) task.controller.abort() + await Promise.all([...this.active.values()].map((task) => task.done)) + await this.scanning + await this.graphRead + } + thaw(): void { + this.frozen = false + } + async close(): Promise { + clearInterval(this.timer) + await this.freeze() + this.cancelHelper() + // Cancellation already reaches the requesting renderer; quit only waits for cleanup. + if (this.helperInstall) await Promise.allSettled([this.helperInstall]) + this.store.close() + this.identity?.signingKey.fill(0) + } + scan(): Promise { + if (this.frozen) return Promise.reject(new Error('Library is paused for a recovery copy.')) + if (this.scanning) return this.scanning + this.scanning = this.scanAll().finally(() => { + this.scanning = null + }) + return this.scanning + } + private async scanAll(): Promise { + const notes = new Map>() + let offset = 0 + let count = 0 + for (;;) { + const nodes = await this.data.listNodes({ + schemaId: SocialContentSchema._schemaId, + orderBy: { createdAt: 'asc' }, + limit: 500, + offset + }) + for (const node of nodes) { + const props = node.properties + if (props.contentKind === 'transcript') continue + if (props.parentContent && props.platformContentKind === 'garden-commentary') { + const parent = string(props.parentContent) + notes.set(parent, [ + ...(notes.get(parent) ?? []), + { + id: node.id, + title: string(props.title), + text: string(props.searchText), + url: string(props.canonicalUrl), + author: string(props.actorHandle) + } + ]) + continue + } + const url = string(props.canonicalUrl) || string(props.platformUrl) + const sourceText = string(props.searchText) || string(props.textPreview) + const remote = /^https?:\/\//.test(url) + if (!remote && !sourceText && !string(props.title)) continue + this.store.seed({ + id: node.id, + ...(!remote ? { kind: 'archive-text' as const } : {}), + platform: string(props.platform) || 'generic', + platformContentId: string(props.platformContentId) || url || node.id, + url, + title: string(props.title) || url, + sourceText, + actor: string(props.actorHandle), + privacy: string(props.privacyClass) || 'unknown', + addedAt: node.createdAt + }) + count++ + } + offset += nodes.length + if (nodes.length < 500) { + for await (const resource of conversationResources(this.data)) { + this.store.seed(resource) + count++ + } + await this.collectPageNotes(notes) + this.store.replaceSourceNotes(notes) + return count + } + } + } + private async collectPageNotes( + notes: Map> + ): Promise { + for (let offset = 0; ; offset += 100) { + const pages = await this.data.listNodes({ + schemaId: PageSchema._schemaId, + limit: 100, + offset, + orderBy: { createdAt: 'asc' } + }) + for (const page of pages) { + const sources = page.properties.sourceResources + if (!Array.isArray(sources) || !sources.length) continue + const bytes = await this.data.getDocumentContent(page.id) + if (!bytes) + throw new Error( + `Source note ${page.id} has no saved document; search was not marked complete.` + ) + const body = readPageText(bytes) + for (const id of sources) + if (typeof id === 'string') + notes.set(id, [ + ...(notes.get(id) ?? []), + { + id: page.id, + title: string(page.properties.title), + text: body, + url: '', + author: page.createdBy, + pageId: page.id + } + ]) + } + if (pages.length < 100) return + } + } + private async tick(): Promise { + if ( + this.frozen || + this.fatal || + this.store.paused || + !this.identity || + process.env.XNET_RECOVERY_OFFLINE === 'true' + ) + return + // Local indexing is already available at import. Drain repair work in bounded batches, + // independently of network work, instead of delaying every source by one second. + const deadline = Date.now() + 25 + for (let count = 0; count < 100 && Date.now() < deadline; count++) { + const job = this.store.next(Date.now(), ['index']) + if (!job) break + const resource = this.store.get(job.resourceId) + try { + if (resource) this.store.index(resource) + } catch (error) { + this.store.finish(job, 'retry', errorMessage(error), Date.now() + 30_000) + throw error + } + this.store.finish( + job, + resource ? 'complete' : 'unavailable', + resource ? null : 'Source resource is missing.' + ) + } + while (this.active.size < 10) { + const capabilities = (['metadata', 'thumbnail', 'transcript'] as const).filter( + (capability) => + [...this.active.values()].filter((task) => task.job.capability === capability).length < + (capability === 'metadata' ? 6 : 2) + ) + const providers = new Map() + for (const task of this.active.values()) { + const source = this.store.get(task.job.resourceId) + if (source) { + const provider = queueProvider(source) + providers.set(provider, (providers.get(provider) ?? 0) + 1) + } + } + // Slow requests from one host cannot occupy every worker. Pacing and Retry-After + // remain enforced by the persistent queue, including after a restart. + const excluded = [...providers] + .filter(([, count]) => count >= 3) + .map(([provider]) => provider) + const job = this.store.next(Date.now(), capabilities, excluded) + if (!job) break + const key = `${job.resourceId}:${job.capability}` + const controller = new AbortController() + const done = this.execute(job, controller.signal) + .catch((error: unknown) => { + this.fatal = errorMessage(error) + }) + .finally(() => { + this.active.delete(key) + }) + this.active.set(key, { job, controller, done }) + } + } + private async persist(drafts: DeterministicNodeImportDraft[]): Promise { + if (!this.identity) throw new Error('Library identity is not ready.') + const identity = this.identity + // Parallel fetches share one ordered projection writer and Lamport allocator. + const next = this.projection.then(async () => { + await this.data.importDeterministicNodes({ + ...identity, + drafts, + policy: { indexMode: 'touched', notificationMode: 'batch', syncMode: 'defer' } + }) + }) + this.projection = next.catch(() => {}) + await next + } + private async project(resource: LibraryResource): Promise { + if (!this.identity) throw new Error('Library identity is not ready.') + const metadata = resource.metadata + if (!metadata) return + const properties: Record = { + platform: resource.platform, + platformContentId: resource.platformContentId, + canonicalUrl: resource.url, + status: 'resolved', + fetchedAt: metadata.fetchedAt, + title: (metadata.title || resource.title).slice(0, 1000), + description: (metadata.description || '').slice(0, 5000), + ...(metadata.provider.startsWith('github') + ? { source: 'data-api' } + : ['oembed', 'open-graph'].includes(metadata.provider) + ? { source: metadata.provider } + : {}), + metadataJson: JSON.stringify({ + fields: metadata.fields, + provider: metadata.provider, + fullTextInDesktopLibrary: true, + thumbnailContentType: resource.thumbnail?.contentType + }) + } + if (metadata.author) properties.authorName = metadata.author.slice(0, 500) + if (metadata.thumbnailUrl) properties.thumbnailUrl = metadata.thumbnailUrl + if (resource.thumbnail) properties.thumbnailBlobCid = resource.thumbnail.cid + await this.persist([ + { + id: createSocialEnrichmentId(resource.platform, resource.platformContentId), + schemaId: SocialEnrichmentSchema._schemaId, + properties + } + ]) + } + private retain( + id: string, + patch: Partial> + ): LibraryResource { + const current = this.store.get(id) + if (!current) throw new Error('Source resource disappeared while enrichment was running.') + // Another capability may have finished while this request was in flight. + const next = { ...current, ...patch } + this.store.put(next) + return next + } + private async execute(job: LibraryJob, signal: AbortSignal): Promise { + const resource = this.store.get(job.resourceId) + if (!resource) { + this.store.finish(job, 'unavailable', 'Source resource is missing.') + return + } + if (resource.kind && job.capability !== 'index') { + this.store.finish( + job, + 'not-applicable', + 'Text is imported locally; there is no public source to fetch.' + ) + return + } + const provider = queueProvider(resource) + const interval = + job.capability === 'thumbnail' + ? 250 + : ['instagram', 'tiktok'].includes(provider) + ? 8000 + : provider === 'youtube' + ? 1000 + : 2000 + if (job.capability !== 'index') + this.store.pauseProvider(`${provider}:${job.capability}`, Date.now() + interval) + try { + if (job.capability === 'index') { + this.store.index(resource) + this.store.finish(job, 'complete') + return + } + if (job.capability === 'metadata') { + const metadata = await fetchLibraryMetadata(resource, signal, this.helperDirectory) + // Retain the full result independently of bounded card projections. + const next = this.retain(resource.id, { metadata }) + this.store.reindex(resource.id) + await this.project(next) + const gaps = Object.entries(metadata.fields).filter( + ([, field]) => field.state !== 'complete' + ) + this.store.finish( + job, + gaps.length ? 'partial' : 'complete', + gaps.length + ? gaps.map(([name, field]) => `${name}: ${field.reason ?? field.state}`).join('; ') + : null + ) + return + } + if (job.capability === 'thumbnail') { + const url = resource.metadata?.thumbnailUrl + if (!url) { + this.store.finish( + job, + 'unavailable', + resource.metadata + ? 'Source returned no thumbnail.' + : 'Metadata is unresolved; retry after it succeeds.' + ) + return + } + const response = await fetchPublic(url, { signal, limit: 8 * 1024 * 1024 }) + const bytes = new Uint8Array(await response.arrayBuffer()) + const contentType = imageContentType(bytes) + if (!contentType) + throw new LibraryProviderError( + 'Thumbnail response is not a supported raster image.', + 'retry' + ) + // Decode every saved poster before acknowledging it; a valid header alone is insufficient. + await sharp(bytes, { failOn: 'warning', limitInputPixels: 40_000_000 }) + .resize({ width: 1920, height: 1920, fit: 'inside', withoutEnlargement: true }) + .raw() + .toBuffer() + const cid = await this.data.putBlob(bytes) + const next = this.retain(resource.id, { + thumbnail: { cid, contentType, bytes: bytes.length } + }) + await this.project(next) + this.store.finish(job, 'complete') + return + } + if (!resource.metadata) { + this.store.finish( + job, + 'blocked', + 'Metadata has not succeeded; caption discovery is pending.' + ) + return + } + if ( + !resource.metadata.fields.captions && + !resource.metadata.tracks?.length && + providerResource(resource).platform !== 'youtube' + ) { + this.store.finish(job, 'not-applicable', 'This page does not expose caption tracks.') + return + } + const result = await fetchLibraryTranscript(resource, signal, this.helperDirectory) + if (result.status === 'unavailable') { + this.store.finish(job, 'unavailable', result.reason) + return + } + const { track, cues, raw, provider: captionProvider } = result.value + const transcript: NonNullable = { + cues, + language: track.language, + autoGenerated: track.autoGenerated, + source: 'captions', + fetchedAt: Date.now(), + provider: captionProvider, + evidence: raw + } + this.retain(resource.id, { transcript }) + this.store.reindex(resource.id) + const digest = createHash('sha256').update(raw).digest('hex') + const splitCues = cues.flatMap((cue) => { + const pieces: typeof cues = [] + for (let offset = 0; offset < cue.text.length; offset += 16000) + pieces.push({ ...cue, text: cue.text.slice(offset, offset + 16000) }) + return pieces + }) + const drafts: DeterministicNodeImportDraft[] = createTranscriptContentDrafts({ + platform: resource.platform, + platformContentId: resource.platformContentId, + videoNodeId: resource.id, + videoTitle: resource.metadata.title || resource.title, + canonicalUrl: resource.url, + cues: splitCues, + language: track.language, + autoGenerated: track.autoGenerated, + fetchedAtMs: transcript.fetchedAt, + privacyClass: resource.privacy + }).map((draft) => ({ + ...draft, + id: `${draft.id}:${digest.slice(0, 16)}`, + schemaId: SocialContentSchema._schemaId, + properties: { + ...draft.properties, + metadataJson: JSON.stringify({ + ...(JSON.parse(draft.properties.metadataJson) as object), + track: { + language: track.language, + autoGenerated: track.autoGenerated, + provider: transcript.provider, + sha256: digest + } + }) + } + })) + if (!this.identity) throw new Error('Library identity is not ready.') + for (let offset = 0; offset < drafts.length; offset += 100) + await this.persist(drafts.slice(offset, offset + 100)) + this.store.finish(job, 'complete') + } catch (error) { + if (signal.aborted) { + this.store.finish(job, 'queued', 'Paused; ready to retry.') + return + } + const failure = + error instanceof LibraryProviderError + ? error + : new LibraryProviderError(errorMessage(error), 'retry') + // Retry-After is a floor; a short hint must not defeat repeated-failure backoff. + const next = Math.max( + failure.retryAt ?? 0, + Date.now() + Math.min(24 * 60 * 60 * 1000, 30_000 * 2 ** Math.min(job.attempts, 10)) + ) + const state = + failure.disposition === 'retry' && failure.scope !== 'provider' && job.attempts >= 5 + ? 'blocked' + : failure.disposition + this.store.finish(job, state, failure.message, next) + if (failure.scope === 'provider') { + const sourceHosts = [ + new URL(resource.url).hostname.replace(/^www\./, ''), + provider.startsWith('web:') ? provider.slice(4) : `${provider}.com`, + ...(provider === 'x' ? ['twitter.com'] : []) + ] + const throttledHost = failure.host + const separateImageHost = + job.capability === 'thumbnail' && + throttledHost && + !sourceHosts.some((host) => throttledHost === host || throttledHost.endsWith(`.${host}`)) + // A CDN's Retry-After applies to its images; repository pages can still progress. + this.store.pauseProvider( + separateImageHost ? `${provider}:thumbnail` : provider, + Math.max(next, Date.now() + 60_000) + ) + } + } + } +} + +export function imageContentType(bytes: Uint8Array): string | null { + const prefix = Buffer.from(bytes.subarray(0, 16)) + if ( + prefix.length >= 8 && + prefix.subarray(0, 8).equals(Buffer.from([137, 80, 78, 71, 13, 10, 26, 10])) + ) + return 'image/png' + if (prefix[0] === 255 && prefix[1] === 216 && prefix[2] === 255) return 'image/jpeg' + if (/^GIF8[79]a/.test(prefix.toString('ascii'))) return 'image/gif' + if (prefix.toString('ascii', 0, 4) === 'RIFF' && prefix.toString('ascii', 8, 12) === 'WEBP') + return 'image/webp' + return null +} diff --git a/apps/electron/src/library/source.ts b/apps/electron/src/library/source.ts new file mode 100644 index 000000000..89154d566 --- /dev/null +++ b/apps/electron/src/library/source.ts @@ -0,0 +1,48 @@ +import type { LibraryResource } from './types' + +/** A citation keeps its imported identity while its URL selects the public provider. */ +export function providerResource(resource: LibraryResource): LibraryResource { + const url = new URL(resource.url) + const host = url.hostname.toLowerCase().replace(/^www\./, '') + const youtube = + host === 'youtu.be' + ? url.pathname.split('/')[1] + : ['youtube.com', 'm.youtube.com'].includes(host) + ? url.searchParams.get('v') || url.pathname.match(/^\/(?:shorts|live|embed)\/([\w-]+)/)?.[1] + : null + if (youtube && /^[\w-]{11}$/.test(youtube)) + return { ...resource, platform: 'youtube', platformContentId: youtube } + const instagram = + host === 'instagram.com' + ? url.pathname.match(/^\/(?:[^/]+\/)?(?:p|reels?|tv)\/([\w-]+)\/?$/)?.[1] + : null + if (instagram) return { ...resource, platform: 'instagram', platformContentId: instagram } + const tweet = ['x.com', 'twitter.com', 'mobile.twitter.com'].includes(host) + ? url.pathname.match(/\/(?:i\/web|[^/]+)\/status\/(\d+)(?:\/|$)/)?.[1] + : null + if (tweet) return { ...resource, platform: 'x', platformContentId: tweet } + const tiktok = ['tiktok.com', 'm.tiktok.com', 'tiktokv.com'].includes(host) + ? url.pathname.match(/\/video\/(\d+)(?:\/|$)/)?.[1] + : null + if (tiktok) + return { + ...resource, + platform: 'tiktok', + platformContentId: tiktok, + url: `https://www.tiktok.com/@_/video/${tiktok}` + } + if (host === 'github.com' && /^\/[^/]+\/[^/]+\/?$/.test(url.pathname)) + return { ...resource, platform: 'github' } + if (['reddit.com', 'old.reddit.com', 'redd.it'].includes(host)) + return { ...resource, platform: 'reddit' } + return resource +} + +export function queueProvider(resource: LibraryResource): string { + if (!/^https?:\/\//.test(resource.url)) return resource.platform + const target = providerResource(resource) + if (['youtube', 'instagram', 'github', 'reddit', 'tiktok', 'x'].includes(target.platform)) + return target.platform + // An archive's provenance is not the host serving its outbound citations. + return `web:${new URL(resource.url).hostname.toLowerCase().replace(/^www\./, '')}` +} diff --git a/apps/electron/src/library/store.test.ts b/apps/electron/src/library/store.test.ts new file mode 100644 index 000000000..45c434ad1 --- /dev/null +++ b/apps/electron/src/library/store.test.ts @@ -0,0 +1,452 @@ +import type { LibraryResource } from './types' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import Database from 'better-sqlite3' +import { beforeEach, afterEach, expect, it } from 'vitest' +import { LibraryStore } from './store' + +let root: string +let path: string +let store: LibraryStore +const resource = (id = 'video-one'): LibraryResource => ({ + id, + platform: 'youtube', + platformContentId: 'abcdefghijk', + title: 'A useful video', + url: 'https://www.youtube.com/watch?v=abcdefghijk', + sourceText: 'Original source notes', + actor: 'Example', + privacy: 'private', + addedAt: 1 +}) +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-library-')) + path = join(root, 'library.db') + store = new LibraryStore(path) +}) +afterEach(async () => { + store.close() + await rm(root, { recursive: true, force: true }) +}) + +it('deduplicates capability work and retains successful metadata across source scans', () => { + store.seed(resource()) + const saved = { + ...resource(), + metadata: { + title: 'Full fetched title', + description: 'Full fetched text', + fields: {}, + provider: 'fixture', + fetchedAt: 2, + evidence: {} + } + } + store.put(saved) + store.seed(resource()) + expect(store.status().counts.reduce((sum, row) => sum + row.count, 0)).toBe(4) + expect(store.get('video-one')?.metadata?.description).toBe('Full fetched text') +}) + +it('recovers interrupted jobs and keeps future retries paused across restart', () => { + store.seed(resource()) + const first = store.next(10)! + expect(first.state).toBe('running') + store.close() + store = new LibraryStore(path) + const replay = store.next(10)! + expect(replay.capability).toBe(first.capability) + store.finish(replay, 'retry', 'Network unavailable', 60_000) + store.close() + store = new LibraryStore(path) + expect( + store + .status() + .recent.some((job) => job.nextAt === 60_000 && job.reason === 'Network unavailable') + ).toBe(true) + expect(store.paused).toBe(true) +}) + +it('finds full descriptions and late transcript cues after restart', () => { + const saved: LibraryResource = { + ...resource(), + metadata: { + description: 'prefix '.repeat(5000) + 'rarewoodlandphrase', + fields: {}, + provider: 'fixture', + fetchedAt: 2, + evidence: {} + }, + transcript: { + cues: [ + { startMs: 0, durationMs: 1000, text: 'opening '.repeat(4000) }, + { startMs: 7200000, durationMs: 1000, text: 'latecopperphrase' } + ], + language: 'de', + autoGenerated: false, + source: 'captions', + fetchedAt: 2, + provider: 'fixture', + evidence: 'raw' + } + } + store.seed(saved) + store.index(saved) + store.close() + store = new LibraryStore(path) + expect(store.search({ text: 'rarewoodlandphrase' })[0].id).toBe(saved.id) + expect(store.search({ text: 'latecopperphrase' })[0].startMs).toBe(7200000) + const card = store.search({ text: 'latecopperphrase' })[0] + expect(card).not.toHaveProperty('transcript') + expect(card.metadata).not.toHaveProperty('evidence') + expect(card.metadata?.description?.length).toBeLessThanOrEqual(600) + expect(store.get(saved.id)?.transcript?.cues).toHaveLength(2) +}) + +it('backfills a new provider version once without resetting completed current work', () => { + store.seed(resource()) + store.close() + const db = new Database(path) + db.prepare("UPDATE work SET version='previous-provider',state='complete'").run() + db.close() + store = new LibraryStore(path) + expect(store.status().counts.every((row) => row.state === 'queued' && row.count === 1)).toBe(true) + const job = store.next(0, ['metadata'])! + store.finish(job, 'complete') + store.close() + store = new LibraryStore(path) + expect(store.next(0, ['metadata'])).toBeNull() + expect(store.status().counts.find((row) => row.capability === 'metadata')?.state).toBe('complete') + const check = new Database(path, { readonly: true }) + expect( + check.prepare("SELECT COUNT(*) AS n FROM work WHERE version='previous-provider'").get() + ).toEqual({ n: 4 }) + check.close() +}) + +it('preserves chronological pagination and large saved text across a browse-index upgrade', () => { + const description = 'Full retained description. '.repeat(10_000) + const rows: LibraryResource[] = [ + { ...resource('youtube-b'), addedAt: 20 }, + { ...resource('web-middle'), platform: 'web', addedAt: 15 }, + { ...resource('youtube-a'), addedAt: 20 }, + { ...resource('youtube-old'), addedAt: 5 }, + { ...resource('web-newest'), platform: 'web', addedAt: 30 } + ] + for (const row of rows) + store.seed({ + ...row, + metadata: { description, fields: {}, provider: 'fixture', fetchedAt: 2, evidence: {} } + }) + store.close() + const older = new Database(path) + older.exec('DROP INDEX IF EXISTS resource_added; DROP INDEX IF EXISTS resource_platform_added;') + older.close() + store = new LibraryStore(path) + + expect(store.search({ offset: 1, limit: 2 }).map((row) => row.id)).toEqual([ + 'youtube-a', + 'youtube-b' + ]) + expect(store.search({ platform: 'youtube', offset: 1, limit: 2 }).map((row) => row.id)).toEqual([ + 'youtube-b', + 'youtube-old' + ]) + expect(store.search({ platform: 'instagram' })).toEqual([]) + expect(store.search({ offset: 99 })).toEqual([]) + expect(store.get('youtube-a')?.metadata?.description).toBe(description) + expect(store.search({ limit: 1 })[0].metadata?.description).toHaveLength(600) +}) + +it('does not consume retry attempts for user pauses or interrupted requests', () => { + store.seed(resource()) + for (let i = 0; i < 10; i++) { + const job = store.next(0, ['metadata'])! + expect(job.attempts).toBe(1) + store.finish(job, 'queued', 'Paused') + } + store.next(0, ['metadata']) + store.close() + store = new LibraryStore(path) + expect(store.next(0, ['metadata'])?.attempts).toBe(1) +}) + +it('upgrades Instagram work without refetching completed YouTube metadata or losing backoff', () => { + store.seed(resource('youtube')) + store.seed({ + ...resource('instagram'), + platform: 'instagram', + url: 'https://www.instagram.com/p/abc123/' + }) + // Both old providers had finished metadata; only Instagram needs the new pass. + for (let i = 0; i < 2; i++) store.finish(store.next(0, ['metadata'])!, 'complete') + store.pauseProvider('youtube', 60_000) + store.setPaused(false) + store.close() + const db = new Database(path) + db.prepare("UPDATE work SET version='desktop-2/youtube-page-1/yt-dlp-2026.07.04'").run() + db.close() + store = new LibraryStore(path) + expect(store.next(0, ['metadata'])?.resourceId).toBe('instagram') + expect(store.next(0, ['metadata'])).toBeNull() + expect(store.next(0, ['thumbnail'])).toBeNull() + expect(store.next(60_000, ['thumbnail'])?.resourceId).toBe('youtube') + expect(store.paused).toBe(false) + store.close() + store = new LibraryStore(path) + expect(store.next(0, ['metadata'])?.resourceId).toBe('instagram') + expect( + store.status().counts.find((row) => row.capability === 'metadata' && row.state === 'complete') + ?.count + ).toBe(1) +}) + +it('paginates every resource and safely handles FTS punctuation', () => { + for (let i = 0; i < 85; i++) store.seed(resource(`source-${i}`)) + const ids = new Set( + [0, 40, 80].flatMap((offset) => store.search({ offset }).map((row) => row.id)) + ) + expect(ids.size).toBe(85) + expect(() => store.search({ text: '" ( OR *' })).not.toThrow() +}) + +it('returns bounded collection cards in request order without dropping missing or repeated items', () => { + const saved = resource() + saved.notes = [ + { id: 'note', title: 'Private note', text: 'Full body', url: '', author: 'fixture' } + ] + saved.transcript = { + cues: [{ startMs: 0, durationMs: 1000, text: 'Full transcript' }], + language: 'en', + autoGenerated: false, + source: 'captions', + fetchedAt: 1, + provider: 'fixture', + evidence: 'Raw provider response' + } + store.seed(saved) + const cards = store.cards([saved.id, 'missing-member', saved.id]) + expect(cards.map((card) => card?.id ?? null)).toEqual([saved.id, null, saved.id]) + expect(cards[0]).not.toHaveProperty('notes') + expect(cards[0]).not.toHaveProperty('transcript') + expect(store.cards([])).toEqual([]) + for (const invalid of [null, [''], [42], Array(101).fill(saved.id)]) + expect(() => store.cards(invalid)).toThrow('at most 100 valid resource IDs') +}) + +it('honors provider backoff while permitting local indexing', () => { + store.seed(resource()) + store.pauseProvider('youtube', 60_000) + const first = store.next(0)! + expect(first.capability).toBe('index') + store.finish(first, 'complete') + expect(store.next(100)).toBeNull() + expect(store.next(60_000)?.capability).toBe('metadata') +}) + +it('isolates capability pacing and never shortens a provider rate limit', () => { + store.seed(resource()) + const metadata = store.next(0, ['metadata'])! + store.finish(metadata, 'complete') + store.pauseProvider('youtube:transcript', 120_000) + expect(store.next(0, ['thumbnail'])?.capability).toBe('thumbnail') + expect(store.next(0, ['transcript'])).toBeNull() + store.pauseProvider('youtube', 180_000) + store.pauseProvider('youtube', 60_000) + store.retry() + expect(store.next(120_000, ['transcript'])).toBeNull() + expect(store.next(180_000, ['transcript'])?.capability).toBe('transcript') +}) + +it('requeues dependent gaps when metadata succeeds and resets explicit retry attempts', () => { + store.seed(resource()) + const metadata = store.next(0, ['metadata'])! + store.finish(metadata, 'blocked', 'Could not discover the video') + const thumbnail = store.next(0, ['thumbnail'])! + store.finish(thumbnail, 'unavailable', 'No metadata') + const transcript = store.next(0, ['transcript'])! + store.finish(transcript, 'blocked', 'No tracks') + store.retry(resource().id) + const retry = store.next(0, ['metadata'])! + expect(retry.attempts).toBe(1) + store.finish(retry, 'complete') + expect(store.next(0, ['thumbnail'])?.state).toBe('running') + expect(store.next(0, ['transcript'])?.state).toBe('running') +}) + +it('refuses future storage without changing its version', () => { + store.close() + const db = new Database(path) + db.pragma('user_version=99') + db.close() + expect(() => new LibraryStore(path)).toThrow('preserved') + const check = new Database(path, { readonly: true }) + expect(check.pragma('user_version', { simple: true })).toBe(99) + check.close() +}) + +it('refuses a missing queue table instead of silently rebuilding it', () => { + store.close() + const db = new Database(path) + db.exec('DROP TABLE attempts') + db.close() + expect(() => new LibraryStore(path)).toThrow('preserved') +}) + +it('indexes authored notes independently of fetched text and refreshes removed notes', () => { + store.seed(resource()) + const notes = [ + { + id: 'garden-note', + title: 'Why I saved it', + text: 'my personal copperbridge observation', + url: 'https://example.com/note', + author: 'fixture' + } + ] + store.replaceSourceNotes(new Map([['video-one', notes]])) + expect(store.search({ text: 'copperbridge' })[0].id).toBe('video-one') + expect(store.search({ text: 'copperbridge' })[0]).not.toHaveProperty('notes') + expect(store.get('video-one')?.notes).toEqual(notes) + store.seed(resource()) + expect(store.get('video-one')?.notes).toEqual(notes) + store.replaceSourceNotes(new Map()) + expect(store.search({ text: 'copperbridge' })).toEqual([]) +}) + +it('paces citations by the network provider while retaining archive provenance', () => { + store.seed({ ...resource('citation'), platform: 'openai' }) + store.pauseProvider('youtube:metadata', 60_000) + expect(store.next(0, ['metadata'])).toBeNull() + expect(store.next(60_000, ['metadata'])?.resourceId).toBe('citation') + expect(store.get('citation')?.platform).toBe('openai') + expect(store.get('citation')?.networkPlatform).toBe('youtube') +}) + +it('preserves successful captions and newest Instagram work across the public-page upgrade', () => { + store.seed({ ...resource(), platform: 'instagram', url: 'https://www.instagram.com/p/abc/' }) + store.close() + const db = new Database(path) + db.prepare( + "UPDATE work SET version='desktop-3/youtube-page-1/instagram-embed-1/yt-dlp-2026.07.04',state='complete'" + ).run() + db.prepare( + "INSERT INTO work SELECT resource_id,capability,'desktop-2/youtube-page-1/yt-dlp-2026.07.04',language,'blocked',attempts,next_at,'older failure' FROM work" + ).run() + db.close() + store = new LibraryStore(path) + expect(store.status().counts).toHaveLength(4) + expect(store.status().counts.every((row) => row.state === 'complete')).toBe(true) + expect(store.next(0)).toBeNull() +}) + +it('bounds un-enriched cards without truncating stored source text', () => { + store.seed({ ...resource(), sourceText: 'word '.repeat(10000) }) + expect(store.search({})[0].sourceText).toHaveLength(600) + expect(store.get('video-one')?.sourceText).toHaveLength(50000) +}) + +it('prioritizes a selected old source without bypassing provider backoff', () => { + store.seed({ ...resource('old'), addedAt: 1 }) + store.seed({ ...resource('new'), addedAt: 2 }) + store.retry('old') + store.pauseProvider('youtube', 60_000) + expect(store.next(0, ['metadata'])).toBeNull() + expect(store.next(60_000, ['metadata'])?.resourceId).toBe('old') +}) + +it('indexes freshly enriched passages ahead of a bulk import backlog', () => { + for (let i = 0; i < 200; i++) store.seed(resource(`bulk-${i}`)) + store.seed(resource('fresh')) + store.reindex('fresh') + expect(store.next(Date.now(), ['index'])?.resourceId).toBe('fresh') +}) + +it('projects every web link for the graph without source bodies or caption payloads', () => { + for (let index = 0; index < 1100; index++) store.put(resource(`graph-${index}`)) + store.put({ ...resource('local'), url: 'xnet://conversation/local', kind: 'conversation' }) + store.put({ + ...resource('tagged'), + sourceText: 'An explicit #learning tag', + metadata: { + title: 'Fetched title', + author: 'Fetched author', + description: 'More #topics', + fields: {}, + provider: 'test', + fetchedAt: 1, + evidence: { large: 'private evidence' } + } + }) + const rows = store.graphResources() + expect(rows).toHaveLength(1101) + expect(rows.find((row) => row.id === 'local')).toBeUndefined() + expect(rows.find((row) => row.id === 'tagged')).toEqual({ + id: 'tagged', + title: 'Fetched title', + url: resource().url, + platform: 'youtube', + provider: 'youtube', + author: 'Fetched author', + hashtags: ['learning', 'topics'] + }) + expect(JSON.stringify(rows)).not.toContain('private evidence') +}) + +it('bounds active hosts without skipping ready work from other providers', () => { + store.seed(resource('video')) + store.seed({ ...resource('repo'), url: 'https://github.com/example/repo', platform: 'github' }) + expect(store.next(0, ['metadata'], ['youtube'])?.resourceId).toBe('repo') + expect(store.next(0, ['metadata'], ['youtube'])).toBeNull() + expect(store.next(0, ['metadata'])?.resourceId).toBe('video') +}) + +it('migrates archive-wide pacing to hosts and queues richer page extraction only once', () => { + store.seed({ + ...resource('web'), + platform: 'openai', + url: 'https://example.org/essay', + metadata: { + title: 'Preview', + description: 'Short preview', + fields: {}, + provider: 'public-page/1', + fetchedAt: 1, + evidence: {} + } + }) + store.finish(store.next(0, ['metadata'])!, 'partial') + store.close() + const db = new Database(path) + db.prepare("DELETE FROM settings WHERE key IN ('queue-hosts-v1','page-text-v2')").run() + db.prepare("UPDATE resource_providers SET platform='openai'").run() + db.close() + store = new LibraryStore(path) + store.pauseProvider('web:example.org', 5000) + expect(store.next(0, ['metadata'])).toBeNull() + const job = store.next(5000, ['metadata'])! + expect(job.resourceId).toBe('web') + expect(store.get('web')?.metadata?.description).toBe('Short preview') + store.finish(job, 'complete') + store.close() + store = new LibraryStore(path) + expect(store.next(6000, ['metadata'])).toBeNull() +}) + +it('reconciles existing passage ownership and removes replaced text from search', () => { + store.seed({ ...resource(), sourceText: 'oldsearchphrase' }) + store.close() + const db = new Database(path) + db.exec('DROP TABLE search_rows') + db.close() + store = new LibraryStore(path) + store.index({ ...resource(), sourceText: 'newsearchphrase' }) + expect(store.search({ text: 'oldsearchphrase' })).toEqual([]) + expect(store.search({ text: 'newsearchphrase' })).toHaveLength(1) + store.close() + store = new LibraryStore(path) + store.index({ ...resource(), sourceText: 'finalsearchphrase' }) + expect(store.search({ text: 'newsearchphrase' })).toEqual([]) + expect(store.search({ text: 'finalsearchphrase' })).toHaveLength(1) +}) diff --git a/apps/electron/src/library/store.ts b/apps/electron/src/library/store.ts new file mode 100644 index 000000000..bc8033cba --- /dev/null +++ b/apps/electron/src/library/store.ts @@ -0,0 +1,537 @@ +import type { CaptureIntent } from './capture' +import type { GraphResource } from './graph' +import type { + Capability, + LibraryJob, + LibraryResource, + LibrarySearchResult, + LibraryStatus, + WorkState +} from './types' +import Database from 'better-sqlite3' +import { requireCompatibleDatabase } from '../storage/compatibility' +import { validateCapture } from './capture' +import { hashtagsIn } from './graph' +import { queueProvider } from './source' +import { CAPABILITIES, LIBRARY_PROVIDER_VERSION } from './types' + +export const inspectLibraryDatabase = (path: string): void => + requireCompatibleDatabase(path, 'library') + +type JobRow = { + resource_id: string + capability: Capability + version: string + language: string + state: WorkState + attempts: number + next_at: number + reason: string | null +} +const jobFor = (row: JobRow): LibraryJob => ({ + resourceId: row.resource_id, + capability: row.capability, + version: row.version, + language: row.language, + state: row.state, + attempts: row.attempts, + nextAt: row.next_at, + reason: row.reason +}) + +const cardFor = (resource: LibraryResource): LibrarySearchResult => { + const { transcript, metadata, notes, ...source } = resource + void transcript + void notes + if (!metadata) return { ...source, sourceText: source.sourceText.slice(0, 600) } + const { evidence, tracks, ...summary } = metadata + void evidence + void tracks + return { + ...source, + sourceText: source.sourceText.slice(0, 600), + metadata: { ...summary, description: summary.description?.slice(0, 600) } + } +} + +/** Device-local work queue; source facts and shared projections stay in NodeStore. */ +export class LibraryStore { + private db: Database.Database + constructor(path: string) { + inspectLibraryDatabase(path) + this.db = new Database(path) + this.db.pragma('journal_mode = WAL') + this.db.pragma('synchronous = FULL') + this.db.exec(` + CREATE TABLE IF NOT EXISTS resources(id TEXT PRIMARY KEY, url TEXT NOT NULL, platform TEXT NOT NULL, title TEXT NOT NULL, payload TEXT NOT NULL, added_at INTEGER NOT NULL); + CREATE INDEX IF NOT EXISTS resource_url ON resources(url); + CREATE INDEX IF NOT EXISTS resource_added ON resources(added_at DESC,id); + CREATE INDEX IF NOT EXISTS resource_platform_added ON resources(platform,added_at DESC,id); + CREATE TABLE IF NOT EXISTS work(resource_id TEXT NOT NULL, capability TEXT NOT NULL, version TEXT NOT NULL, language TEXT NOT NULL, state TEXT NOT NULL, attempts INTEGER NOT NULL DEFAULT 0, next_at INTEGER NOT NULL DEFAULT 0, reason TEXT, PRIMARY KEY(resource_id, capability, version, language)); + CREATE INDEX IF NOT EXISTS work_due ON work(state, next_at); + CREATE INDEX IF NOT EXISTS work_pending_capability ON work(version,capability,next_at) WHERE state IN ('queued','retry'); + CREATE TABLE IF NOT EXISTS settings(key TEXT PRIMARY KEY, value TEXT NOT NULL); + CREATE TABLE IF NOT EXISTS provider_pause(platform TEXT PRIMARY KEY, until_ms INTEGER NOT NULL); + CREATE TABLE IF NOT EXISTS resource_providers(resource_id TEXT PRIMARY KEY, platform TEXT NOT NULL); + INSERT OR IGNORE INTO resource_providers SELECT id,COALESCE(json_extract(payload,'$.networkPlatform'),platform) FROM resources; + CREATE TABLE IF NOT EXISTS attempts(resource_id TEXT NOT NULL, capability TEXT NOT NULL, version TEXT NOT NULL, at_ms INTEGER NOT NULL, state TEXT NOT NULL, reason TEXT); + CREATE VIRTUAL TABLE IF NOT EXISTS search USING fts5(resource_id UNINDEXED, title, body, start_ms UNINDEXED, tokenize='unicode61'); + CREATE TABLE IF NOT EXISTS search_rows(row_id INTEGER PRIMARY KEY, resource_id TEXT NOT NULL); + CREATE INDEX IF NOT EXISTS search_rows_resource ON search_rows(resource_id); + PRAGMA user_version = 1; + `) + // FTS's UNINDEXED resource_id otherwise scans every saved passage on each update. + // Reconcile on open so an older app can still write this disposable search cache. + this.db.transaction(() => { + this.db.exec(` + DELETE FROM search_rows WHERE row_id NOT IN (SELECT rowid FROM search); + INSERT OR REPLACE INTO search_rows SELECT rowid,resource_id FROM search; + `) + })() + if (!this.db.prepare("SELECT 1 FROM settings WHERE key='queue-hosts-v1'").get()) { + this.db.transaction(() => { + const update = this.db.prepare( + 'UPDATE resource_providers SET platform=? WHERE resource_id=?' + ) + for (let offset = 0; ; offset += 500) { + const rows = this.db + .prepare('SELECT id,payload FROM resources ORDER BY id LIMIT 500 OFFSET ?') + .all(offset) as { id: string; payload: string }[] + for (const source of rows) + update.run(queueProvider(JSON.parse(source.payload) as LibraryResource), source.id) + if (rows.length < 500) break + } + this.db.prepare("INSERT INTO settings VALUES ('queue-hosts-v1','true')").run() + })() + } + // A new provider version gets a complete, resumable pass over existing resources. + // Retain older jobs as evidence; current successful work is never reset on restart. + this.db.transaction(() => { + // Preserve results from the newest compatible pass. Only newly supported + // providers and unresolved captions need another attempt in desktop-4. + for (const previous of [ + 'desktop-3/youtube-page-1/instagram-embed-1/yt-dlp-2026.07.04', + 'desktop-2/youtube-page-1/yt-dlp-2026.07.04' + ]) { + const unchanged = previous.startsWith('desktop-3') + ? "('youtube','instagram')" + : "('youtube')" + this.db + .prepare( + ` + INSERT OR IGNORE INTO work(resource_id,capability,version,language,state,attempts,next_at,reason) + SELECT w.resource_id,w.capability,?,w.language,w.state,w.attempts,w.next_at,w.reason + FROM work w JOIN resources r ON r.id=w.resource_id + WHERE w.version=? AND ( + w.capability='index' + OR (w.capability='metadata' AND r.platform IN ${unchanged}) + OR (w.capability='thumbnail' AND (w.state='complete' OR r.platform IN ${unchanged})) + OR (w.capability='transcript' AND w.state='complete')) + ` + ) + .run(LIBRARY_PROVIDER_VERSION, previous) + } + for (const capability of CAPABILITIES) + this.db + .prepare( + "INSERT OR IGNORE INTO work(resource_id,capability,version,language,state) SELECT id,?,?,'preferred','queued' FROM resources" + ) + .run(capability, LIBRARY_PROVIDER_VERSION) + })() + if (!this.db.prepare("SELECT 1 FROM settings WHERE key='page-text-v2'").get()) { + this.db.transaction(() => { + this.db + .prepare( + `UPDATE work SET state='queued',attempts=0,next_at=0,reason=NULL + WHERE version=? AND capability='metadata' AND state IN ('partial','complete') + AND resource_id IN (SELECT id FROM resources WHERE json_extract(payload,'$.metadata.provider') IN ('github-page/1','public-page/1'))` + ) + .run(LIBRARY_PROVIDER_VERSION) + this.db.prepare("INSERT INTO settings VALUES ('page-text-v2','true')").run() + })() + } + this.db + .prepare( + "UPDATE work SET state = 'queued', attempts=MAX(0,attempts-1), reason = 'Interrupted; ready to resume' WHERE state = 'running'" + ) + .run() + this.db + .prepare( + `UPDATE work SET next_at=-1 WHERE version=? + AND capability IN ('thumbnail','transcript') AND state='queued' AND next_at=0 + AND EXISTS (SELECT 1 FROM work m WHERE m.resource_id=work.resource_id + AND m.version=work.version AND m.capability='metadata' AND m.state IN ('complete','partial'))` + ) + .run(LIBRARY_PROVIDER_VERSION) + } + close(): void { + if (this.db.open) this.db.close() + } + private readCapture(raw: string): CaptureIntent { + const value = JSON.parse(raw) as CaptureIntent + if ( + value.version !== 1 || + !value.authorDID || + !value.result?.pageId || + !value.result.resourceId || + !value.resource?.url || + typeof value.completed !== 'boolean' || + !Array.isArray(value.document) || + value.document.some((byte) => !Number.isInteger(byte) || byte < 0 || byte > 255) + ) + throw new Error('Capture recovery record is unreadable; it was preserved.') + validateCapture(value.input) + return value + } + captureIntent(requestId: string): CaptureIntent | null { + const row = this.db + .prepare('SELECT value FROM settings WHERE key=?') + .get(`capture:${requestId}`) as { value: string } | undefined + return row ? this.readCapture(row.value) : null + } + pendingCaptures(): CaptureIntent[] { + const rows = this.db.prepare("SELECT value FROM settings WHERE key LIKE 'capture:%'").all() as { + value: string + }[] + return rows.map((row) => this.readCapture(row.value)).filter((intent) => !intent.completed) + } + saveCaptureIntent(intent: CaptureIntent): void { + const value = JSON.stringify(intent) + this.readCapture(value) + this.db + .prepare('INSERT OR REPLACE INTO settings(key,value) VALUES (?,?)') + .run(`capture:${intent.input.requestId}`, value) + } + get paused(): boolean { + return ( + ( + this.db.prepare("SELECT value FROM settings WHERE key='paused'").get() as + | { value: string } + | undefined + )?.value !== 'false' + ) + } + setPaused(value: boolean): void { + this.db.prepare("INSERT OR REPLACE INTO settings VALUES ('paused', ?)").run(String(value)) + } + get(id: string): LibraryResource | null { + const row = this.db.prepare('SELECT payload FROM resources WHERE id=?').get(id) as + | { payload: string } + | undefined + return row ? (JSON.parse(row.payload) as LibraryResource) : null + } + byUrl(url: string): LibraryResource | null { + const row = this.db.prepare('SELECT payload FROM resources WHERE url=? LIMIT 1').get(url) as + | { payload: string } + | undefined + return row ? (JSON.parse(row.payload) as LibraryResource) : null + } + cards(ids: unknown): (LibrarySearchResult | null)[] { + if ( + !Array.isArray(ids) || + ids.length > 100 || + ids.some((id) => typeof id !== 'string' || !id || id.length > 500) + ) + throw new Error('Library cards require at most 100 valid resource IDs.') + return ids.map((id: string) => { + const resource = this.get(id) + return resource ? cardFor(resource) : null + }) + } + graphResources(): GraphResource[] { + // Read text only while extracting explicit hashtags; never send transcripts, + // source bodies or provider evidence with the overview's compact nodes. + const rows = this.db + .prepare( + `SELECT id,url,title,platform, + COALESCE(json_extract(payload,'$.networkPlatform'),platform) AS provider, + COALESCE(NULLIF(json_extract(payload,'$.metadata.author'),''),json_extract(payload,'$.actor'),'') AS author, + COALESCE(json_extract(payload,'$.sourceText'),'') || char(10) || + COALESCE(json_extract(payload,'$.metadata.description'),'') AS body + FROM resources WHERE lower(url) LIKE 'https://%' OR lower(url) LIKE 'http://%' ORDER BY id` + ) + .iterate() + return Array.from(rows, (value) => { + const { body, ...resource } = value as Omit & { body: string } + return { ...resource, hashtags: hashtagsIn(body) } + }) + } + put(resource: LibraryResource): void { + resource = { ...resource, networkPlatform: queueProvider(resource) } + this.db.transaction(() => { + this.db + .prepare( + 'INSERT INTO resources VALUES (@id,@url,@platform,@title,@payload,@addedAt) ON CONFLICT(id) DO UPDATE SET url=excluded.url, platform=excluded.platform, title=excluded.title, payload=excluded.payload' + ) + .run({ + ...resource, + title: resource.metadata?.title || resource.title, + payload: JSON.stringify(resource) + }) + this.db + .prepare( + 'INSERT INTO resource_providers VALUES (?,?) ON CONFLICT(resource_id) DO UPDATE SET platform=excluded.platform' + ) + .run(resource.id, resource.networkPlatform) + })() + } + seed(resource: LibraryResource): void { + this.db.transaction(() => { + const previous = this.get(resource.id) + const next = previous + ? { + ...previous, + ...resource, + kind: resource.kind, + metadata: previous.metadata, + thumbnail: previous.thumbnail, + transcript: previous.transcript, + notes: previous.notes, + addedAt: previous.addedAt + } + : resource + if (!previous || JSON.stringify(previous) !== JSON.stringify(next)) this.put(next) + if ( + !previous || + previous.sourceText !== resource.sourceText || + previous.title !== resource.title + ) + this.index(this.get(resource.id)!) + for (const capability of CAPABILITIES) this.enqueue(resource.id, capability) + if (previous?.kind && !resource.kind) + this.db + .prepare( + "UPDATE work SET state='queued',reason=NULL WHERE resource_id=? AND version=? AND capability!='index' AND state='not-applicable'" + ) + .run(resource.id, LIBRARY_PROVIDER_VERSION) + if (resource.kind) + this.db + .prepare( + "UPDATE work SET state='not-applicable',reason='Text is imported locally; there is no public source to fetch.' WHERE resource_id=? AND version=? AND capability!='index'" + ) + .run(resource.id, LIBRARY_PROVIDER_VERSION) + })() + } + enqueue(id: string, capability: Capability, language = 'preferred'): void { + this.db + .prepare( + "INSERT OR IGNORE INTO work(resource_id,capability,version,language,state) VALUES (?,?,?,?,'queued')" + ) + .run(id, capability, LIBRARY_PROVIDER_VERSION, language) + } + replaceSourceNotes(notes: Map>): void { + this.db.transaction(() => { + const rows = this.db.prepare('SELECT id FROM resources').all() as { id: string }[] + for (const row of rows) { + const resource = this.get(row.id)! + const next = notes.get(resource.id) ?? [] + if (JSON.stringify(resource.notes ?? []) === JSON.stringify(next)) continue + const updated = { ...resource, notes: next } + this.put(updated) + this.index(updated) + } + })() + } + reindex(id: string): void { + this.db + .prepare( + "UPDATE work SET state='queued',next_at=?,reason=NULL WHERE resource_id=? AND capability='index' AND version=?" + ) + .run(-Date.now(), id, LIBRARY_PROVIDER_VERSION) + } + next( + now: number, + capabilities: readonly Capability[] = CAPABILITIES, + excludedProviders: readonly string[] = [] + ): LibraryJob | null { + if (!capabilities.length) return null + let row: JobRow | undefined + // The index supplies queue order directly; never sort the entire import backlog per claim. + const ordered = capabilities.includes('index') + ? ['index' as const, ...capabilities.filter((capability) => capability !== 'index')] + : capabilities + for (const capability of ordered) { + row = + capability === 'index' + ? (this.db + .prepare( + "SELECT * FROM work WHERE version=? AND capability='index' AND state IN ('queued','retry') AND next_at<=? ORDER BY next_at LIMIT 1" + ) + .get(LIBRARY_PROVIDER_VERSION, now) as JobRow | undefined) + : (this.db + .prepare( + `SELECT w.* FROM work w INDEXED BY work_pending_capability + JOIN resource_providers rp ON rp.resource_id=w.resource_id + LEFT JOIN provider_pause p ON p.platform=rp.platform + LEFT JOIN provider_pause lane ON lane.platform=rp.platform || ':' || w.capability + WHERE w.version=? AND w.capability=? AND w.state IN ('queued','retry') AND w.next_at<=? + AND (p.until_ms IS NULL OR p.until_ms<=?) AND (lane.until_ms IS NULL OR lane.until_ms<=?) + ${excludedProviders.length ? `AND rp.platform NOT IN (${excludedProviders.map(() => '?').join(',')})` : ''} + AND (w.capability='metadata' OR NOT EXISTS (SELECT 1 FROM work m WHERE m.resource_id=w.resource_id AND m.capability='metadata' AND m.version=w.version AND m.state IN ('queued','running','retry'))) + ORDER BY w.next_at LIMIT 1` + ) + .get(LIBRARY_PROVIDER_VERSION, capability, now, now, now, ...excludedProviders) as + | JobRow + | undefined) + if (row) break + } + if (!row) return null + const job = jobFor(row) + this.db + .prepare( + "UPDATE work SET state='running', attempts=attempts+1 WHERE resource_id=? AND capability=? AND version=? AND language=?" + ) + .run(job.resourceId, job.capability, job.version, job.language) + return { ...job, state: 'running', attempts: job.attempts + 1 } + } + finish(job: LibraryJob, state: WorkState, reason: string | null = null, nextAt = 0): void { + this.db.transaction(() => { + this.db + .prepare( + 'UPDATE work SET state=?,reason=?,next_at=? WHERE resource_id=? AND capability=? AND version=? AND language=?' + ) + .run(state, reason, nextAt, job.resourceId, job.capability, job.version, job.language) + if (state === 'queued' && job.state === 'running') + this.db + .prepare( + 'UPDATE work SET attempts=MAX(0,attempts-1) WHERE resource_id=? AND capability=? AND version=? AND language=?' + ) + .run(job.resourceId, job.capability, job.version, job.language) + if (job.capability === 'metadata' && (state === 'complete' || state === 'partial')) + this.db + .prepare( + "UPDATE work SET state='queued',attempts=0,next_at=MIN(next_at,-1),reason=NULL WHERE resource_id=? AND version=? AND capability IN ('thumbnail','transcript') AND state IN ('queued','blocked','unavailable')" + ) + .run(job.resourceId, job.version) + this.db + .prepare('INSERT INTO attempts VALUES (?,?,?,?,?,?)') + .run(job.resourceId, job.capability, job.version, Date.now(), state, reason) + })() + } + pauseProvider(platform: string, until: number): void { + this.db + .prepare( + 'INSERT INTO provider_pause VALUES (?,?) ON CONFLICT(platform) DO UPDATE SET until_ms=MAX(provider_pause.until_ms,excluded.until_ms)' + ) + .run(platform, until) + } + retry(id?: string): void { + if (id) + this.db + .prepare( + "UPDATE work SET state='queued',attempts=0,next_at=?,reason=NULL WHERE resource_id=? AND version=? AND state IN ('queued','blocked','retry','partial','unavailable')" + ) + .run(-Date.now(), id, LIBRARY_PROVIDER_VERSION) + else + this.db + .prepare( + "UPDATE work SET state='queued',attempts=0,next_at=0,reason=NULL WHERE version=? AND state IN ('blocked','retry','partial')" + ) + .run(LIBRARY_PROVIDER_VERSION) + } + index(resource: LibraryResource): void { + this.db.transaction(() => { + this.db + .prepare( + 'DELETE FROM search WHERE rowid IN (SELECT row_id FROM search_rows WHERE resource_id=?)' + ) + .run(resource.id) + this.db.prepare('DELETE FROM search_rows WHERE resource_id=?').run(resource.id) + const insert = this.db.prepare( + 'INSERT INTO search(resource_id,title,body,start_ms) VALUES (?,?,?,?)' + ) + const owner = this.db.prepare('INSERT INTO search_rows(row_id,resource_id) VALUES (?,?)') + const passage = (title: string, body: string, startMs: number | null) => { + const result = insert.run(resource.id, title, body, startMs) + owner.run(result.lastInsertRowid, resource.id) + } + const title = resource.metadata?.title || resource.title + passage( + title, + [ + resource.url, + resource.actor, + resource.sourceText, + resource.metadata?.description, + ...(resource.notes ?? []).map((note) => `${note.title}\n${note.text}\n${note.url}`) + ] + .filter(Boolean) + .join('\n'), + null + ) + for (const cue of resource.transcript?.cues ?? []) passage(title, cue.text, cue.startMs) + })() + } + search(options: { + text?: string + platform?: string + offset?: number + limit?: number + }): LibrarySearchResult[] { + const limit = Math.min(100, Math.max(1, options.limit ?? 40)) + const offset = Math.max(0, options.offset ?? 0) + const platform = options.platform || '' + const words = options.text?.trim().split(/\s+/).filter(Boolean) ?? [] + if (words.length) { + const query = words.map((word) => `"${word.replaceAll('"', '""')}"`).join(' AND ') + const rows = this.db + .prepare( + `SELECT r.payload, snippet(search,2,'','', ' … ',32) AS snippet, search.start_ms FROM search JOIN resources r ON r.id=search.resource_id WHERE search MATCH ? AND (?='' OR r.platform=?) ORDER BY rank LIMIT ? OFFSET ?` + ) + .all(query, platform, platform, limit, offset) as { + payload: string + snippet: string + start_ms: number | null + }[] + return rows.map((row) => ({ + ...cardFor(JSON.parse(row.payload) as LibraryResource), + snippet: row.snippet, + ...(row.start_ms !== null ? { startMs: row.start_ms } : {}) + })) + } + return ( + this.db + .prepare( + `SELECT payload FROM resources ${platform ? 'WHERE platform=?' : ''} ORDER BY added_at DESC,id LIMIT ? OFFSET ?` + ) + .all(...(platform ? [platform] : []), limit, offset) as { payload: string }[] + ).map((row) => cardFor(JSON.parse(row.payload) as LibraryResource)) + } + status(): LibraryStatus { + const counts = this.db + .prepare( + 'SELECT capability,state,COUNT(*) AS count FROM work WHERE version=? GROUP BY capability,state' + ) + .all(LIBRARY_PROVIDER_VERSION) as LibraryStatus['counts'] + const recent = this.db + .prepare( + "SELECT w.*, r.title FROM work w JOIN resources r ON r.id=w.resource_id WHERE w.version=? AND w.state IN ('blocked','retry','partial','unavailable') ORDER BY w.next_at DESC LIMIT 20" + ) + .all(LIBRARY_PROVIDER_VERSION) as (JobRow & { title: string })[] + const running = this.db + .prepare( + "SELECT w.*,r.title FROM work w JOIN resources r ON r.id=w.resource_id WHERE w.version=? AND w.state='running' LIMIT 8" + ) + .all(LIBRARY_PROVIDER_VERSION) as (JobRow & { title: string })[] + const due = this.db + .prepare( + `SELECT MIN(MAX(w.next_at,COALESCE(p.until_ms,0),COALESCE(lane.until_ms,0))) AS at + FROM work w JOIN resources r ON r.id=w.resource_id + LEFT JOIN resource_providers rp ON rp.resource_id=r.id + LEFT JOIN provider_pause p ON p.platform=COALESCE(rp.platform,r.platform) + LEFT JOIN provider_pause lane ON lane.platform=COALESCE(rp.platform,r.platform) || ':' || w.capability + WHERE w.version=? AND w.state IN ('queued','retry') AND w.capability!='index' + AND (w.capability='metadata' OR NOT EXISTS (SELECT 1 FROM work m WHERE m.resource_id=w.resource_id AND m.capability='metadata' AND m.version=w.version AND m.state IN ('queued','running','retry')))` + ) + .get(LIBRARY_PROVIDER_VERSION) as { at: number | null } + return { + paused: this.paused, + running: running.map((row) => ({ ...jobFor(row), title: row.title })), + nextAt: due.at, + resources: (this.db.prepare('SELECT COUNT(*) AS n FROM resources').get() as { n: number }).n, + counts, + recent: recent.map((row) => ({ ...jobFor(row), title: row.title })), + providerVersion: LIBRARY_PROVIDER_VERSION + } + } +} diff --git a/apps/electron/src/library/transcripts.test.ts b/apps/electron/src/library/transcripts.test.ts new file mode 100644 index 000000000..e62511d26 --- /dev/null +++ b/apps/electron/src/library/transcripts.test.ts @@ -0,0 +1,105 @@ +import type { LibraryResource } from './types' +import { beforeEach, expect, it, vi } from 'vitest' +import { LibraryProviderError } from './provider-error' +import { fetchLibraryTranscript } from './transcripts' + +const stubs = vi.hoisted(() => ({ fetch: vi.fn(), helper: vi.fn() })) +vi.mock('./providers', async (original) => ({ + ...(await original()), + fetchPublic: stubs.fetch, + fetchExtractorMetadata: stubs.helper +})) +const track = { + url: 'https://example.com/old', + language: 'en', + format: 'vtt' as const, + autoGenerated: false +} +const resource: LibraryResource = { + id: 'video', + platform: 'youtube', + platformContentId: 'abcdefghijk', + url: 'https://youtu.be/abcdefghijk', + title: 'Video', + sourceText: '', + actor: '', + privacy: 'private', + addedAt: 0, + metadata: { + provider: 'youtube-page/1', + fetchedAt: 1, + evidence: {}, + tracks: [track], + fields: { captions: { state: 'complete' } } + } +} +const vtt = 'WEBVTT\n\n00:01.000 --> 00:02.500\nA spoken caption.\n' +beforeEach(() => { + stubs.fetch.mockReset() + stubs.helper.mockReset() +}) + +it('refreshes inaccessible YouTube captions through the helper while preserving title metadata', async () => { + stubs.fetch.mockResolvedValueOnce(new Response('')).mockResolvedValueOnce(new Response(vtt)) + stubs.helper.mockResolvedValue({ + provider: 'yt-dlp/fixture', + tracks: [{ ...track, url: 'https://example.com/fresh', autoGenerated: true }], + fields: { captions: { state: 'complete' } } + }) + const result = await fetchLibraryTranscript(resource, new AbortController().signal) + expect(result).toMatchObject({ + status: 'available', + value: { + provider: 'yt-dlp/fixture', + track: { autoGenerated: true }, + cues: [{ startMs: 1000, durationMs: 1500, text: 'A spoken caption.' }] + } + }) + expect(resource.metadata?.provider).toBe('youtube-page/1') + expect(stubs.helper).toHaveBeenCalledOnce() +}) + +it('retrieves TikTok and ordinary-page tracks without a helper', async () => { + for (const url of ['https://www.tiktokv.com/share/video/12345', 'https://example.com/lecture']) { + stubs.fetch.mockResolvedValue(new Response(vtt)) + const result = await fetchLibraryTranscript( + { ...resource, platform: 'generic', url }, + new AbortController().signal + ) + expect(result.status).toBe('available') + } + expect(stubs.helper).not.toHaveBeenCalled() +}) + +it('does not hide empty caption bodies or retry a provider rate limit through the helper', async () => { + stubs.fetch.mockResolvedValue(new Response('')) + stubs.helper.mockResolvedValue({ + provider: 'helper', + tracks: [], + fields: { captions: { state: 'complete' } } + }) + await expect(fetchLibraryTranscript(resource, new AbortController().signal)).rejects.toThrow( + 'empty caption' + ) + stubs.helper.mockClear() + stubs.fetch.mockRejectedValue(new LibraryProviderError('Rate limited', 'retry', 100, 'provider')) + await expect( + fetchLibraryTranscript(resource, new AbortController().signal) + ).rejects.toMatchObject({ scope: 'provider' }) + expect(stubs.helper).not.toHaveBeenCalled() +}) + +it('tries another advertised format when one caption track is unreadable', async () => { + stubs.fetch + .mockResolvedValueOnce(new Response('not captions')) + .mockResolvedValueOnce(new Response(vtt)) + const result = await fetchLibraryTranscript( + { + ...resource, + metadata: { ...resource.metadata!, tracks: [{ ...track, format: 'json3' }, track] } + }, + new AbortController().signal + ) + expect(result.status).toBe('available') + expect(stubs.helper).not.toHaveBeenCalled() +}) diff --git a/apps/electron/src/library/transcripts.ts b/apps/electron/src/library/transcripts.ts new file mode 100644 index 000000000..1d96a96a8 --- /dev/null +++ b/apps/electron/src/library/transcripts.ts @@ -0,0 +1,96 @@ +import type { CaptionTrack, Cue, LibraryResource } from './types' +import { LibraryProviderError } from './provider-error' +import { chooseTrack, fetchExtractorMetadata, fetchPublic, parseCaptionBody } from './providers' +import { providerResource } from './source' + +type RetrievedTranscript = { track: CaptionTrack; cues: Cue[]; raw: string; provider: string } +type TranscriptResult = + | { status: 'available'; value: RetrievedTranscript } + | { status: 'unavailable'; reason: string } + +async function readTracks( + tracks: readonly CaptionTrack[], + preferred: string, + provider: string, + signal: AbortSignal +): Promise { + const remaining = [...tracks] + let failure: unknown = new LibraryProviderError( + 'No readable caption track was discovered.', + 'unavailable' + ) + for (let attempt = 0; attempt < 4 && remaining.length; attempt++) { + const track = chooseTrack(remaining, preferred)! + remaining.splice(remaining.indexOf(track), 1) + try { + const response = await fetchPublic(track.url, { + signal, + limit: 32 * 1024 * 1024, + timeoutMs: 20_000 + }) + const raw = await response.text() + const cues = raw.trim() ? parseCaptionBody(raw, track.format) : [] + if (!cues.length) + throw new LibraryProviderError('The source returned an empty caption track.', 'blocked') + return { track, raw, cues, provider } + } catch (error) { + if (signal.aborted || (error instanceof LibraryProviderError && error.scope === 'provider')) + throw error + failure = error + } + } + throw failure +} + +/** Caption discovery and retrieval may need a fresh provider response after signed URLs expire. */ +export async function fetchLibraryTranscript( + resource: LibraryResource, + signal: AbortSignal, + helperDirectory?: string +): Promise { + const target = providerResource(resource) + const metadata = resource.metadata + const preferred = metadata?.language || 'en' + const tracks = metadata?.tracks ?? [] + let failure: unknown + if (tracks.length) { + try { + return { + status: 'available', + value: await readTracks(tracks, preferred, metadata!.provider, signal) + } + } catch (error) { + if (signal.aborted || (error instanceof LibraryProviderError && error.scope === 'provider')) + throw error + failure = error + } + } + if (target.platform === 'youtube') { + const fresh = await fetchExtractorMetadata(target, signal, helperDirectory) + if (fresh.tracks?.length) + return { + status: 'available', + value: await readTracks(fresh.tracks, preferred, fresh.provider, signal) + } + if (fresh.fields.captions?.state !== 'complete') + throw new LibraryProviderError( + fresh.fields.captions?.reason ?? 'Caption discovery did not complete.', + 'blocked' + ) + if (failure) throw failure + return { + status: 'unavailable', + reason: 'The source reported no readable caption tracks. Local transcription has not run.' + } + } + if (failure) throw failure + if (metadata?.fields.captions?.state !== 'complete') + throw new LibraryProviderError( + metadata?.fields.captions?.reason ?? 'Caption discovery has not succeeded.', + 'blocked' + ) + return { + status: 'unavailable', + reason: 'No readable caption track was discovered. Local transcription has not run.' + } +} diff --git a/apps/electron/src/library/types.ts b/apps/electron/src/library/types.ts new file mode 100644 index 000000000..da1f43ae3 --- /dev/null +++ b/apps/electron/src/library/types.ts @@ -0,0 +1 @@ +export * from '../shared/library' diff --git a/apps/electron/src/library/youtube.test.ts b/apps/electron/src/library/youtube.test.ts new file mode 100644 index 000000000..54847f240 --- /dev/null +++ b/apps/electron/src/library/youtube.test.ts @@ -0,0 +1,94 @@ +import { expect, it } from 'vitest' +import { parseYouTubePage } from './providers' + +const page = (value: unknown) => + `` +const details = { + videoId: 'abcdefghijk', + title: 'A title with "quotes" and {braces}', + shortDescription: 'Full text '.repeat(3000), + author: 'Original author', + lengthSeconds: '171', + thumbnail: { + thumbnails: [ + { url: 'https://i.ytimg.com/small.jpg', width: 120 }, + { url: 'https://i.ytimg.com/large.jpg', width: 1280 } + ] + } +} + +it('reads full public metadata and caption tracks without evaluating scripts or needing a helper', () => { + const result = parseYouTubePage( + page({ + videoDetails: details, + playabilityStatus: { status: 'OK' }, + captions: { + playerCaptionsTracklistRenderer: { + captionTracks: [ + { + baseUrl: 'https://www.youtube.com/api/timedtext?v=abcdefghijk&sig=original', + languageCode: 'de', + kind: 'asr' + } + ] + } + } + }), + 'abcdefghijk' + ) + expect(result.title).toBe(details.title) + expect(result.description).toBe(details.shortDescription) + expect(result.thumbnailUrl).toBe('https://i.ytimg.com/large.jpg') + expect(result.durationSeconds).toBe(171) + expect(result.tracks?.[0]).toMatchObject({ language: 'de', autoGenerated: true, format: 'json3' }) + expect(new URL(result.tracks![0].url).searchParams.get('sig')).toBe('original') + expect(new URL(result.tracks![0].url).searchParams.get('fmt')).toBe('json3') +}) + +it('distinguishes a known empty caption inventory from restricted playback', () => { + const normal = parseYouTubePage( + page({ videoDetails: details, playabilityStatus: { status: 'OK' } }), + 'abcdefghijk' + ) + expect(normal.tracks).toEqual([]) + expect(normal.fields.captions.state).toBe('complete') + const restricted = parseYouTubePage( + page({ + videoDetails: details, + playabilityStatus: { status: 'LOGIN_REQUIRED', reason: 'Sign in to confirm your age' } + }), + 'abcdefghijk' + ) + expect(restricted.title).toBe(details.title) + expect(restricted.fields.captions).toEqual({ + state: 'partial', + reason: 'Sign in to confirm your age' + }) +}) + +it('rejects wrong videos, deleted videos, malformed JSON, and truncated caption inventories', () => { + expect(() => parseYouTubePage(page({ videoDetails: details }), 'differentID')).toThrow( + 'requested video' + ) + expect(() => + parseYouTubePage( + page({ playabilityStatus: { status: 'ERROR', reason: 'Video has been removed' } }), + 'abcdefghijk' + ) + ).toThrow('removed') + expect(() => + parseYouTubePage('', 'abcdefghijk') + ).toThrow() + expect(() => + parseYouTubePage( + page({ + videoDetails: details, + captions: { playerCaptionsTracklistRenderer: { captionTracks: [{ languageCode: 'en' }] } } + }), + 'abcdefghijk' + ) + ).toThrow('inventory') + expect(() => parseYouTubePage('Consent or login', 'abcdefghijk')).toThrow( + 'did not return' + ) +}) diff --git a/apps/electron/src/main/data-process-manager.ts b/apps/electron/src/main/data-process-manager.ts index 76b7fdc04..dc0367885 100644 --- a/apps/electron/src/main/data-process-manager.ts +++ b/apps/electron/src/main/data-process-manager.ts @@ -90,6 +90,7 @@ export async function spawnDataProcess(dbPath: string): Promise { } log('Spawning data process...') + isShuttingDown = false return new Promise((resolve, reject) => { try { @@ -173,8 +174,9 @@ export async function spawnDataProcess(dbPath: string): Promise { isReady = true log('Data process ready') - // Initialize with database path - sendRequest('init', { dbPath }) + // Startup inspects both databases and reconciles the Library search cache. + // Large libraries need minutes; ordinary requests keep their shorter deadline. + sendRequest('init', { dbPath }, 600_000) .then(() => { log('Data process initialized') resolve() @@ -201,17 +203,26 @@ export async function stopDataProcess(): Promise { try { await sendRequest('shutdown', {}, 5000) - } catch { - log('Shutdown request failed, killing process') + } catch (error) { + isShuttingDown = false + throw error } - if (dataProcess) { - dataProcess.kill() - dataProcess = null + const stopping = dataProcess + if (stopping) { + await new Promise((resolve, reject) => { + const timeout = setTimeout( + () => reject(new Error('Data process did not exit after saving.')), + 5000 + ) + stopping.once('exit', () => { + clearTimeout(timeout) + resolve() + }) + stopping.kill() + }) } - isReady = false - isShuttingDown = false } /** @@ -240,7 +251,8 @@ async function sendRequest( timeout: timeoutHandle }) - dataProcess!.postMessage({ type, requestId, ...payload }) + // Transport identity must remain authoritative even when a payload has its own retry key. + dataProcess!.postMessage({ ...payload, type, requestId }) }) } @@ -510,6 +522,11 @@ export function setupDataProcessIPC(getMainWindow: () => BrowserWindow | null): }) // Change log operations + ipcMain.handle('xnet:nodes:applyNodeBatch', async (_event, opts: { input: unknown }) => { + const result = (await sendRequest('nodes:applyNodeBatch', opts)) as { result: unknown } + return result.result + }) + ipcMain.handle('xnet:nodes:appendChange', async (_event, opts: { change: unknown }) => { await sendRequest('nodes:appendChange', opts) }) diff --git a/apps/electron/src/main/data-process-startup.test.ts b/apps/electron/src/main/data-process-startup.test.ts new file mode 100644 index 000000000..dad0e3322 --- /dev/null +++ b/apps/electron/src/main/data-process-startup.test.ts @@ -0,0 +1,79 @@ +import { afterEach, beforeEach, expect, it, vi } from 'vitest' + +const child = vi.hoisted(() => ({ + handlers: new Map void>(), + postMessage: vi.fn(), + on: vi.fn((event: string, handler: (value: unknown) => void) => { + child.handlers.set(event, handler) + }) +})) +vi.mock('electron', () => ({ + utilityProcess: { fork: () => child }, + app: { getAppPath: () => '/test/app' }, + ipcMain: { handle: vi.fn() }, + MessageChannelMain: vi.fn() +})) + +beforeEach(() => { + vi.resetModules() + vi.useFakeTimers() + child.handlers.clear() + child.postMessage.mockClear() +}) +afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() +}) + +const ready = () => child.handlers.get('message')!({ type: 'ready' }) +const respond = (error?: string) => + child.handlers.get('message')!({ + type: 'response', + requestId: child.postMessage.mock.lastCall![0].requestId, + error + }) + +it('waits for a large workspace inspection beyond the ordinary request deadline', async () => { + const manager = await import('./data-process-manager') + let state = 'waiting' + const opened = manager.spawnDataProcess('/test/data.db').then( + () => { + state = 'ready' + }, + (error: Error) => { + state = error.message + } + ) + ready() + await vi.advanceTimersByTimeAsync(360_000) + expect(state).toBe('waiting') + respond() + await opened + expect(state).toBe('ready') + + const request = expect(manager.sendDataProcessRequest('query', {})).rejects.toThrow( + 'Request query timed out' + ) + await vi.advanceTimersByTimeAsync(30_000) + await request +}) + +it('still fails an initialization that never replies within ten minutes', async () => { + const manager = await import('./data-process-manager') + const opened = expect(manager.spawnDataProcess('/test/data.db')).rejects.toThrow( + 'Request init timed out' + ) + ready() + await vi.advanceTimersByTimeAsync(600_000) + await opened +}) + +it('propagates an integrity-check failure immediately', async () => { + const manager = await import('./data-process-manager') + const opened = expect(manager.spawnDataProcess('/test/data.db')).rejects.toThrow( + 'SQLite integrity check failed' + ) + ready() + respond('SQLite integrity check failed') + await opened +}) diff --git a/apps/electron/src/main/identity-seed.test.ts b/apps/electron/src/main/identity-seed.test.ts index b2ac4a585..cc55f3abb 100644 --- a/apps/electron/src/main/identity-seed.test.ts +++ b/apps/electron/src/main/identity-seed.test.ts @@ -86,4 +86,40 @@ describe('identity-seed', () => { getOrCreateIdentitySeed(dir, makeSafeStorage(true), { profile: 'default' }) ).toThrow(/invalid/) }) + + it.each(['data.db', 'xnet.db'])('refuses a replacement identity when %s remains', (name) => { + const dir = tempDir() + writeFileSync(join(dir, name), 'original bytes') + expect(() => getOrCreateIdentitySeed(dir, makeSafeStorage(), { profile: 'default' })).toThrow( + 'identity is missing' + ) + expect(readFileSync(join(dir, name), 'utf8')).toBe('original bytes') + expect(() => readFileSync(join(dir, 'identity-seed.json'))).toThrow() + }) + + it('preserves the encrypted seed when the key store is locked', () => { + const dir = tempDir() + getOrCreateIdentitySeed(dir, makeSafeStorage(), { profile: 'default' }) + const original = readFileSync(join(dir, 'identity-seed.json')) + expect(() => + getOrCreateIdentitySeed(dir, makeSafeStorage(false), { profile: 'default' }) + ).toThrow('key store is unavailable') + expect(readFileSync(join(dir, 'identity-seed.json'))).toEqual(original) + }) + + it('rejects corrupt base64 that permissive decoding would silently accept', () => { + const dir = tempDir() + writeFileSync( + join(dir, 'identity-seed.json'), + JSON.stringify({ + version: 1, + plaintext: true, + updatedAt: Date.now(), + payload: Buffer.alloc(32).toString('base64') + '!' + }) + ) + expect(() => getOrCreateIdentitySeed(dir, makeSafeStorage(), { profile: 'default' })).toThrow( + 'encoding is invalid' + ) + }) }) diff --git a/apps/electron/src/main/identity-seed.ts b/apps/electron/src/main/identity-seed.ts index 63cf056da..ac5d62e2f 100644 --- a/apps/electron/src/main/identity-seed.ts +++ b/apps/electron/src/main/identity-seed.ts @@ -17,8 +17,18 @@ */ import type { SafeStorageLike } from './secure-seed' -import { randomBytes } from 'node:crypto' -import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { randomBytes, randomUUID } from 'node:crypto' +import { + closeSync, + fsyncSync, + linkSync, + lstatSync, + mkdirSync, + openSync, + readFileSync, + rmSync, + writeFileSync +} from 'node:fs' import { join } from 'node:path' export type IdentitySeedMode = 'secure' | 'plaintext' | 'test' @@ -46,7 +56,8 @@ const isStoredIdentitySeed = (value: unknown): value is StoredIdentitySeed => { return ( candidate.version === 1 && typeof candidate.payload === 'string' && - typeof candidate.updatedAt === 'number' + Number.isFinite(candidate.updatedAt) && + (candidate.plaintext === undefined || typeof candidate.plaintext === 'boolean') ) } @@ -80,20 +91,45 @@ export function getOrCreateIdentitySeed( const filePath = join(dataDir, SEED_FILE_NAME) const encryptionAvailable = safeStorage.isEncryptionAvailable() - if (existsSync(filePath)) { - const parsed = JSON.parse(readFileSync(filePath, 'utf8')) as unknown + let stored: string | undefined + try { + stored = readFileSync(filePath, 'utf8') + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error + } + if (stored !== undefined) { + const parsed = JSON.parse(stored) as unknown if (!isStoredIdentitySeed(parsed)) { throw new Error(`Stored identity seed at ${filePath} is invalid`) } - const bytes = parsed.plaintext - ? Buffer.from(parsed.payload, 'base64') - : Buffer.from(safeStorage.decryptString(Buffer.from(parsed.payload, 'base64')), 'base64') + if (!parsed.plaintext && !encryptionAvailable) + throw new Error( + 'The platform key store is unavailable. Unlock it before opening this workspace.' + ) + const encoded = parsed.plaintext + ? parsed.payload + : safeStorage.decryptString(Buffer.from(parsed.payload, 'base64')) + const bytes = Buffer.from(encoded, 'base64') + if (bytes.toString('base64') !== encoded) + throw new Error('Stored identity seed encoding is invalid') if (bytes.length !== 32) { throw new Error(`Stored identity seed at ${filePath} has length ${bytes.length}, expected 32`) } return { seed: new Uint8Array(bytes), mode: parsed.plaintext ? 'plaintext' : 'secure' } } + for (const name of ['data.db', 'xnet.db']) { + try { + lstatSync(join(dataDir, name)) + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') continue + throw error + } + throw new Error( + 'The workspace identity is missing. Restore its identity and data together; a new identity was not created.' + ) + } + const seed = randomBytes(32) const seedB64 = Buffer.from(seed).toString('base64') mkdirSync(dataDir, { recursive: true }) @@ -104,7 +140,7 @@ export function getOrCreateIdentitySeed( payload: safeStorage.encryptString(seedB64).toString('base64'), updatedAt: Date.now() } - writeFileSync(filePath, JSON.stringify(record), { encoding: 'utf8', mode: 0o600 }) + persistNewIdentity(filePath, dataDir, record) return { seed: new Uint8Array(seed), mode: 'secure' } } @@ -121,6 +157,29 @@ export function getOrCreateIdentitySeed( plaintext: true, updatedAt: Date.now() } - writeFileSync(filePath, JSON.stringify(record), { encoding: 'utf8', mode: 0o600 }) + persistNewIdentity(filePath, dataDir, record) return { seed: new Uint8Array(seed), mode: 'plaintext' } } + +function persistNewIdentity(path: string, directory: string, record: StoredIdentitySeed): void { + const temporary = `${path}.incomplete-${randomUUID()}` + try { + const file = openSync(temporary, 'wx', 0o600) + try { + writeFileSync(file, JSON.stringify(record), 'utf8') + fsyncSync(file) + } finally { + closeSync(file) + } + // Exclusive promotion cannot replace a seed created by another writer. + linkSync(temporary, path) + const parent = openSync(directory, 'r') + try { + fsyncSync(parent) + } finally { + closeSync(parent) + } + } finally { + rmSync(temporary, { force: true }) + } +} diff --git a/apps/electron/src/main/index.ts b/apps/electron/src/main/index.ts index b5ecd8d4f..f6ce5248f 100644 --- a/apps/electron/src/main/index.ts +++ b/apps/electron/src/main/index.ts @@ -1,10 +1,15 @@ /** * Electron main process entry point */ -import { appendFileSync, readlinkSync } from 'fs' +import { appendFileSync, existsSync, readlinkSync } from 'fs' import { join, dirname } from 'path' import { fileURLToPath } from 'url' -import { app, BrowserWindow } from 'electron' +import { app, BrowserWindow, dialog, safeStorage } from 'electron' +import { inspectCheckpoints } from '../storage/checkpoints' +import { requireCompatibleDatabase } from '../storage/compatibility' +import { readDesktopSettings } from '../storage/desktop-settings' +import { prepareWorkspaceUpgrade } from '../storage/migrations' +import { recoverPendingRestore, restoreCheckpoint } from '../storage/restore' import { setupAgentBridgeIPC, startAgentBridge, stopAgentBridge } from './agent-bridge-manager' import { setupCloudflareTunnelIPC, stopCloudflareTunnel } from './cloudflare-tunnel-ipc' import { installMainCrashLog } from './crash-log' @@ -17,16 +22,22 @@ import { import { parseConnectDeepLink, type CloudConnectPayload } from './deep-link' import { attachDevLogWindow, installDevLogBridge } from './dev-log-bridge' import { titleSuffix } from './dev-scope' -import { setupIPC, getOrCreateStorage } from './ipc' +import { getOrCreateIdentitySeed } from './identity-seed' +import { setupIPC, getOrCreateStorage, closeStorage } from './ipc' +import { configureLibrary, setupLibraryIPC } from './library-ipc' import { startLocalAPI, stopLocalAPI, setupLocalAPIIPC } from './local-api' import { setupMeetingCaptureIPC } from './meeting-capture-ipc' -import { setupRecordingCaptureIPC, shutdownRecordingCapture } from './recording-capture-ipc' import { createMenu } from './menu' import { dataPath, profile } from './profile' +import { createQuitBarrier } from './quit-barrier' +import { setupRecordingCaptureIPC, shutdownRecordingCapture } from './recording-capture-ipc' +import { checkpointWorkspace, recoveryPath, recoveryIsBusy, setupRecovery } from './recovery' +import { flushRenderers, resumeRenderers, setupRendererFlush } from './renderer-flush' import { setupServiceIPC, cleanupServices } from './service-ipc' -import { setupSocialImportIPC } from './social-import-ipc' +import { setupSocialImportIPC, hasActiveSocialImports } from './social-import-ipc' +import { showStartupRecovery } from './startup-recovery' import { setupStorybookIPC, stopStorybook } from './storybook-ipc' -import { initAutoUpdater } from './updater' +import { hasDownloadedUpdate, initAutoUpdater, installDownloadedUpdate } from './updater' // Capture main-process console output for the renderer console (0413). First // statement after the imports so a failure during early module init is still @@ -59,6 +70,88 @@ let mainWindow: BrowserWindow | null = null let pendingSharePayload: string | null = null let pendingCloudConnect: CloudConnectPayload | null = null let cleanupTunnelIPC: (() => void) | null = null +let installRequested = false +let workspaceReady = false +let writersStopped = false + +async function stopWorkspaceWriters(): Promise { + await shutdownRecordingCapture() + await stopAgentBridge() + await stopLocalAPI() + await cleanupServices() + await stopCloudflareTunnel() + await stopStorybook() + await stopDataProcess() + await closeStorage() + writersStopped = true +} + +async function restartWorkspaceWriters(): Promise { + await recoverPendingRestore(dataPath, recoveryPath, { + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + process.env.XNET_RECOVERY_OFFLINE = existsSync(join(recoveryPath, 'review-required.json')) + ? 'true' + : 'false' + requireCompatibleDatabase(dbPath) + requireCompatibleDatabase(join(dataPath, 'xnet.db'), 'blobs') + requireCompatibleDatabase(join(dataPath, 'library.db'), 'library') + await readDesktopSettings(dataPath, safeStorage) + await getOrCreateStorage().open() + await spawnDataProcess(dbPath) + await configureLibrary() + writersStopped = false + if (process.env.XNET_RECOVERY_OFFLINE !== 'true') { + await startLocalAPI() + await startAgentBridge() + } + for (const window of BrowserWindow.getAllWindows()) { + window.webContents.once('did-finish-load', () => setupWindowChannel(window)) + window.reload() + } +} + +const quitBarrier = createQuitBarrier({ + prepare: async () => { + if (!workspaceReady) { + await stopDataProcess() + await closeStorage() + return + } + if (hasActiveSocialImports()) + throw new Error('An import is still running. Finish or cancel it before quitting.') + if (recoveryIsBusy()) + throw new Error('Wait for the current recovery operation before quitting.') + await flushRenderers() + await stopWorkspaceWriters() + cleanupTunnelIPC?.() + cleanupTunnelIPC = null + await checkpointWorkspace({ writersStopped: true, resume: false }) + }, + finish: () => { + if (installRequested || hasDownloadedUpdate()) installDownloadedUpdate() + else app.quit() + }, + failed: async (error) => { + installRequested = false + if (writersStopped) { + try { + await restartWorkspaceWriters() + } catch (restartError) { + workspaceReady = false + await showStartupRecovery(restartError, dataPath) + app.exit(1) + return + } + } + resumeRenderers() + dialog.showErrorBox( + 'xNet is still open', + error instanceof Error ? error.message : String(error) + ) + } +}) +setupRendererFlush() const DEEP_LINK_PROTOCOL = 'xnet' @@ -275,6 +368,25 @@ async function createWindow() { mainWindow = null }) + const window = mainWindow + let closing = false + window.on('close', (event) => { + if (quitBarrier.approved) return + event.preventDefault() + if (closing) return + closing = true + void flushRenderers([window]) + .then(() => window.destroy()) + .catch((error: unknown) => { + closing = false + resumeRenderers() + dialog.showErrorBox( + 'Your workspace is still open', + error instanceof Error ? error.message : String(error) + ) + }) + }) + mainWindow.webContents.on('did-finish-load', () => { bootTrace('renderer loaded') // Flush everything main logged before a renderer existed (0413). Done here @@ -321,91 +433,157 @@ process.on('unhandledRejection', (reason) => { installMainCrashLog(app.getPath('userData')) bootTrace('main module loaded') -app.whenReady().then(async () => { - bootTrace('whenReady fired') - app.setAsDefaultProtocolClient(DEEP_LINK_PROTOCOL) +app + .whenReady() + .then(async () => { + bootTrace('whenReady fired') + app.setAsDefaultProtocolClient(DEEP_LINK_PROTOCOL) - for (const arg of process.argv) { - if (arg.startsWith(`${DEEP_LINK_PROTOCOL}://`)) { - handleDeepLink(arg) - break + for (const arg of process.argv) { + if (arg.startsWith(`${DEEP_LINK_PROTOCOL}://`)) { + handleDeepLink(arg) + break + } } - } - - // Create storage early so IPC can use it - const storage = getOrCreateStorage() - bootTrace('opening storage') - await storage.open() - - // Spawn the data utility process (SQLite, Yjs, WebSocket sync) - // This runs data operations off the main thread - bootTrace('spawning data process') - await spawnDataProcess(dbPath) - bootTrace('data process ready') - // Setup IPC handlers for main process operations - setupIPC() - - // Setup IPC handlers that proxy to data utility process - setupDataProcessIPC(() => mainWindow) + await recoverPendingRestore(dataPath, recoveryPath, { + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + requireCompatibleDatabase(join(dataPath, 'library.db'), 'library') + await readDesktopSettings(dataPath, safeStorage) + await prepareWorkspaceUpgrade({ + dataPath, + recoveryPath, + profile, + appVersion: app.getVersion(), + testIdentity: process.env.XNET_TEST_BYPASS === 'true', + requireIdentity: () => { + getOrCreateIdentitySeed(dataPath, safeStorage, { + profile, + testMode: process.env.XNET_TEST_BYPASS === 'true' + }) + } + }) + process.env.XNET_RECOVERY_OFFLINE = existsSync(join(recoveryPath, 'review-required.json')) + ? 'true' + : 'false' + + // Resolve identity before opening stores: an unavailable key must not look like a fresh app. + getOrCreateIdentitySeed(dataPath, safeStorage, { + profile, + testMode: process.env.XNET_TEST_BYPASS === 'true' + }) - // Setup service IPC for plugin background processes - setupServiceIPC() + // Create storage early so IPC can use it + const storage = getOrCreateStorage() + bootTrace('opening storage') + await storage.open() - // Setup Local API IPC handlers - setupLocalAPIIPC() + // Spawn the data utility process (SQLite, Yjs, WebSocket sync) + // This runs data operations off the main thread + bootTrace('spawning data process') + await spawnDataProcess(dbPath) + bootTrace('data process ready') + await configureLibrary() + setupLibraryIPC(() => mainWindow) - // Setup local social import IPC handlers - setupSocialImportIPC(() => mainWindow) + // Setup IPC handlers for main process operations + setupIPC() + setupRecovery({ stopWriters: stopWorkspaceWriters, restartWriters: restartWorkspaceWriters }) + workspaceReady = true - // Setup meeting capture IPC (system-audio loopback + native STT engines) - setupMeetingCaptureIPC() + // Setup IPC handlers that proxy to data utility process + setupDataProcessIPC(() => mainWindow) - // Setup recording capture IPC (ScreenCaptureKit helper, exploration 0414) - setupRecordingCaptureIPC() + // Setup service IPC for plugin background processes + setupServiceIPC() - // Setup Cloudflare tunnel IPC handlers - cleanupTunnelIPC = setupCloudflareTunnelIPC() + // Setup Local API IPC handlers + setupLocalAPIIPC() - // Setup agent bridge IPC handlers (drives the user's claude/codex CLI) - setupAgentBridgeIPC() + // Setup local social import IPC handlers + setupSocialImportIPC(() => mainWindow) - // Setup dev-only Storybook IPC handlers - if (process.env.NODE_ENV === 'development') { - setupStorybookIPC() - } + // Setup meeting capture IPC (system-audio loopback + native STT engines) + setupMeetingCaptureIPC() - // Start Local API server (for external integrations) - bootTrace('starting local API') - await startLocalAPI() + // Setup recording capture IPC (ScreenCaptureKit helper, exploration 0414) + setupRecordingCaptureIPC() - // Start the agent bridge daemon (no-op if the agent CLI isn't installed). - // Fire-and-forget: a slow `--version` probe must not delay window creation. - void startAgentBridge().catch(() => undefined) + // Setup Cloudflare tunnel IPC handlers + cleanupTunnelIPC = setupCloudflareTunnelIPC() - // Create menu - createMenu() + // Setup agent bridge IPC handlers (drives the user's claude/codex CLI) + setupAgentBridgeIPC() - // Create window - bootTrace('creating window') - await createWindow() - bootTrace('window created') + // Setup dev-only Storybook IPC handlers + if (process.env.NODE_ENV === 'development') { + setupStorybookIPC() + } - // Setup MessagePort channel between renderer and data process - if (mainWindow) { - setupWindowChannel(mainWindow) - initAutoUpdater(mainWindow) - } + // Start Local API server (for external integrations) + bootTrace('starting local API') + if (process.env.XNET_RECOVERY_OFFLINE !== 'true') await startLocalAPI() + + // Start the agent bridge daemon (no-op if the agent CLI isn't installed). + // Fire-and-forget: a slow `--version` probe must not delay window creation. + if (process.env.XNET_RECOVERY_OFFLINE !== 'true') void startAgentBridge().catch(() => undefined) + + // Create menu + createMenu() + + // Create window + bootTrace('creating window') + await createWindow() + bootTrace('window created') + + // Setup MessagePort channel between renderer and data process + if (mainWindow) { + setupWindowChannel(mainWindow) + initAutoUpdater(mainWindow, async () => { + if (!hasDownloadedUpdate()) throw new Error('No downloaded update is ready to install.') + installRequested = true + await quitBarrier.request() + }) + } - app.on('activate', async () => { - if (BrowserWindow.getAllWindows().length === 0) { - await createWindow() - if (mainWindow) { - setupWindowChannel(mainWindow) + app.on('activate', async () => { + if (BrowserWindow.getAllWindows().length === 0) { + await createWindow() + if (mainWindow) { + setupWindowChannel(mainWindow) + } } + }) + }) + .catch(async (error: unknown) => { + workspaceReady = false + await stopDataProcess() + await closeStorage() + let latest: string | undefined + try { + const listing = await inspectCheckpoints(recoveryPath) + for (const point of listing.unreadable) + console.error('[Recovery] Preserved unreadable point:', point.id, point.reason) + latest = listing.checkpoints.find((point) => point.profile === profile)?.id + } catch (listingError) { + console.error('[Recovery] Could not list recovery copies:', listingError) } + await showStartupRecovery( + error, + dataPath, + latest + ? () => + restoreCheckpoint({ + id: latest!, + dataPath, + recoveryPath, + profile, + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + : undefined + ) }) -}) app.on('window-all-closed', () => { if (process.platform !== 'darwin') { @@ -413,29 +591,8 @@ app.on('window-all-closed', () => { } }) -app.on('before-quit', async () => { - // Finalize any in-flight recording so its files are closed, not truncated - await shutdownRecordingCapture() - - // Stop the agent bridge daemon - await stopAgentBridge() - - // Stop Local API server - await stopLocalAPI() - - // Stop all plugin services - await cleanupServices() - - // Stop cloudflare tunnel process - await stopCloudflareTunnel() - - // Remove tunnel event listeners - cleanupTunnelIPC?.() - cleanupTunnelIPC = null - - // Stop Storybook dev runtime - await stopStorybook() - - // Stop data utility process (handles BSM, SQLite, Yjs cleanup) - await stopDataProcess() +app.on('before-quit', (event) => { + if (quitBarrier.approved) return + event.preventDefault() + void quitBarrier.request() }) diff --git a/apps/electron/src/main/ipc.ts b/apps/electron/src/main/ipc.ts index fa3627aa7..aa6dc966f 100644 --- a/apps/electron/src/main/ipc.ts +++ b/apps/electron/src/main/ipc.ts @@ -9,6 +9,12 @@ import { SQLiteAdapter } from './storage' let storage: SQLiteAdapter | null = null +export async function closeStorage(): Promise { + if (!storage) return + await storage.close() + storage = null +} + export function getOrCreateStorage(): SQLiteAdapter { if (storage) return storage diff --git a/apps/electron/src/main/library-ipc.ts b/apps/electron/src/main/library-ipc.ts new file mode 100644 index 000000000..e4f0c4c06 --- /dev/null +++ b/apps/electron/src/main/library-ipc.ts @@ -0,0 +1,96 @@ +import { identityFromPrivateKey } from '@xnetjs/identity' +import { app, BrowserWindow, clipboard, globalShortcut, ipcMain, safeStorage } from 'electron' +import { sendDataProcessRequest } from './data-process-manager' +import { getOrCreateIdentitySeed } from './identity-seed' +import { dataPath, profile } from './profile' +import { withWorkspaceWriteBarrier } from './recovery' + +export async function configureLibrary(): Promise { + const { seed } = getOrCreateIdentitySeed(dataPath, safeStorage, { + profile, + testMode: process.env.XNET_TEST_BYPASS === 'true' + }) + try { + await sendDataProcessRequest('library:configure', { + authorDID: identityFromPrivateKey(seed).did, + signingKey: Array.from(seed) + }) + await freezeLibrary() + try { + await sendDataProcessRequest('library:recover-captures', {}, 120_000) + } finally { + await thawLibrary() + } + } finally { + seed.fill(0) + } +} +export const freezeLibrary = () => sendDataProcessRequest('library:freeze', {}, 120_000) +export const thawLibrary = () => sendDataProcessRequest('library:thaw', {}, 120_000) +export const refreshLibrarySources = () => + sendDataProcessRequest('library:scan', {}, 10 * 60 * 1000) +export function setupLibraryIPC(getWindow: () => BrowserWindow | null): void { + let captureIntent: { url: string } | null = null + let returnToPreviousApp = false + const accelerator = 'CommandOrControl+Shift+L' + const shortcutRegistered = globalShortcut.register(accelerator, () => { + returnToPreviousApp = BrowserWindow.getFocusedWindow() === null + const text = clipboard.readText().trim() + captureIntent = { url: /^https?:\/\/\S+$/i.test(text) && text.length <= 500 ? text : '' } + const window = getWindow() + if (!window || window.isDestroyed()) { + app.emit('activate') + return + } + if (window.isMinimized()) window.restore() + window.show() + window.focus() + window.webContents.send('xnet:library:capture-ready') + }) + app.once('will-quit', () => { + if (shortcutRegistered) globalShortcut.unregister(accelerator) + }) + ipcMain.handle('xnet:library:capture-intent', () => { + const intent = captureIntent + captureIntent = null + return intent + }) + ipcMain.handle('xnet:library:capture-shortcut', () => ({ + accelerator, + registered: shortcutRegistered + })) + ipcMain.handle('xnet:library:capture-closed', (_event, restoreFocus = true) => { + if (returnToPreviousApp && restoreFocus && process.platform === 'darwin') app.hide() + returnToPreviousApp = false + }) + ipcMain.handle('xnet:library:capture', (_event, payload: Record) => + withWorkspaceWriteBarrier(async () => { + const response = await sendDataProcessRequest('library:capture', { input: payload }, 120_000) + return (response as { value: unknown }).value + }) + ) + for (const action of [ + 'status', + 'search', + 'get', + 'cards', + 'graph', + 'graph-detail', + 'lookup', + 'scan', + 'pause', + 'resume', + 'retry', + 'helper-status', + 'helper-install', + 'helper-cancel' + ]) { + ipcMain.handle( + `xnet:library:${action}`, + async (_event, payload: Record = {}) => { + const response = await sendDataProcessRequest(`library:${action}`, payload, 10 * 60 * 1000) + return (response as { value: unknown }).value + } + ) + } +} diff --git a/apps/electron/src/main/profile-path.test.ts b/apps/electron/src/main/profile-path.test.ts new file mode 100644 index 000000000..0fef8ba2c --- /dev/null +++ b/apps/electron/src/main/profile-path.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { resolveProfilePath } from './profile-path' + +const daily = '/profiles/xnet-desktop' + +describe('protected daily profile', () => { + it('keeps packaged default data and identity in their existing home', () => { + expect(resolveProfilePath(daily, true)).toEqual({ profile: 'default', userData: daily }) + }) + + it.each(['default', 'user2', 'wt-feature', 'daily'])( + 'isolates a source launch even with an explicit %s profile', + (profile) => { + const dev = resolveProfilePath(daily, false, profile) + expect(dev.userData).not.toBe(daily) + expect(dev.userData).not.toBe(resolveProfilePath(daily, true, profile).userData) + expect(dev.profile).toBe(`dev-${profile}`) + } + ) + + it('prevents a packaged profile from aliasing a development profile', () => { + expect(() => resolveProfilePath(daily, true, 'dev-default')).toThrow('reserved') + expect(resolveProfilePath(daily, false, 'dev-default').profile).toBe('dev-dev-default') + }) + + it.each(['../default', '../../xnet-desktop', '/profiles', '', 'a/b', 'a\\b'])( + 'rejects profile path traversal: %s', + (profile) => + expect(() => resolveProfilePath(daily, false, profile)).toThrow('Invalid xNet profile') + ) +}) diff --git a/apps/electron/src/main/profile-path.ts b/apps/electron/src/main/profile-path.ts new file mode 100644 index 000000000..38e581d69 --- /dev/null +++ b/apps/electron/src/main/profile-path.ts @@ -0,0 +1,22 @@ +import { dirname, join } from 'node:path' + +/** Packaged data stays put; every source launch gets a separate namespace. */ +export function resolveProfilePath( + defaultUserData: string, + packaged: boolean, + requested = 'default' +) { + if (!/^[a-zA-Z0-9][a-zA-Z0-9_-]{0,100}$/.test(requested)) { + throw new Error('Invalid xNet profile name. Use letters, numbers, underscores, or hyphens.') + } + if (packaged && requested.startsWith('dev-')) + throw new Error('Profile names beginning with dev- are reserved for source builds.') + const profile = packaged ? requested : `dev-${requested}` + return { + profile, + userData: + packaged && requested === 'default' + ? defaultUserData + : join(dirname(defaultUserData), `xnet-desktop-${profile}`) + } +} diff --git a/apps/electron/src/main/profile.ts b/apps/electron/src/main/profile.ts index 50fc30a02..1cc9ebde8 100644 --- a/apps/electron/src/main/profile.ts +++ b/apps/electron/src/main/profile.ts @@ -1,13 +1,19 @@ import { join } from 'path' import { app } from 'electron' import { devScope } from './dev-scope' +import { resolveProfilePath } from './profile-path' // Profile support for running multiple instances with separate data. // // A linked git worktree scopes itself automatically — the dev launcher resolves // `wt-` and exports it as XNET_PROFILE (0413). Set XNET_PROFILE by // hand to override, as `dev:user2` does. -export const profile = devScope.profile || process.env.XNET_PROFILE || 'default' +const selection = resolveProfilePath( + app.getPath('userData'), + app.isPackaged, + devScope.profile || process.env.XNET_PROFILE || 'default' +) +export const profile = selection.profile // Set separate user data path for each profile before app readiness. // This isolates local app storage, localStorage, cookies, etc. between profiles. @@ -18,9 +24,9 @@ export const profile = devScope.profile || process.env.XNET_PROFILE || 'default' // lets two worktrees run at once. It works because this file is imported for // its side effect at module scope. `profile-lock-order.test.ts` pins it; do not // move either half into a function. -if (profile !== 'default') { - const userDataPath = join(app.getPath('userData'), '..', `xnet-desktop-${profile}`) - app.setPath('userData', userDataPath) +app.setPath('userData', selection.userData) +if (!app.isPackaged) { + process.env.XNET_DEV_SCOPE = JSON.stringify({ ...devScope, profile }) } export const dataPath = join(app.getPath('userData'), 'xnet-data') diff --git a/apps/electron/src/main/quit-barrier.test.ts b/apps/electron/src/main/quit-barrier.test.ts new file mode 100644 index 000000000..16bf19a2d --- /dev/null +++ b/apps/electron/src/main/quit-barrier.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from 'vitest' +import { createQuitBarrier } from './quit-barrier' + +describe('quit barrier', () => { + it('holds quit until saving completes and coalesces repeated requests', async () => { + const events: string[] = [] + let release!: () => void + const saved = new Promise((resolve) => { + release = resolve + }) + const barrier = createQuitBarrier({ + prepare: async () => { + events.push('saving') + await saved + }, + finish: () => { + events.push('quit') + }, + failed: () => { + events.push('failed') + } + }) + const first = barrier.request() + expect(barrier.request()).toBe(first) + await Promise.resolve() + expect(events).toEqual(['saving']) + expect(barrier.approved).toBe(false) + release() + await first + expect(events).toEqual(['saving', 'quit']) + expect(barrier.approved).toBe(true) + }) + + it('leaves the app usable on failure and allows retry', async () => { + let fail = true + const events: string[] = [] + const barrier = createQuitBarrier({ + prepare: async () => { + if (fail) throw new Error('disk full') + }, + finish: () => { + events.push('quit') + }, + failed: () => { + events.push('failed') + } + }) + await barrier.request() + expect(events).toEqual(['failed']) + expect(barrier.approved).toBe(false) + fail = false + await barrier.request() + expect(events).toEqual(['failed', 'quit']) + }) +}) diff --git a/apps/electron/src/main/quit-barrier.ts b/apps/electron/src/main/quit-barrier.ts new file mode 100644 index 000000000..65afcd40b --- /dev/null +++ b/apps/electron/src/main/quit-barrier.ts @@ -0,0 +1,32 @@ +/** One quit attempt at a time. A failed save leaves the process and editor alive. */ +export function createQuitBarrier(options: { + prepare: () => Promise + finish: () => void + failed: (error: unknown) => void | Promise +}) { + let approved = false + let pending: Promise | null = null + return { + get approved() { + return approved + }, + request(): Promise { + if (approved) return Promise.resolve() + if (pending) return pending + pending = Promise.resolve() + .then(options.prepare) + .then(() => { + approved = true + options.finish() + }) + .catch((error: unknown) => { + approved = false + return options.failed(error) + }) + .finally(() => { + pending = null + }) + return pending + } + } +} diff --git a/apps/electron/src/main/recovery.test.ts b/apps/electron/src/main/recovery.test.ts new file mode 100644 index 000000000..fd8f70d0b --- /dev/null +++ b/apps/electron/src/main/recovery.test.ts @@ -0,0 +1,112 @@ +import { beforeEach, expect, it, vi } from 'vitest' + +const mocks = vi.hoisted(() => ({ + create: vi.fn(), + retain: vi.fn(), + flush: vi.fn(), + freeze: vi.fn(), + thaw: vi.fn(), + resume: vi.fn(), + send: vi.fn() +})) +vi.mock('electron', () => ({ + app: { getPath: () => '/test/profile', getVersion: () => '3.0.0' }, + BrowserWindow: { + getAllWindows: () => [{ isDestroyed: () => false, webContents: { send: mocks.send } }] + }, + dialog: {}, + ipcMain: {}, + safeStorage: {}, + shell: {} +})) +vi.mock('./profile', () => ({ dataPath: '/test/data', profile: 'test' })) +vi.mock('./library-ipc', () => ({ freezeLibrary: mocks.freeze, thawLibrary: mocks.thaw })) +vi.mock('./renderer-flush', () => ({ flushRenderers: mocks.flush, resumeRenderers: mocks.resume })) +vi.mock('./social-import-ipc', () => ({ hasActiveSocialImports: () => false })) +vi.mock('../storage/checkpoints', () => ({ + createCheckpoint: mocks.create, + retainCheckpoints: mocks.retain, + inspectCheckpoints: vi.fn(), + workspaceFingerprint: vi.fn() +})) +vi.mock('../storage/compatibility', () => ({ + inspectDatabase: vi.fn(), + WorkspaceRecoveryRequired: class extends Error {} +})) +vi.mock('../storage/desktop-settings', () => ({ + readDesktopSettings: vi.fn(), + writeDesktopSettings: vi.fn() +})) +vi.mock('../storage/portable', () => ({ + exportPortableCheckpoint: vi.fn(), + unpackPortableCheckpoint: vi.fn() +})) +vi.mock('../storage/restore', () => ({ restoreCheckpoint: vi.fn() })) + +beforeEach(() => { + vi.resetModules() + vi.resetAllMocks() + mocks.create.mockResolvedValue({ id: 'saved-point' }) + mocks.retain.mockResolvedValue(undefined) + mocks.flush.mockResolvedValue(undefined) + mocks.freeze.mockResolvedValue(undefined) + mocks.thaw.mockResolvedValue(undefined) +}) + +it('restores editing and reports failure if resuming the Library times out', async () => { + const recovery = await import('./recovery') + mocks.thaw.mockRejectedValue(new Error('Request library:thaw timed out')) + await expect(recovery.checkpointWorkspace()).rejects.toThrow('library:thaw timed out') + expect(mocks.retain).toHaveBeenCalledOnce() + expect(mocks.resume).toHaveBeenCalledOnce() + expect(mocks.send).toHaveBeenLastCalledWith( + 'xnet:recovery:error', + 'Request library:thaw timed out' + ) + expect(recovery.recoveryIsBusy()).toBe(false) +}) + +it('keeps the recovery barrier until the Library acknowledges resuming', async () => { + const recovery = await import('./recovery') + let release!: () => void + mocks.thaw.mockReturnValueOnce( + new Promise((resolve) => { + release = resolve + }) + ) + const first = recovery.checkpointWorkspace() + await vi.waitFor(() => expect(mocks.thaw).toHaveBeenCalledOnce()) + expect(recovery.recoveryIsBusy()).toBe(true) + expect(mocks.resume).not.toHaveBeenCalled() + const second = recovery.checkpointWorkspace() + const third = recovery.checkpointWorkspace() + await Promise.resolve() + expect(mocks.create).toHaveBeenCalledOnce() + release() + await Promise.all([first, second, third]) + expect(mocks.create).toHaveBeenCalledTimes(3) + expect(mocks.resume).toHaveBeenCalledTimes(3) + expect(recovery.recoveryIsBusy()).toBe(false) +}) + +it('resumes both writers after a failed copy and clears the error after a successful retry', async () => { + const recovery = await import('./recovery') + mocks.create.mockRejectedValueOnce(new Error('Disk write failed')) + await expect(recovery.checkpointWorkspace()).rejects.toThrow('Disk write failed') + expect(mocks.thaw).toHaveBeenCalledOnce() + expect(mocks.resume).toHaveBeenCalledOnce() + expect(mocks.send).toHaveBeenLastCalledWith('xnet:recovery:error', 'Disk write failed') + await expect(recovery.checkpointWorkspace()).resolves.toEqual({ id: 'saved-point' }) + expect(mocks.send).toHaveBeenLastCalledWith('xnet:recovery:error', null) +}) + +it('keeps writers stopped for a final quit copy', async () => { + const recovery = await import('./recovery') + await recovery.checkpointWorkspace({ writersStopped: true, resume: false }) + expect(mocks.create).toHaveBeenCalledOnce() + expect(mocks.flush).not.toHaveBeenCalled() + expect(mocks.freeze).not.toHaveBeenCalled() + expect(mocks.thaw).not.toHaveBeenCalled() + expect(mocks.resume).not.toHaveBeenCalled() + expect(recovery.recoveryIsBusy()).toBe(false) +}) diff --git a/apps/electron/src/main/recovery.ts b/apps/electron/src/main/recovery.ts new file mode 100644 index 000000000..b8a6e880e --- /dev/null +++ b/apps/electron/src/main/recovery.ts @@ -0,0 +1,376 @@ +import { randomUUID } from 'node:crypto' +import { mkdir, open, readFile, realpath, rm } from 'node:fs/promises' +import { join, sep } from 'node:path' +import { app, BrowserWindow, dialog, ipcMain, safeStorage, shell } from 'electron' +import { validateDesktopSettings } from '../shared/desktop-settings' +import { createCheckpointSchedule } from '../storage/checkpoint-policy' +import { + createCheckpoint, + inspectCheckpoints, + retainCheckpoints, + workspaceFingerprint, + type CheckpointManifest +} from '../storage/checkpoints' +import { inspectDatabase, WorkspaceRecoveryRequired } from '../storage/compatibility' +import { readDesktopSettings, writeDesktopSettings } from '../storage/desktop-settings' +import { exportPortableCheckpoint, unpackPortableCheckpoint } from '../storage/portable' +import { restoreCheckpoint } from '../storage/restore' +import { freezeLibrary, thawLibrary } from './library-ipc' +import { dataPath, profile } from './profile' +import { flushRenderers, resumeRenderers } from './renderer-flush' +import { hasActiveSocialImports } from './social-import-ipc' + +export const recoveryPath = join(app.getPath('userData'), 'xnet-recovery') +let inFlight: Promise | null = null +let lastFailure: string | null = null +let busy = false +export const recoveryIsBusy = () => busy || inFlight !== null + +/** Small native mutations share the same writer barrier as recovery copies. */ +export async function withWorkspaceWriteBarrier(write: () => Promise): Promise { + if (recoveryIsBusy() || hasActiveSocialImports()) + throw new Error('Wait for the current import or recovery operation, then retry saving.') + busy = true + try { + await flushRenderers() + await freezeLibrary() + return await write() + } finally { + try { + await thawLibrary() + } finally { + resumeRenderers() + busy = false + } + } +} + +function reportFailure(message: string | null): void { + lastFailure = message + for (const window of BrowserWindow.getAllWindows()) + if (!window.isDestroyed()) window.webContents.send('xnet:recovery:error', message) +} + +export async function checkpointWorkspace( + options: { + writersStopped?: boolean + resume?: boolean + } = {} +): Promise { + // A quit checkpoint must be newer than a manual copy that was already running. + while (inFlight) await inFlight + inFlight = (async () => { + try { + let point: CheckpointManifest + try { + if (hasActiveSocialImports()) + throw new Error('Finish or cancel the current import before making a recovery copy.') + if (!options.writersStopped) { + await flushRenderers() + await freezeLibrary() + } + point = await createCheckpoint({ + dataPath, + recoveryPath, + profile, + appVersion: app.getVersion(), + testIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + await retainCheckpoints(recoveryPath, point.id, { + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + } finally { + if (options.resume !== false) { + try { + if (!options.writersStopped) await thawLibrary() + } finally { + // A failed acknowledgement must not leave the workspace inert. + resumeRenderers() + } + } + } + reportFailure(null) + return point + } catch (error) { + reportFailure(error instanceof Error ? error.message : String(error)) + throw error + } + })() + try { + return await inFlight + } finally { + inFlight = null + } +} + +export function setupRecovery(options: { + stopWriters: () => Promise + restartWriters: () => Promise +}): void { + let settingsWrite = Promise.resolve() + ipcMain.handle('xnet:settings:save', async (_event, value: unknown) => { + const settings = validateDesktopSettings(value) + const next = settingsWrite.then(() => writeDesktopSettings(dataPath, settings, safeStorage)) + // A failed save remains an error to its caller, while a later retry can run. + settingsWrite = next.catch(() => {}) + await next + }) + ipcMain.handle('xnet:settings:recovery', async () => { + await settingsWrite + const settings = await readDesktopSettings(dataPath, safeStorage) + let restoreId: string | null = null + try { + const review: unknown = JSON.parse( + await readFile(join(recoveryPath, 'review-required.json'), 'utf8') + ) + if ( + !review || + typeof review !== 'object' || + !('version' in review) || + review.version !== 1 || + !('restoredAt' in review) || + typeof review.restoredAt !== 'string' + ) + throw new Error('Unreadable settings restore marker. Workspace copies have been preserved.') + restoreId = 'id' in review && typeof review.id === 'string' ? review.id : review.restoredAt + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error + } + return { settings, restoreId } + }) + const tick = createCheckpointSchedule({ + now: Date.now, + busy: () => recoveryIsBusy() || hasActiveSocialImports(), + latest: async () => (await inspectCheckpoints(recoveryPath)).checkpoints[0] ?? null, + fingerprint: async () => { + if (recoveryIsBusy() || hasActiveSocialImports()) return workspaceFingerprint(dataPath) + busy = true + try { + // Settings-only edits must make an overdue point eligible for replacement. + await flushRenderers() + return await workspaceFingerprint(dataPath) + } finally { + resumeRenderers() + busy = false + } + }, + create: () => checkpointWorkspace(), + failed: (error) => reportFailure(error instanceof Error ? error.message : String(error)) + }) + // The first tick also catches an overdue copy after launch. Recheck once a minute after failures. + const timer = setInterval(() => void tick(), 60_000) + timer.unref() + app.once('will-quit', () => clearInterval(timer)) + ipcMain.handle('xnet:recovery:status', async () => ({ + ...(await inspectCheckpoints(recoveryPath)), + busy: busy || inFlight !== null, + error: lastFailure, + protection: 'local-only', + networkPaused: process.env.XNET_RECOVERY_OFFLINE === 'true', + coverage: + 'Native workspace, desktop identity, known desktop preferences, AI provider key, and capture draft. Device sign-in sessions and unlisted browser state are not included. Older copies may lack settings.' + })) + ipcMain.handle('xnet:recovery:resume-network', async () => { + if (busy) throw new Error('A recovery operation is already running.') + if (process.env.XNET_RECOVERY_OFFLINE !== 'true') return + const { response } = await dialog.showMessageBox({ + type: 'question', + title: 'Reconnect this restored workspace?', + message: 'Have you finished reviewing the restored workspace?', + detail: + 'Sync may bring newer work back from your other devices. xNet will restart and reconnect. The preserved workspace stays in the recovery folder.', + buttons: ['Stay offline', 'Reconnect and restart'], + defaultId: 0, + cancelId: 0 + }) + if (response !== 1) return + busy = true + try { + await flushRenderers() + await options.stopWriters() + await checkpointWorkspace({ writersStopped: true, resume: false }) + await rm(join(recoveryPath, 'review-required.json')) + const directory = await open(recoveryPath, 'r') + try { + await directory.sync() + } finally { + await directory.close() + } + app.relaunch() + app.exit(0) + } catch (error) { + await options.restartWriters() + throw error + } finally { + busy = false + resumeRenderers() + } + }) + ipcMain.handle('xnet:recovery:create', async () => { + if (busy) throw new Error('A recovery operation is already running.') + busy = true + try { + return await checkpointWorkspace() + } finally { + busy = false + } + }) + ipcMain.handle('xnet:recovery:show', async () => { + const error = await shell.openPath(recoveryPath) + if (error) throw new Error(error) + }) + ipcMain.handle('xnet:recovery:export', async (_event, password: string) => { + if (recoveryIsBusy() || hasActiveSocialImports()) + throw new Error('Finish the current import or recovery operation first.') + busy = true + try { + const picked = await dialog.showOpenDialog({ + title: 'Choose a folder for the encrypted backup', + buttonLabel: 'Save encrypted backup here', + properties: ['openDirectory', 'createDirectory'] + }) + if (picked.canceled || !picked.filePaths[0]) return null + const destination = await realpath(picked.filePaths[0]) + await mkdir(recoveryPath, { recursive: true, mode: 0o700 }) + for (const source of [dataPath, recoveryPath]) { + const root = await realpath(source) + if (destination === root || destination.startsWith(root + sep)) + throw new Error( + 'Choose a backup folder outside this workspace and its local recovery folder.' + ) + } + const point = await checkpointWorkspace() + return await exportPortableCheckpoint({ + checkpointPath: join(recoveryPath, point.id), + destination, + password, + safeStorage, + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + } catch (error) { + reportFailure(error instanceof Error ? error.message : String(error)) + throw error + } finally { + busy = false + } + }) + ipcMain.handle('xnet:recovery:import', async (_event, password: string) => { + if (recoveryIsBusy() || hasActiveSocialImports()) + throw new Error('Finish the current import or recovery operation first.') + busy = true + let stopped = false + const incoming = join(recoveryPath, 'incoming', randomUUID()) + try { + const picked = await dialog.showOpenDialog({ + title: 'Choose the encrypted .xnetbackup folder', + properties: ['openDirectory'] + }) + if (picked.canceled || !picked.filePaths[0]) return { restored: false } + await mkdir(join(recoveryPath, 'incoming'), { recursive: true, mode: 0o700 }) + const recovered = await unpackPortableCheckpoint({ + path: picked.filePaths[0], + output: incoming, + password, + safeStorage, + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + for (const [name, kind] of [ + ['data.db', 'workspace'], + ['xnet.db', 'blobs'], + ['library.db', 'library'] + ] as const) { + const path = join(recovered.workspace, name) + const compatibility = inspectDatabase(path, kind) + if ( + compatibility.status !== 'supported' && + !(kind === 'library' && compatibility.status === 'missing') + ) + throw new WorkspaceRecoveryRequired(path, compatibility) + } + const { response } = await dialog.showMessageBox({ + type: 'warning', + title: 'Restore this encrypted backup?', + message: `Restore the backup from ${new Date(recovered.source.createdAt).toLocaleString()}?`, + detail: + 'The password, files, and databases have been verified. xNet will keep your current workspace, then restart offline with the recovered identity and data. New backups also restore desktop preferences, the AI provider key, and capture draft. Device sessions require signing in again; older backups may lack settings.', + buttons: ['Cancel', 'Restore and restart'], + defaultId: 0, + cancelId: 0 + }) + if (response !== 1) return { restored: false } + const point = await createCheckpoint({ + dataPath: recovered.workspace, + recoveryPath, + profile, + appVersion: app.getVersion(), + testIdentity: recovered.source.identity === 'test', + pinned: true + }) + await flushRenderers() + await options.stopWriters() + stopped = true + await checkpointWorkspace({ writersStopped: true, resume: false }) + await restoreCheckpoint({ + id: point.id, + dataPath, + recoveryPath, + profile, + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + await rm(incoming, { recursive: true, force: true }) + app.relaunch() + app.exit(0) + return { restored: true } + } catch (error) { + reportFailure(error instanceof Error ? error.message : String(error)) + if (stopped) await options.restartWriters() + throw error + } finally { + await rm(incoming, { recursive: true, force: true }) + busy = false + resumeRenderers() + } + }) + ipcMain.handle('xnet:recovery:restore', async (_event, id: string) => { + if (busy) throw new Error('A recovery operation is already running.') + busy = true + let stopped = false + try { + const { checkpoints: points } = await inspectCheckpoints(recoveryPath) + const point = points.find((point) => point.id === id && point.profile === profile) + if (!point) throw new Error('Recovery point was not found in this workspace.') + const { response } = await dialog.showMessageBox({ + type: 'warning', + title: 'Restore this workspace?', + message: `Restore the local copy from ${new Date(point.createdAt).toLocaleString()}?`, + detail: + 'xNet will restart. Your current workspace, including newer edits, will be kept in a separate recovery folder. This copy does not include browser settings or sign-in sessions.', + buttons: ['Cancel', 'Restore and restart'], + defaultId: 0, + cancelId: 0 + }) + if (response !== 1) return { restored: false } + await flushRenderers() + await options.stopWriters() + stopped = true + await checkpointWorkspace({ writersStopped: true, resume: false }) + await restoreCheckpoint({ + id, + dataPath, + recoveryPath, + profile, + allowTestIdentity: process.env.XNET_TEST_BYPASS === 'true' + }) + app.relaunch() + app.exit(0) + return { restored: true } + } catch (error) { + reportFailure(error instanceof Error ? error.message : String(error)) + if (stopped) await options.restartWriters() + throw error + } finally { + busy = false + resumeRenderers() + } + }) +} diff --git a/apps/electron/src/main/renderer-flush.ts b/apps/electron/src/main/renderer-flush.ts new file mode 100644 index 000000000..bd96d6b75 --- /dev/null +++ b/apps/electron/src/main/renderer-flush.ts @@ -0,0 +1,60 @@ +import { randomUUID } from 'node:crypto' +import { BrowserWindow, ipcMain } from 'electron' + +const ready = new Set() + +export function setupRendererFlush(): void { + ipcMain.on('xnet:flush-ready', (event) => { + if (ready.has(event.sender.id)) return + ready.add(event.sender.id) + event.sender.once('destroyed', () => ready.delete(event.sender.id)) + }) +} + +/** Keep the renderer alive until it acknowledges all document writes. */ +export async function flushRenderers(windows = BrowserWindow.getAllWindows()): Promise { + const results = await Promise.allSettled( + windows.map( + (window) => + new Promise((resolve, reject) => { + if (window.isDestroyed()) + return reject(new Error('A workspace window closed before saving.')) + const requestId = randomUUID() + const senderId = window.webContents.id + const channel = `xnet:flush-result:${requestId}` + const cleanup = () => { + clearTimeout(timeout) + ipcMain.removeListener(channel, listener) + } + const listener = ( + event: Electron.IpcMainEvent, + result: { ok?: boolean; error?: string } + ) => { + if (event.sender.id !== senderId) return + cleanup() + if (result?.ok === true) resolve() + else reject(new Error(result?.error || 'Document save failed.')) + } + const timeout = setTimeout(() => { + cleanup() + reject(new Error('The workspace did not finish saving. Please retry.')) + }, 30_000) + ipcMain.on(channel, listener) + if (!ready.has(senderId)) { + cleanup() + reject( + new Error('The workspace is still starting. Wait for it to open before closing.') + ) + return + } + window.webContents.send('xnet:flush-documents', requestId) + }) + ) + ) + const failed = results.find((result) => result.status === 'rejected') + if (failed?.status === 'rejected') throw failed.reason +} + +export function resumeRenderers(): void { + for (const window of BrowserWindow.getAllWindows()) window.webContents.send('xnet:resume-editing') +} diff --git a/apps/electron/src/main/social-import-ipc.ts b/apps/electron/src/main/social-import-ipc.ts index 054ed0fde..cd274e9bb 100644 --- a/apps/electron/src/main/social-import-ipc.ts +++ b/apps/electron/src/main/social-import-ipc.ts @@ -1,3 +1,11 @@ +import type { + ElectronStagedSocialImport, + SocialImportArchivePreview, + SocialImportStageRequest, + SocialImportStageResult, + SocialImportCommitJobRequest, + SocialImportCommitJobSnapshot +} from '../shared/social-import' /** * Main-process IPC for local social graph archive imports. */ @@ -8,16 +16,15 @@ import type { NodeBatchWriteTimings } from '@xnetjs/data' import type { - ArchiveManifest, SocialImportArchivePreview as SharedSocialImportArchivePreview, SocialImportNodeDraft as SharedSocialImportNodeDraft, SocialImportJobCheckpointSnapshot, SocialImportNodeDraftStreamResult, SocialImportJobMetrics, - SocialImportJobPhase, - SocialImportJobProgress + SocialImportJobPhase } from '@xnetjs/social/import/core' import type { BrowserWindow, OpenDialogOptions } from 'electron' +import { extname } from 'node:path' import { createSocialImportJobCheckpointAccumulator, resolveSocialImportCommitPolicy, @@ -26,66 +33,71 @@ import { } from '@xnetjs/social/import/core' import { createSocialArchivePreview, - createZipJsonEntryReader, - createZipTextEntryReader, - readZipArchiveManifest, + openSocialImportSource, streamSocialImportNodeDrafts } from '@xnetjs/social/import/node' import { builtInSocialImportAdapters } from '@xnetjs/social/importers' import { dialog, ipcMain } from 'electron' +import { + loadImportJournals, + saveImportJournal, + retainedJournalSource, + type ImportJournal +} from '../storage/import-journal' +import { retainImportSource } from '../storage/import-sources' import { sendDataProcessRequest } from './data-process-manager' - -export type SocialImportArchivePreview = Omit & { - archivePath: string -} - -export type SocialImportNodeDraft = SharedSocialImportNodeDraft - -export type SocialImportStageRequest = { - archivePath: string - buckets?: string[] - includeSensitive?: boolean -} - -export type SocialImportStageResult = Omit & { - archive: SocialImportArchivePreview - stageId: string -} - -export type SocialImportCommitJobRequest = { - stageId: string - includeSourceRecords: boolean - authorDID: string - signingKey: number[] -} - -export type SocialImportCommitJobSummary = { - created: number - updated: number - batches: number -} - -export type SocialImportCommitJobSnapshot = SocialImportJobProgress & { - summary?: SocialImportCommitJobSummary -} +import { freezeLibrary, thawLibrary, refreshLibrarySources } from './library-ipc' +import { dataPath } from './profile' +import { recoveryIsBusy } from './recovery' + +export type { + ElectronStagedSocialImport, + SocialImportArchivePreview, + SocialImportNodeDraft, + SocialImportStageRequest, + SocialImportStageResult, + SocialImportCommitJobRequest, + SocialImportCommitJobSummary, + SocialImportCommitJobSnapshot +} from '../shared/social-import' const adapters = builtInSocialImportAdapters const approvedArchivePaths = new Set() const stagedResults = new Map() const commitJobs = new Map() +const journals = new Map() +let journalsLoaded = false const cancelledCommitJobIds = new Set() + +function ensureCommitJobsLoaded(): void { + if (journalsLoaded) return + const saved = loadImportJournals(dataPath) + for (const journal of saved) { + journals.set(journal.job.jobId, journal) + commitJobs.set(journal.job.jobId, journal.job) + } + journalsLoaded = true +} let queuedTestArchivePath: string | null = null const COMMIT_BATCH_SIZE = 2500 -type ElectronStagedSocialImport = Omit & { - archive: SocialImportArchivePreview - archivePath: string - manifest: ArchiveManifest - stageRequest: SocialImportStageRequest - importedAt: string +export function hasActiveSocialImports(): boolean { + return [...commitJobs.values()].some((job) => job.status === 'queued' || job.status === 'running') } export function setupSocialImportIPC(getWindow: () => BrowserWindow | null): void { + // The preload obtains this path from a disk-backed File selected or dropped + // by the user. Renderer-created File objects have no filesystem path. + ipcMain.handle('xnet:social-import:previewSelectedFile', async (_event, archivePath: string) => { + if ( + typeof archivePath !== 'string' || + !['.zip', '.json'].includes(extname(archivePath).toLowerCase()) + ) + throw new Error('Choose a ZIP or JSON archive.') + const preview = await createArchivePreview(archivePath) + approvedArchivePaths.add(archivePath) + return preview + }) ipcMain.handle('xnet:social-import:pickArchive', async () => { if (process.env.XNET_TEST_BYPASS === 'true' && queuedTestArchivePath) { const archivePath = queuedTestArchivePath @@ -136,6 +148,56 @@ export function setupSocialImportIPC(getWindow: () => BrowserWindow | null): voi startCommitJob(request, getWindow) ) + ipcMain.handle( + 'xnet:social-import:resumeCommitJob', + async ( + _event, + request: { + jobId: string + authorDID: string + signingKey: number[] + } + ) => { + ensureCommitJobsLoaded() + if (recoveryIsBusy()) + throw new Error('Wait for workspace recovery to finish before resuming.') + if (hasActiveSocialImports()) throw new Error('Finish or pause the current import first.') + const journal = journals.get(request.jobId) + const job = commitJobs.get(request.jobId) + if (!journal || !job || job.status === 'completed') + throw new Error('No unfinished import found.') + if (request.authorDID !== journal.authorDID || request.signingKey.length !== 32) + throw new Error('Resume with the same workspace identity that started this import.') + const adapter = journal.stage.archive.adapter + if ( + !adapters.some( + (candidate) => candidate.id === adapter?.id && candidate.version === adapter.version + ) + ) + throw new Error( + 'This importer changed. Review the retained source as a new import before continuing.' + ) + const stagedResult = { + ...journal.stage, + archivePath: retainedJournalSource(dataPath, journal) + } + const next = updateCommitJob( + job.jobId, + { status: 'queued', error: null, completedAt: null }, + getWindow + ) + void runCommitJob({ + jobId: job.jobId, + stagedResult, + request: { ...request, stageId: '', includeSourceRecords: journal.includeSourceRecords }, + totalRecords: job.totalRecords ?? 0, + getWindow, + resume: job + }) + return next + } + ) + ipcMain.handle( 'xnet:social-import:listCommitJobs', async (): Promise => listCommitJobs() @@ -143,8 +205,10 @@ export function setupSocialImportIPC(getWindow: () => BrowserWindow | null): voi ipcMain.handle( 'xnet:social-import:getCommitJob', - async (_event, jobId: string): Promise => - commitJobs.get(jobId) ?? null + async (_event, jobId: string): Promise => { + ensureCommitJobsLoaded() + return commitJobs.get(jobId) ?? null + } ) ipcMain.handle( @@ -157,18 +221,18 @@ export function setupSocialImportIPC(getWindow: () => BrowserWindow | null): voi const archiveDialogOptions: OpenDialogOptions = { title: 'Select social archive', properties: ['openFile'], - filters: [{ name: 'ZIP archives', extensions: ['zip'] }] + filters: [{ name: 'Social exports, stars, and garden snapshots', extensions: ['zip', 'json'] }] } async function createArchivePreview(archivePath: string): Promise { - const manifest = await readZipArchiveManifest(archivePath, { hashEntries: false }) + const { manifest } = await openSocialImportSource(archivePath) return requireArchivePath(await createSocialArchivePreview({ adapters, manifest }), archivePath) } async function stageArchive(request: SocialImportStageRequest): Promise { - const manifest = await readZipArchiveManifest(request.archivePath, { hashEntries: false }) - const readJsonEntry = await createZipJsonEntryReader(request.archivePath) - const readTextEntry = await createZipTextEntryReader(request.archivePath) + const { manifest, readJsonEntry, readTextEntry } = await openSocialImportSource( + request.archivePath + ) const importedAt = new Date().toISOString() const streamResults: SocialImportNodeDraftStreamResult[] = [] @@ -209,6 +273,8 @@ async function stageArchive(request: SocialImportStageRequest): Promise BrowserWindow | null ): SocialImportCommitJobSnapshot { + ensureCommitJobsLoaded() + if (recoveryIsBusy()) throw new Error('Wait for workspace recovery to finish before importing.') + if (hasActiveSocialImports()) throw new Error('Finish or pause the current import first.') const stagedResult = stagedResults.get(request.stageId) if (!stagedResult) { throw new Error(`No staged social import found for ${request.stageId}`) @@ -258,6 +327,7 @@ function startCommitJob( } function listCommitJobs(): SocialImportCommitJobSnapshot[] { + ensureCommitJobsLoaded() return [...commitJobs.values()].sort((a, b) => b.updatedAt - a.updatedAt) } @@ -269,8 +339,9 @@ function cancelCommitJob( if (!job || !isActiveJob(job)) return job ?? null cancelledCommitJobIds.add(jobId) - const next = updateCommitJob(jobId, { status: 'cancelled', updatedAt: Date.now() }, getWindow) - return next + // The current batch may still be writing. Keep quit/recovery blocked until it acknowledges. + publishCommitJob(job, getWindow) + return job } async function runCommitJob(input: { @@ -279,6 +350,7 @@ async function runCommitJob(input: { request: SocialImportCommitJobRequest totalRecords: number getWindow: () => BrowserWindow | null + resume?: SocialImportCommitJobSnapshot }): Promise { const startedAt = Date.now() const totalRecords = input.totalRecords @@ -308,18 +380,41 @@ async function runCommitJob(input: { totalScalarRowsWritten: 0, totalFtsRowsWritten: 0 } - let created = 0 - let updated = 0 - let processedRecords = 0 - let currentChunk = 0 + let created = input.resume?.created ?? 0 + let updated = input.resume?.updated ?? 0 + let processedRecords = input.resume?.processedRecords ?? 0 + let currentChunk = input.resume?.currentChunk ?? 0 + const resumeCursor = processedRecords + let streamedRecords = 0 + let streamedChunks = 0 let draftBatch: SharedSocialImportNodeDraft[] = [] const checkpointAccumulator = createSocialImportJobCheckpointAccumulator() try { + await freezeLibrary() updateCommitJob(input.jobId, { status: 'running', phase: 'checking' }, input.getWindow) - const readJsonEntry = await createZipJsonEntryReader(input.stagedResult.archivePath) - const readTextEntry = await createZipTextEntryReader(input.stagedResult.archivePath) + const expectedHash = input.stagedResult.manifest.archiveHash + if (!expectedHash) throw new Error('Import preview has no source fingerprint. Review it again.') + const retainedPath = await retainImportSource({ + dataPath, + sourcePath: input.stagedResult.archivePath, + expectedHash + }) + const { manifest, readJsonEntry, readTextEntry } = await openSocialImportSource(retainedPath) + if (manifest.archiveHash !== expectedHash) + throw new Error('Retained archive fingerprint does not match the preview.') + const journal: ImportJournal = { + version: 1, + job: commitJobs.get(input.jobId)!, + stage: input.stagedResult, + authorDID: input.request.authorDID, + sourceHash: expectedHash, + sourceExtension: extname(retainedPath) as '.zip' | '.json', + includeSourceRecords: input.request.includeSourceRecords + } + saveImportJournal(dataPath, journal) + journals.set(input.jobId, journal) const flushDraftBatch = async (): Promise => { if (draftBatch.length === 0) return @@ -327,6 +422,19 @@ async function runCommitJob(input: { assertCommitJobNotCancelled(input.jobId) const draftChunk = draftBatch draftBatch = [] + streamedRecords += draftChunk.length + streamedChunks += 1 + if (streamedRecords <= resumeCursor) { + checkpointAccumulator.add(draftChunk, { + processedRecords: streamedRecords, + currentChunk: streamedChunks + }) + return + } + if (streamedRecords - draftChunk.length < resumeCursor) + throw new Error( + 'Import cursor does not align with the reviewed source. No batch was skipped.' + ) const nextChunk = currentChunk + 1 const checkStartedAt = performance.now() @@ -377,8 +485,6 @@ async function runCommitJob(input: { processedRecords, currentChunk }) - assertCommitJobNotCancelled(input.jobId) - reportCommitJobProgress({ jobId: input.jobId, phase: currentChunk >= totalChunks ? 'finalizing' : 'checking', @@ -414,8 +520,10 @@ async function runCommitJob(input: { } await flushDraftBatch() + await thawLibrary() + await refreshLibrarySources() - if (processedRecords !== totalRecords) { + if (processedRecords !== totalRecords || streamedRecords !== totalRecords) { throw new Error( `Social import streamed ${processedRecords} records but expected ${totalRecords}` ) @@ -439,16 +547,29 @@ async function runCommitJob(input: { ) } catch (error) { const cancelled = error instanceof SocialImportCommitCancelledError - updateCommitJob( - input.jobId, - { - status: cancelled ? 'cancelled' : 'failed', - completedAt: Date.now(), - error: cancelled ? null : error instanceof Error ? error.message : String(error) - }, - input.getWindow - ) + const failed: Partial = { + status: cancelled ? 'paused' : 'failed', + completedAt: Date.now(), + error: cancelled + ? 'Paused after the last saved batch.' + : error instanceof Error + ? error.message + : String(error) + } + try { + updateCommitJob(input.jobId, failed, input.getWindow) + } catch (journalError) { + const current = commitJobs.get(input.jobId)! + const next = { + ...current, + ...failed, + error: `Import stopped; progress could not be saved: ${String(journalError)}` + } + commitJobs.set(input.jobId, next) + publishCommitJob(next, input.getWindow) + } } finally { + await thawLibrary() cancelledCommitJobIds.delete(input.jobId) } } @@ -532,6 +653,12 @@ function updateCommitJob( ...patch, updatedAt: patch.updatedAt ?? Date.now() } + const journal = journals.get(jobId) + if (journal) { + const saved = { ...journal, job: next } + saveImportJournal(dataPath, saved) + journals.set(jobId, saved) + } commitJobs.set(jobId, next) publishCommitJob(next, getWindow) return next diff --git a/apps/electron/src/main/startup-recovery.ts b/apps/electron/src/main/startup-recovery.ts new file mode 100644 index 000000000..566ce57d0 --- /dev/null +++ b/apps/electron/src/main/startup-recovery.ts @@ -0,0 +1,43 @@ +import { dirname } from 'node:path' +import { app, dialog, shell } from 'electron' +import { WorkspaceRecoveryRequired } from '../storage/compatibility' + +/** Available before renderer, identity, sync, or normal workspace startup. */ +export async function showStartupRecovery( + error: unknown, + dataPath: string, + restore?: () => Promise +): Promise { + const detail = error instanceof Error ? error.message : String(error) + console.error('[Recovery] Workspace startup stopped:', error) + const result = await dialog.showMessageBox({ + type: 'error', + title: 'Your workspace was preserved', + message: 'xNet could not safely open this workspace.', + detail: `${detail}\n\nNo database has been reset. Keep this folder when reinstalling a compatible version of xNet.\n\n${dataPath}`, + buttons: [ + 'Quit', + 'Show workspace folder', + ...(restore ? ['Restore latest recovery copy'] : []) + ], + defaultId: 1, + cancelId: 0, + noLink: true + }) + if (result.response === 1) { + const path = error instanceof WorkspaceRecoveryRequired ? dirname(error.path) : dataPath + const failure = await shell.openPath(path) + if (failure) dialog.showErrorBox('Could not open workspace folder', `${failure}\n${path}`) + } + if (result.response === 2 && restore) { + try { + await restore() + app.relaunch() + app.exit(0) + return + } catch (restoreError) { + dialog.showErrorBox('Could not restore this workspace', String(restoreError)) + } + } + app.quit() +} diff --git a/apps/electron/src/main/storage.ts b/apps/electron/src/main/storage.ts index fc5656b8c..77eda7134 100644 --- a/apps/electron/src/main/storage.ts +++ b/apps/electron/src/main/storage.ts @@ -7,6 +7,7 @@ export class SQLiteAdapter implements StorageAdapter { constructor(path: string) { this.db = new Database(path) + this.db.pragma('synchronous = FULL') } async open(): Promise { diff --git a/apps/electron/src/main/updater.ts b/apps/electron/src/main/updater.ts index a434aaa05..b40a8ffcf 100644 --- a/apps/electron/src/main/updater.ts +++ b/apps/electron/src/main/updater.ts @@ -12,9 +12,10 @@ const { autoUpdater } = pkg // ─── Configuration ────────────────────────────────────────── -// Disable auto download — we ask the user first -autoUpdater.autoDownload = false -autoUpdater.autoInstallOnAppQuit = true +// Download in the background; installation waits for a verified recovery copy. +autoUpdater.autoDownload = true +// Only the main-process save/checkpoint barrier may hand control to the installer. +autoUpdater.autoInstallOnAppQuit = false // Check interval: every 4 hours const CHECK_INTERVAL_MS = 4 * 60 * 60 * 1000 @@ -36,10 +37,22 @@ function safeSend(window: BrowserWindow, channel: string, data: unknown): void { let checkInterval: ReturnType | null = null let initialTimeout: ReturnType | null = null let initialized = false +let downloadedUpdateReady = false -export function initAutoUpdater(mainWindow: BrowserWindow): void { +export function hasDownloadedUpdate(): boolean { + return downloadedUpdateReady +} + +export function installDownloadedUpdate(): void { + autoUpdater.quitAndInstall() +} + +export function initAutoUpdater( + mainWindow: BrowserWindow, + requestInstall: () => Promise +): void { // Skip in development - if (process.env.NODE_ENV === 'development') { + if (!app.isPackaged) { return } @@ -80,23 +93,6 @@ export function initAutoUpdater(mainWindow: BrowserWindow): void { version: info.version, releaseNotes: info.releaseNotes }) - - if (mainWindow.isDestroyed()) return - - dialog - .showMessageBox(mainWindow, { - type: 'info', - title: 'Update Available', - message: `Version ${info.version} is available.`, - detail: 'Would you like to download and install it now?', - buttons: ['Download', 'Later'], - defaultId: 0 - }) - .then(({ response }: { response: number }) => { - if (response === 0) { - autoUpdater.downloadUpdate() - } - }) }) autoUpdater.on('download-progress', (progress: any) => { @@ -113,6 +109,7 @@ export function initAutoUpdater(mainWindow: BrowserWindow): void { }) autoUpdater.on('update-downloaded', (info: any) => { + downloadedUpdateReady = true if (process.platform === 'darwin') { app.dock?.setBadge('') } @@ -134,7 +131,7 @@ export function initAutoUpdater(mainWindow: BrowserWindow): void { }) .then(({ response }: { response: number }) => { if (response === 0) { - autoUpdater.quitAndInstall() + void requestInstall() } }) }) @@ -148,19 +145,16 @@ export function initAutoUpdater(mainWindow: BrowserWindow): void { // ─── IPC handlers for manual update control ───────────── ipcMain.handle('check-for-updates', async () => { - try { - const result = await autoUpdater.checkForUpdates() - return result?.updateInfo ?? null - } catch { - return null - } + const result = await autoUpdater.checkForUpdates() + if (!result) throw new Error('Update checks are unavailable in this installation.') + return result.updateInfo }) ipcMain.handle('download-update', () => { - autoUpdater.downloadUpdate() + return autoUpdater.downloadUpdate() }) ipcMain.handle('install-update', () => { - autoUpdater.quitAndInstall() + return requestInstall() }) } diff --git a/apps/electron/src/preload/index.ts b/apps/electron/src/preload/index.ts index 563540656..527220f93 100644 --- a/apps/electron/src/preload/index.ts +++ b/apps/electron/src/preload/index.ts @@ -2,19 +2,105 @@ * Preload script - exposes xNet API to renderer */ import type { CloudConnectPayload } from '../main/deep-link' +import type { DesktopSettings, SettingsRecovery } from '../shared/desktop-settings' +import type { + CaptureInput, + CaptureResult, + LibraryResource, + LibraryHelperStatus, + LibrarySearchResult, + LibraryStatus +} from '../shared/library' +import type { LibraryGraphDetail } from '../shared/library-graph' +import type { SerializedNodeBatch } from '../shared/node-batch' +import type { CheckpointManifest } from '../shared/recovery' import type { SocialImportArchivePreview, SocialImportCommitJobRequest, SocialImportCommitJobSnapshot, SocialImportStageRequest, SocialImportStageResult -} from '../main/social-import-ipc' +} from '../shared/social-import' +import type { ApplyNodeBatchResult } from '@xnetjs/data' import type { SyncReplicationConfig } from '@xnetjs/sync' -import { contextBridge, ipcRenderer } from 'electron' +import { contextBridge, ipcRenderer, webUtils } from 'electron' // Expose xNet API to renderer contextBridge.exposeInMainWorld('xnet', { + getSettingsRecovery: () => ipcRenderer.invoke('xnet:settings:recovery'), + saveDesktopSettings: (settings: DesktopSettings) => + ipcRenderer.invoke('xnet:settings:save', settings), + getRecoveryStatus: () => ipcRenderer.invoke('xnet:recovery:status'), + libraryStatus: () => ipcRenderer.invoke('xnet:library:status'), + libraryCards: (ids: string[]) => ipcRenderer.invoke('xnet:library:cards', { ids }), + libraryGraph: () => ipcRenderer.invoke('xnet:library:graph'), + libraryGraphDetail: (id: string) => ipcRenderer.invoke('xnet:library:graph-detail', { id }), + libraryHelperStatus: () => ipcRenderer.invoke('xnet:library:helper-status'), + installLibraryHelper: () => ipcRenderer.invoke('xnet:library:helper-install'), + cancelLibraryHelper: () => ipcRenderer.invoke('xnet:library:helper-cancel'), + libraryCapture: (input: CaptureInput) => ipcRenderer.invoke('xnet:library:capture', input), + libraryLookup: (url: string) => ipcRenderer.invoke('xnet:library:lookup', { url }), + libraryCaptureShortcut: () => ipcRenderer.invoke('xnet:library:capture-shortcut'), + closeLibraryCapture: (returnToPreviousApp = true) => { + void ipcRenderer.invoke('xnet:library:capture-closed', returnToPreviousApp) + }, + onLibraryCapture: (handler: (url: string) => void) => { + let active = true + const receive = () => { + void ipcRenderer + .invoke('xnet:library:capture-intent') + .then((intent: { url: string } | null) => { + if (active && intent) handler(intent.url) + }) + } + ipcRenderer.on('xnet:library:capture-ready', receive) + receive() + return () => { + active = false + ipcRenderer.removeListener('xnet:library:capture-ready', receive) + } + }, + librarySearch: (options: { text?: string; platform?: string; offset?: number; limit?: number }) => + ipcRenderer.invoke('xnet:library:search', options), + libraryGet: (id: string) => ipcRenderer.invoke('xnet:library:get', { id }), + libraryScan: () => ipcRenderer.invoke('xnet:library:scan'), + libraryPause: () => ipcRenderer.invoke('xnet:library:pause'), + libraryResume: () => ipcRenderer.invoke('xnet:library:resume'), + libraryRetry: (id?: string) => ipcRenderer.invoke('xnet:library:retry', { id }), + resumeRecoveryNetwork: () => ipcRenderer.invoke('xnet:recovery:resume-network'), + createRecoveryCopy: () => ipcRenderer.invoke('xnet:recovery:create'), + showRecoveryFolder: () => ipcRenderer.invoke('xnet:recovery:show'), + restoreRecoveryCopy: (id: string) => ipcRenderer.invoke('xnet:recovery:restore', id), + exportEncryptedBackup: (password: string) => ipcRenderer.invoke('xnet:recovery:export', password), + restoreEncryptedBackup: (password: string) => + ipcRenderer.invoke('xnet:recovery:import', password), + onRecoveryError: (handler: (message: string | null) => void) => { + const listener = (_event: Electron.IpcRendererEvent, message: string | null) => handler(message) + ipcRenderer.on('xnet:recovery:error', listener) + return () => ipcRenderer.removeListener('xnet:recovery:error', listener) + }, getProfile: () => ipcRenderer.invoke('xnet:getProfile'), + onFlushDocuments: (flush: () => Promise) => { + const handler = async (_event: unknown, requestId: string) => { + if (typeof requestId !== 'string' || !/^[a-f0-9-]{36}$/.test(requestId)) return + try { + await flush() + ipcRenderer.send(`xnet:flush-result:${requestId}`, { ok: true }) + } catch (error) { + ipcRenderer.send(`xnet:flush-result:${requestId}`, { + ok: false, + error: error instanceof Error ? error.message : String(error) + }) + } + } + ipcRenderer.on('xnet:flush-documents', handler) + ipcRenderer.send('xnet:flush-ready') + return () => ipcRenderer.removeListener('xnet:flush-documents', handler) + }, + onResumeEditing: (callback: () => void) => { + ipcRenderer.on('xnet:resume-editing', callback) + return () => ipcRenderer.removeListener('xnet:resume-editing', callback) + }, getIdentitySeed: () => ipcRenderer.invoke('xnet:identity:getSeed'), setSeedPhrase: (mnemonic: string) => ipcRenderer.invoke('xnet:seed:set', { mnemonic }), getSeedPhrase: () => ipcRenderer.invoke('xnet:seed:get'), @@ -440,6 +526,13 @@ contextBridge.exposeInMainWorld('xnetTunnel', { }) contextBridge.exposeInMainWorld('xnetSocialImport', { + previewArchiveFile: (file: File): Promise => { + const path = webUtils.getPathForFile(file) + if (!path) throw new Error('Choose an archive file from this computer.') + return ipcRenderer.invoke('xnet:social-import:previewSelectedFile', path) + }, + resumeCommitJob: (request: { jobId: string; authorDID: string; signingKey: number[] }) => + ipcRenderer.invoke('xnet:social-import:resumeCommitJob', request), pickArchive: (): Promise => ipcRenderer.invoke('xnet:social-import:pickArchive'), queueArchiveForTest: (archivePath: string): Promise => @@ -467,6 +560,8 @@ contextBridge.exposeInMainWorld('xnetSocialImport', { // This enables persistent node storage in Electron (replacing MemoryNodeStorageAdapter). contextBridge.exposeInMainWorld('xnetNodes', { + applyNodeBatch: (input: SerializedNodeBatch) => + ipcRenderer.invoke('xnet:nodes:applyNodeBatch', { input }), // Change log operations appendChange: (change: unknown) => ipcRenderer.invoke('xnet:nodes:appendChange', { change }), getChanges: (nodeId: string) => ipcRenderer.invoke('xnet:nodes:getChanges', { nodeId }), @@ -507,7 +602,56 @@ contextBridge.exposeInMainWorld('xnetNodes', { }) // Type declaration for renderer +export interface RecoveryStatus { + unreadable: { id: string; reason: string }[] + checkpoints: CheckpointManifest[] + busy: boolean + error: string | null + protection: 'local-only' + networkPaused: boolean + coverage: string +} + export interface XNetAPI { + libraryCaptureShortcut(): Promise<{ accelerator: string; registered: boolean }> + closeLibraryCapture(returnToPreviousApp?: boolean): void + onLibraryCapture(handler: (url: string) => void): () => void + libraryCapture(input: CaptureInput): Promise + libraryLookup( + url: string + ): Promise<{ id: string; title: string; notes: { pageId: string; title: string }[] } | null> + libraryStatus(): Promise + libraryHelperStatus(): Promise + installLibraryHelper(): Promise + cancelLibraryHelper(): Promise + librarySearch(options: { + text?: string + platform?: string + offset?: number + limit?: number + }): Promise + libraryGet(id: string): Promise + libraryCards(ids: string[]): Promise<(LibrarySearchResult | null)[]> + libraryGraph(): Promise + libraryGraphDetail(id: string): Promise + libraryScan(): Promise + libraryPause(): Promise + libraryResume(): Promise + libraryRetry(id?: string): Promise + getRecoveryStatus(): Promise + resumeRecoveryNetwork(): Promise + createRecoveryCopy(): Promise + showRecoveryFolder(): Promise + restoreRecoveryCopy(id: string): Promise<{ restored: boolean }> + exportEncryptedBackup( + password: string + ): Promise<{ path: string; createdAt: string; files: number; bytes: number } | null> + restoreEncryptedBackup(password: string): Promise<{ restored: boolean }> + onRecoveryError(handler: (message: string | null) => void): () => void + getSettingsRecovery(): Promise + saveDesktopSettings(settings: DesktopSettings): Promise + onFlushDocuments(flush: () => Promise): () => void + onResumeEditing(callback: () => void): () => void getProfile(): Promise getIdentitySeed(): Promise<{ seedB64: string; mode: 'secure' | 'plaintext' | 'test' }> setSeedPhrase(mnemonic: string): Promise<{ ok: true }> @@ -520,7 +664,13 @@ export interface XNetAPI { } export interface XNetSocialImportAPI { + resumeCommitJob(request: { + jobId: string + authorDID: string + signingKey: number[] + }): Promise pickArchive(): Promise + previewArchiveFile(file: File): Promise queueArchiveForTest(archivePath: string): Promise stageArchive(request: SocialImportStageRequest): Promise startCommitJob(request: SocialImportCommitJobRequest): Promise @@ -655,6 +805,7 @@ export interface XNetTunnelAPI { // Node Storage API types (for IPC-based NodeStorageAdapter) export interface XNetNodesAPI { + applyNodeBatch(input: SerializedNodeBatch): Promise // Change log operations appendChange(change: unknown): Promise getChanges(nodeId: string): Promise diff --git a/apps/electron/src/renderer/App.tsx b/apps/electron/src/renderer/App.tsx index d761831a7..6f821ade3 100644 --- a/apps/electron/src/renderer/App.tsx +++ b/apps/electron/src/renderer/App.tsx @@ -19,6 +19,7 @@ import { BundledPluginInstaller } from './components/BundledPluginInstaller' import { CanvasView } from './components/CanvasView' import { ConnectHubDialog } from './components/ConnectHubDialog' import { setPersistedHubUrl } from './lib/hub-url' +import { useNativeNodeChanges } from './lib/use-native-node-changes' import { useDesktopPlatformPort } from './shell/desktop-platform' import { registerDesktopHostedViews } from './shell/hosted-views' import { STORIES_ENABLED, useDocumentShell } from './shell/use-document-shell' @@ -81,6 +82,29 @@ export function App(): React.ReactElement { const [showAddSharedDialog, setShowAddSharedDialog] = useState(false) const [prefilledShareValue, setPrefilledShareValue] = useState('') const [connectRequest, setConnectRequest] = useState(null) + const [recoveryNotice, setRecoveryNotice] = useState(null) + const [recoveryError, setRecoveryError] = useState(null) + const nativeChangeError = useNativeNodeChanges() + + useEffect( + () => + window.xnet.onLibraryCapture((url) => { + window.dispatchEvent(new CustomEvent('xnet:open-library-capture', { detail: { url } })) + }), + [] + ) + + useEffect(() => { + void window.xnet.getRecoveryStatus().then( + (status) => { + setRecoveryError(status.error) + if (status.networkPaused) + setRecoveryNotice('Restored workspace: sync is paused while you review your data.') + }, + (error: unknown) => setRecoveryNotice(`Could not read recovery status: ${String(error)}`) + ) + return window.xnet.onRecoveryError(setRecoveryError) + }, []) useEffect(() => { const cleanup = window.xnet.onSharePayload((payload) => { @@ -132,9 +156,33 @@ export function App(): React.ReactElement { title: 'Open Stories', run: () => handleOpenStories() }) - return () => disposable.dispose() + return () => { + void disposable.dispose() + } }, [handleOpenStories]) + useEffect(() => { + const disposable = getCommandRegistry().register({ + id: 'desktop.importArchive', + title: 'Import social archive, GitHub stars, or garden', + run: handleOpenSocialImport + }) + return () => { + void disposable.dispose() + } + }, [handleOpenSocialImport]) + + useEffect(() => { + const disposable = getCommandRegistry().register({ + id: 'desktop.library', + title: 'Open Library', + run: handleOpenDataWorkspace + }) + return () => { + void disposable.dispose() + } + }, [handleOpenDataWorkspace]) + if (homeCanvasBootstrapError && !homeCanvasId) { return (
@@ -178,7 +226,31 @@ export function App(): React.ReactElement { starts below a slim drag strip instead of underneath them. The frames subtract --titlebar-height so the bottom islands stay on-screen. */} -
+
+ +
+ {nativeChangeError && ( +

+ {nativeChangeError} +

+ )} + {(recoveryError || recoveryNotice) && ( +
+ {recoveryError ? `Recovery copy failed: ${recoveryError}` : recoveryNotice} + +
+ )}
{/* Focused surfaces are lazy chunks (cold-open budget); the null diff --git a/apps/electron/src/renderer/bootstrap.ts b/apps/electron/src/renderer/bootstrap.ts new file mode 100644 index 000000000..95fc73cfa --- /dev/null +++ b/apps/electron/src/renderer/bootstrap.ts @@ -0,0 +1,21 @@ +import { restoreDesktopSettings } from '../shared/desktop-settings' + +async function boot(): Promise { + const recovery = await window.xnet.getSettingsRecovery() + restoreDesktopSettings(localStorage, recovery) + // Workbench and consent stores hydrate at module scope. + await import('./main') +} + +void boot().catch((error: unknown) => { + const root = document.getElementById('root') + if (!root) throw error + const heading = document.createElement('h1') + heading.textContent = 'Desktop settings could not be recovered' + const detail = document.createElement('p') + detail.textContent = error instanceof Error ? error.message : String(error) + const help = document.createElement('p') + help.textContent = + 'Your workspace has not been reset. Unlock the Mac key store if needed, then restart xNet.' + root.replaceChildren(heading, detail, help) +}) diff --git a/apps/electron/src/renderer/components/DataWorkspaceView.tsx b/apps/electron/src/renderer/components/DataWorkspaceView.tsx index 27a3727ab..efc88711a 100644 --- a/apps/electron/src/renderer/components/DataWorkspaceView.tsx +++ b/apps/electron/src/renderer/components/DataWorkspaceView.tsx @@ -11,8 +11,10 @@ import { DataWorkspaceBody, type SavedViewCanvasFrameInput } from '@xnetjs/views' +import { useNavigateTo } from '@xnetjs/workbench' import { Database, Import, Loader2, X } from 'lucide-react' -import React, { useEffect, useMemo } from 'react' +import React, { useEffect, useMemo, useState } from 'react' +import { LibraryView } from './LibraryView' export type { SavedViewCanvasFrameInput } @@ -21,7 +23,22 @@ type DataWorkspaceViewProps = { onInsertSavedLensAsCanvasFrame?: (input: SavedViewCanvasFrameInput) => void } -export function DataWorkspaceView({ +export function DataWorkspaceView(props: DataWorkspaceViewProps): React.ReactElement { + const [graph, setGraph] = useState(false) + const navigate = useNavigateTo() + return graph ? ( + setGraph(false)} /> + ) : ( + setGraph(true)} + onImport={() => navigate({ kind: 'path', path: '/social-import' })} + onOpenPage={(nodeId) => navigate({ kind: 'node', nodeType: 'page', nodeId })} + /> + ) +} + +function DataWorkspaceGraphView({ onClose, onInsertSavedLensAsCanvasFrame }: DataWorkspaceViewProps): React.ReactElement { diff --git a/apps/electron/src/renderer/components/LibraryCollections.tsx b/apps/electron/src/renderer/components/LibraryCollections.tsx new file mode 100644 index 000000000..cfcc6d828 --- /dev/null +++ b/apps/electron/src/renderer/components/LibraryCollections.tsx @@ -0,0 +1,260 @@ +import type { LibrarySearchResult } from '../../shared/library' +import { useQuery } from '@xnetjs/react' +import { SocialCollectionItemSchema, SocialCollectionSchema } from '@xnetjs/social/schemas' +import { useEffect, useState } from 'react' +import { LibraryResourceCard } from './LibraryResourceCard' + +const PAGE_SIZE = 40 +const button = + 'rounded-md border border-border px-3 py-1.5 text-sm hover:bg-accent disabled:opacity-50' + +function Pagination({ + offset, + hasMore, + loading, + onOffset +}: { + offset: number + hasMore: boolean + loading: boolean + onOffset: (value: number) => void +}) { + return ( +
+ + +
+ ) +} + +export function LibraryCollections({ + onSelectResource +}: { + onSelectResource: (id: string) => void +}) { + const [query, setQuery] = useState('') + const [offset, setOffset] = useState(0) + const [selected, setSelected] = useState<{ id: string; title: string } | null>(null) + const collections = useQuery(SocialCollectionSchema, { + search: query.trim() || undefined, + orderBy: { title: 'asc' }, + page: { first: PAGE_SIZE, count: 'exact' }, + offset, + source: 'local', + enabled: selected === null + }) + if (selected) + return ( + setSelected(null)} + onSelectResource={onSelectResource} + /> + ) + return ( +
+
+

Your collections

+

+ Playlists and saved groups from your imports. Their entries stay connected to the original + sources. +

+
+ { + setQuery(event.target.value) + setOffset(0) + }} + /> + {collections.error && ( +

+ {collections.error.message} +

+ )} + {collections.loading &&

Loading collections…

} + {!collections.loading && !collections.error && !collections.data.length && ( +

+ {query + ? 'No matching collections.' + : 'No imported collections yet. Import an archive that includes playlists or saved groups.'} +

+ )} + {collections.completeness?.level === 'partial' && + collections.completeness.reason !== 'page-limited' && ( +

+ Only part of the collection list is available:{' '} + {collections.completeness.reason ?? 'incomplete query'}. +

+ )} +
+ {collections.data.map((collection) => ( + + ))} +
+ +
+ ) +} + +function CollectionMembers({ + collection, + onBack, + onSelectResource +}: { + collection: { id: string; title: string } + onBack: () => void + onSelectResource: (id: string) => void +}) { + const [offset, setOffset] = useState(0) + const [cards, setCards] = useState(new Map()) + const [loadingCards, setLoadingCards] = useState(true) + const [error, setError] = useState(null) + const members = useQuery(SocialCollectionItemSchema, { + where: { collection: collection.id }, + // A single sort field keeps archive order authoritative across query normalization. + orderBy: { sortKey: 'asc' }, + page: { first: PAGE_SIZE, count: 'exact' }, + offset, + source: 'local' + }) + const idsKey = JSON.stringify([ + ...new Set( + members.data + .map((member) => member.item) + .filter((id) => typeof id === 'string' && id.length > 0) + ) + ]) + useEffect(() => { + let active = true + let request = 0 + const ids = JSON.parse(idsKey) as string[] + setCards(new Map()) + setLoadingCards(true) + setError(null) + const load = async () => { + const current = ++request + try { + const next = await window.xnet.libraryCards(ids) + if (active && request === current) { + setCards(new Map(ids.map((id, index) => [id, next[index]]))) + setError(null) + } + } catch (cause) { + if (active && request === current) + setError(cause instanceof Error ? cause.message : String(cause)) + } finally { + if (active && request === current) setLoadingCards(false) + } + } + void load() + const timer = setInterval(() => void load(), 5000) + return () => { + active = false + clearInterval(timer) + } + }, [idsKey]) + return ( +
+ +
+

{collection.title}

+

+ {members.totalCount === null + ? 'Saved entries' + : `${members.totalCount.toLocaleString()} saved entries`} + . Repeated saves remain separate. Order follows the export when available. +

+
+ {(members.error || error) && ( +

+ {members.error?.message || error} +

+ )} + {(members.loading || loadingCards) && ( +

Loading saved entries…

+ )} + {members.completeness?.level === 'partial' && + members.completeness.reason !== 'page-limited' && ( +

+ Some entries could not be loaded: {members.completeness.reason ?? 'incomplete query'}. +

+ )} + {!members.loading && !members.error && !members.data.length && ( +

+ No membership records are available in this local view. +

+ )} + {!members.loading && !loadingCards && !error && !members.error && ( +
+ {members.data.map((member, index) => { + const resource = member.item ? cards.get(member.item) : null + return resource ? ( + onSelectResource(resource.id)} + /> + ) : ( +
+

Entry {offset + index + 1}

+

Source details unavailable

+

+ This entry is not indexed as a Library resource. In Resources, choose Find + imported links, or inspect it in Graph & saved views. +

+
+ ) + })} +
+ )} + +
+ ) +} diff --git a/apps/electron/src/renderer/components/LibraryGraphView.tsx b/apps/electron/src/renderer/components/LibraryGraphView.tsx new file mode 100644 index 000000000..cdc59cb16 --- /dev/null +++ b/apps/electron/src/renderer/components/LibraryGraphView.tsx @@ -0,0 +1,664 @@ +import type { GraphControls } from './library-graph/GraphCanvas' +import type { + LibraryGraph, + LibraryGraphDetail, + LibraryGraphNode, + LibraryGraphRelation +} from '../../shared/library-graph' +import { useCallback, useEffect, useMemo, useRef, useState } from 'react' +import { createPortal } from 'react-dom' +import { GraphCanvas } from './library-graph/GraphCanvas' +import { GraphGroups } from './library-graph/GraphGroups' +import { GraphSearch } from './library-graph/GraphSearch' +import { graphAdjacency, graphColor, relationKinds, selectGraph } from './library-graph/model' +import { groupNames } from './library-graph/navigation' +import { LibraryThumbnail } from './LibraryResourceCard' + +const button = + 'rounded-md border border-border px-2.5 py-1.5 text-xs hover:bg-accent disabled:opacity-50' +const count = (value: number) => value.toLocaleString() +const errorText = (error: unknown) => (error instanceof Error ? error.message : String(error)) + +function GraphDetail({ + node, + neighbors, + edges, + pinned, + onPin, + onChoose, + onNeighborhood, + onOpenResource +}: { + node: LibraryGraphNode + neighbors: LibraryGraphNode[] + edges: LibraryGraph['edges'] + pinned: boolean + onPin: () => void + onChoose: (node: LibraryGraphNode) => void + onNeighborhood: () => void + onOpenResource: (id: string) => void +}) { + const [detail, setDetail] = useState(null) + const [error, setError] = useState(null) + const [metadataOpen, setMetadataOpen] = useState(false) + const [limit, setLimit] = useState(30) + const [filter, setFilter] = useState('') + useEffect(() => { + let active = true + setDetail(null) + setError(null) + setLimit(30) + setFilter('') + if (node.kind !== 'link') return + // Fast pointer movement should not enqueue a full metadata read for every dot. + const timer = setTimeout(() => { + void window.xnet.libraryGraphDetail(node.id).then( + (value) => { + if (active) setDetail(value) + }, + (reason) => { + if (active) setError(errorText(reason)) + } + ) + }, 160) + return () => { + active = false + clearTimeout(timer) + } + }, [node.id, node.kind]) + const resource = detail?.resource + const filtered = neighbors.filter((neighbor) => + neighbor.label.toLocaleLowerCase().includes(filter.toLocaleLowerCase()) + ) + const evidence = [...new Set(edges.map((edge) => edge.evidence))] + return ( +
+
+ + {pinned ? 'Pinned detail' : 'Hover preview'} · {node.kind} + + +
+ {node.kind === 'link' && ( + + )} +
+

+ {resource?.metadata?.title || node.label} +

+

+ {node.platform} + {resource?.metadata?.author || resource?.actor + ? ` · ${resource.metadata?.author || resource.actor}` + : ''} +

+
+ {error && ( +

+ {error} +

+ )} + {node.kind === 'link' && !detail && !error && ( +

Loading saved details…

+ )} + {detail && !resource && ( +

+ This source is no longer in the Library. Reload the graph. +

+ )} + {resource && ( + <> +

+ {resource.metadata?.description || resource.sourceText || 'No description saved yet.'} +

+
+
Added to Library
+
{new Date(resource.addedAt).toLocaleDateString()}
+
Source privacy
+
{resource.privacy}
+ {resource.metadata?.durationSeconds !== undefined && ( + <> +
Duration
+
{Math.round(resource.metadata.durationSeconds / 60)} min
+ + )} + {resource.metadata?.language && ( + <> +
Language
+
{resource.metadata.language}
+ + )} +
Transcript
+
+ {resource.transcript + ? `${resource.transcript.language} · ${count(resource.transcript.cues.length)} cues` + : 'Not saved'} +
+
+
+ + Open original + + +
+ + )} +
+
+

{count(neighbors.length)} connections

+ +
+

+ {node.kind === 'creator' + ? 'Links with this author name on this platform. Matching names are not a verified identity.' + : 'Connections come from imported memberships, source tags, categories, or matching creator names.'} + {evidence.includes('hashtag') ? ' Hashtags are read from saved source text.' : ''} +

+ {neighbors.length > 30 && ( + { + setFilter(event.target.value) + setLimit(30) + }} + className="w-full rounded border border-border bg-background p-2 text-xs" + /> + )} +
+ {filtered.slice(0, limit).map((neighbor) => ( + + ))} +
+ {!neighbors.length && ( +

+ No known connections with these filters. The link is still included. +

+ )} + {filtered.length > limit && ( + + )} +
+ {detail && ( +
setMetadataOpen(event.currentTarget.open)} + > + All saved metadata & provenance +

+ Original imported fields and saved enrichment, including field coverage and caption + evidence. +

+
+            {metadataOpen ? JSON.stringify(detail, null, 2) : null}
+          
+
+ )} +
+ ) +} + +export function LibraryGraphView({ + onOpenResource, + onClose +}: { + onOpenResource: (id: string) => void + onClose: () => void +}) { + const searchInput = useRef(null) + useEffect(() => { + const root = document.getElementById('root') + const previousInert = root?.inert ?? false + const previousFocus = document.activeElement + if (root) root.inert = true + searchInput.current?.focus() + return () => { + if (root) root.inert = previousInert + if (previousFocus instanceof HTMLElement && previousFocus.isConnected) previousFocus.focus() + } + }, []) + const [showInspector, setShowInspector] = useState(true) + const [showGroups, setShowGroups] = useState(true) + const [groups, setGroups] = useState([]) + const [match, setMatch] = useState<'any' | 'all'>('any') + const [graph, setGraph] = useState(null) + const [loading, setLoading] = useState(true) + const [error, setError] = useState(null) + const [renderError, setRenderError] = useState(null) + const [revision, setRevision] = useState(0) + const [platform, setPlatform] = useState('') + const [kinds, setKinds] = useState(relationKinds) + const [focus, setFocus] = useState(null) + const [previewId, setPreviewId] = useState(null) + const [pinned, setPinned] = useState(false) + const [paused, setPaused] = useState( + () => window.matchMedia('(prefers-reduced-motion: reduce)').matches + ) + const [progress, setProgress] = useState(0) + const controller = useRef(null) + useEffect(() => { + let active = true + setLoading(true) + setError(null) + setRenderError(null) + void window.xnet + .libraryGraph() + .then((serialized) => { + if (active) { + setGraph(JSON.parse(serialized) as LibraryGraph) + setLoading(false) + } + }) + .catch((reason: unknown) => { + if (active) { + setError(errorText(reason)) + setLoading(false) + } + }) + return () => { + active = false + } + }, [revision]) + const visible = useMemo( + () => (graph ? selectGraph(graph, platform, kinds, focus, groups, match) : null), + [graph, platform, kinds, focus, groups, match] + ) + const adjacency = useMemo(() => (visible ? graphAdjacency(visible) : []), [visible]) + const indices = useMemo(() => new Map(visible?.nodes.map((node, i) => [node.id, i])), [visible]) + const previewIndex = previewId ? indices.get(previewId) : undefined + const preview = previewIndex === undefined ? null : visible!.nodes[previewIndex] + const groupLabels = useMemo( + () => + new Map( + graph?.nodes.filter((node) => node.kind !== 'link').map((node) => [node.id, node.label]) + ), + [graph] + ) + const hubs = useMemo( + () => + visible?.nodes + .filter((node) => node.kind !== 'link') + .sort((a, b) => adjacency[indices.get(b.id)!].length - adjacency[indices.get(a.id)!].length) + .slice(0, 12) ?? [], + [visible, adjacency, indices] + ) + const choose = useCallback( + (node: LibraryGraphNode) => { + setPreviewId(node.id) + setPinned(true) + setShowInspector(true) + const index = indices.get(node.id) + if (index !== undefined) controller.current?.focus(index) + }, + [indices] + ) + const resetPreview = () => { + setPreviewId(null) + setPinned(false) + setRenderError(null) + } + const toggleGroup = (id: string) => { + setGroups((current) => + current.includes(id) ? current.filter((value) => value !== id) : [...current, id] + ) + setFocus(null) + resetPreview() + } + const clearFilters = () => { + setGroups([]) + setPlatform('') + setFocus(null) + setMatch('any') + resetPreview() + } + const platforms = useMemo( + () => + [ + ...new Set(graph?.nodes.filter((node) => node.kind === 'link').map((node) => node.platform)) + ].sort(), + [graph] + ) + return createPortal( +
{ + if (event.key === 'Escape') { + event.stopPropagation() + onClose() + } + }} + > +
+ + Library · 3D graph + +
+
+ + + + + {focus && ( + + )} +
+
+ Show connections: + {relationKinds.map((kind) => ( + + ))} + Source relationships · no AI categories yet +
+ {(groups.length > 0 || platform || focus) && ( +
+ {groups.length > 0 && ( + <> + + {groups.map((id) => ( + + ))} + + )} + {platform && Source: {platform}} + {focus && Within a neighborhood} + +
+ )} + {error && ( +

+ {error} +

+ )} + {graph?.warnings.map((warning) => ( +

+ {warning} +

+ ))} +
+ {showGroups && graph && ( + + )} +
+ {visible && !loading && !renderError && ( + { + if (!pinned) setPreviewId(visible.nodes[index].id) + }} + onSelect={(index) => { + setPreviewId(visible.nodes[index].id) + setPinned(true) + }} + onProgress={setProgress} + onError={setRenderError} + /> + )} +
+

+ {focus + ? 'A closer look' + : groups.length || platform + ? 'Filtered links' + : 'Your link universe'} +

+

+ {loading + ? 'Reading all saved links and relationships…' + : visible + ? `${count(visible.linkCount)} of ${count(graph!.linkCount)} links · ${count(visible.nodes.length - visible.linkCount)} groups · ${count(visible.edges.length)} connections` + : 'Graph unavailable'} +

+ {!loading && visible && ( +

+ {paused + ? 'Layout paused' + : progress < 1 + ? `Arranging in 3D · ${Math.round(progress * 100)}%` + : 'Layout settled'} + {focus ? ' · Neighborhood view' : ''} +

+ )} +
+ {renderError && ( +
+ {renderError} Search and metadata remain available. +
+ )} + {!loading && visible?.linkCount === 0 && ( +

+ No web links match this view. Change the filters or import saved links. +

+ )} +
+ + + + +
+

+ Drag to orbit · Scroll to zoom · Right-drag to pan · Click to pin +

+
+ {showInspector && ( + + )} +
+
, + document.body + ) +} diff --git a/apps/electron/src/renderer/components/LibraryResourceCard.tsx b/apps/electron/src/renderer/components/LibraryResourceCard.tsx new file mode 100644 index 000000000..fbc53e87e --- /dev/null +++ b/apps/electron/src/renderer/components/LibraryResourceCard.tsx @@ -0,0 +1,92 @@ +import type { LibraryResource, LibrarySearchResult } from '../../shared/library' +import { useEffect, useState } from 'react' + +export const libraryTimestamp = (ms: number) => + `${Math.floor(ms / 60000)}:${String(Math.floor(ms / 1000) % 60).padStart(2, '0')}` + +export function LibraryThumbnail({ + resource +}: { + resource: Pick +}) { + const [url, setUrl] = useState(null) + const [failed, setFailed] = useState(false) + const cid = resource.thumbnail?.cid + const contentType = resource.thumbnail?.contentType + useEffect(() => { + let active = true + let objectUrl: string | null = null + setUrl(null) + setFailed(false) + if (cid) { + void window.xnetBSM + .getBlob(cid) + .then((bytes) => { + if (!active) return + if (!bytes) { + setFailed(true) + return + } + objectUrl = URL.createObjectURL(new Blob([new Uint8Array(bytes)], { type: contentType })) + setUrl(objectUrl) + }) + .catch(() => { + if (active) setFailed(true) + }) + } + return () => { + active = false + if (objectUrl) URL.revokeObjectURL(objectUrl) + } + }, [cid, contentType]) + return url && !failed ? ( + setFailed(true)} + /> + ) : ( +
+ {resource.platform} · {failed ? 'Saved image could not be read' : 'No saved thumbnail yet'} +
+ ) +} + +export function LibraryResourceCard({ + resource, + onSelect, + caption +}: { + resource: LibrarySearchResult + onSelect: () => void + caption?: string +}) { + return ( + + ) +} diff --git a/apps/electron/src/renderer/components/LibraryView.tsx b/apps/electron/src/renderer/components/LibraryView.tsx new file mode 100644 index 000000000..1a3ac9c3f --- /dev/null +++ b/apps/electron/src/renderer/components/LibraryView.tsx @@ -0,0 +1,592 @@ +import type { + LibraryResource, + LibrarySearchResult, + LibraryStatus, + LibraryHelperStatus +} from '../../shared/library' +import { getCommandRegistry } from '@xnetjs/plugins' +import { flushDocumentWrites } from '@xnetjs/react/internal' +import { lazy, Suspense, useEffect, useState } from 'react' +import { LibraryCollections } from './LibraryCollections' +import { + LibraryResourceCard, + LibraryThumbnail, + libraryTimestamp as timestamp +} from './LibraryResourceCard' + +const LibraryGraphView = lazy(() => + import('./LibraryGraphView').then((module) => ({ default: module.LibraryGraphView })) +) + +const button = + 'rounded-md border border-border px-3 py-1.5 text-sm hover:bg-accent disabled:opacity-50' +const label = (value: string) => value.replaceAll('-', ' ') +const errorText = (error: unknown) => + (error instanceof Error ? error.message : String(error)) + .replace(/^Error: /, '') + .replace(/^Error invoking remote method '[^']+': (?:Error: )?/, '') +const atTime = (resource: LibraryResource, ms: number) => { + if ((resource.networkPlatform ?? resource.platform) !== 'youtube') return resource.url + const url = new URL(resource.url) + url.searchParams.set('t', `${Math.floor(ms / 1000)}s`) + return url.href +} + +export function LibraryView({ + onOpenGraph, + onImport, + onClose, + onOpenPage +}: { + onOpenGraph: () => void + onImport: () => void + onClose: () => void + onOpenPage: (id: string) => void +}) { + const [status, setStatus] = useState<(LibraryStatus & { error: string | null }) | null>(null) + const [results, setResults] = useState([]) + const [query, setQuery] = useState('') + const [section, setSection] = useState<'resources' | 'collections' | 'graph'>('resources') + const [platform, setPlatform] = useState('') + const [offset, setOffset] = useState(0) + const [selectedId, setSelectedId] = useState(null) + const [selectedCue, setSelectedCue] = useState(null) + const [cueLimit, setCueLimit] = useState(100) + const [selected, setSelected] = useState(null) + const [busy, setBusy] = useState(false) + const [error, setError] = useState(null) + const [showProgress, setShowProgress] = useState(false) + const [helper, setHelper] = useState(null) + const [installingHelper, setInstallingHelper] = useState(false) + const [shortcut, setShortcut] = useState(null) + useEffect(() => { + let active = true + void window.xnet.libraryCaptureShortcut().then( + (value) => { + if (active) setShortcut(value.registered) + }, + (error) => { + if (active) setError(errorText(error)) + } + ) + void flushDocumentWrites() + .then(() => window.xnet.libraryScan()) + .catch((error) => { + if (active) setError(errorText(error)) + }) + return () => { + active = false + } + }, []) + const refresh = async () => { + setStatus(await window.xnet.libraryStatus()) + setHelper(await window.xnet.libraryHelperStatus()) + setResults(await window.xnet.librarySearch({ text: query, platform, offset, limit: 40 })) + } + useEffect(() => { + let active = true + let loading = false + const load = async () => { + if (!active || loading) return + loading = true + try { + const nextStatus = await window.xnet.libraryStatus() + const helperState = await window.xnet.libraryHelperStatus() + const rows = + section === 'resources' + ? await window.xnet.librarySearch({ text: query, platform, offset, limit: 40 }) + : [] + if (active) { + setStatus(nextStatus) + setHelper(helperState) + setResults(rows) + } + } catch (error) { + if (active) setError(errorText(error)) + } finally { + loading = false + } + } + void load() + const timer = setInterval(() => void load(), 5000) + return () => { + active = false + clearInterval(timer) + } + }, [query, platform, offset, section]) + useEffect(() => { + let active = true + setSelected(null) + setCueLimit(100) + const load = async () => { + if (!selectedId) return + try { + const resource = await window.xnet.libraryGet(selectedId) + if (active) setSelected(resource) + } catch (error) { + if (active) setError(errorText(error)) + } + } + void load() + const timer = setInterval(() => void load(), 5000) + return () => { + active = false + clearInterval(timer) + } + }, [selectedId]) + const run = async (operation: () => Promise) => { + setBusy(true) + setError(null) + try { + await operation() + await refresh() + } catch (error) { + setError(errorText(error)) + } finally { + setBusy(false) + } + } + const count = (capability: string, state: string) => + status?.counts.find((row) => row.capability === capability && row.state === state)?.count ?? 0 + return ( +
+
+
+

Library

+

+ Keep what matters. Find it again with context. +

+ {shortcut !== null && ( +

+ {shortcut + ? 'Capture a copied link with Command/Ctrl + Shift + L.' + : 'The capture shortcut is unavailable. Use Save a link.'} +

+ )} +
+
+ + + + +
+
+ + {section === 'resources' && ( +
+ { + setQuery(event.target.value) + setOffset(0) + }} + placeholder="Search titles, descriptions, URLs, and transcripts" + className="min-w-64 flex-1 rounded-md border border-border bg-background px-3 py-2 text-sm" + /> + + +
+ )} +
+ + {status?.resources.toLocaleString() ?? '…'} resources ·{' '} + {status?.error + ? 'Enrichment stopped' + : status?.paused + ? 'Enrichment paused' + : status?.running.length + ? 'Enriching sources' + : status?.nextAt !== null + ? 'Waiting for next source request' + : 'Pass finished — review Coverage & gaps'} + + + +
+ {status && ( +
+

+ Metadata: {count('metadata', 'complete').toLocaleString()} complete,{' '} + {count('metadata', 'partial').toLocaleString()} partial,{' '} + {( + count('metadata', 'queued') + + count('metadata', 'running') + + count('metadata', 'retry') + ).toLocaleString()}{' '} + pending + {' · '}Thumbnails: {count('thumbnail', 'complete').toLocaleString()} + {' · '}Captions: {count('transcript', 'complete').toLocaleString()} +

+ {status.running.map((job) => ( +

+ {label(job.capability)} · {job.title} +

+ ))} + {!status.paused && + !status.running.length && + status.nextAt !== null && + status.nextAt > Date.now() && ( +

+ Next request {new Date(status.nextAt).toLocaleTimeString()}. Provider pacing and + retry delays are preserved when you restart. +

+ )} +
+ )} + {(error || status?.error) && ( +

+ {error || status?.error} +

+ )} + {showProgress && ( +
+

+ Enrichment requests source websites from this Mac. Saved text and images remain + available offline. YouTube titles, descriptions, thumbnails, and available caption + tracks are fetched directly, with a helper fallback for inaccessible YouTube captions. + Instagram and TikTok public posts supply written captions, authors, and thumbnails. + Available TikTok and web subtitle tracks are saved and indexed. Written captions are + separate from spoken transcripts. Restricted or unavailable posts are listed below and + do not stop the remaining videos. Automatic local transcription is not connected yet. +

+
+

+ Helper for X/Twitter and YouTube caption fallback:{' '} + {helper ? label(helper.state) : 'checking…'} + {helper ? ` · yt-dlp ${helper.version}` : ''} +

+

+ Download the tested helper directly from its official GitHub release (about 38 MB). + xNet verifies it before use. This does not read browser cookies, download video media, + or start enrichment. A compatible existing yt-dlp installation can also be used. +

+ {helper?.reason &&

{helper.reason}

} + {helper && !['ready', 'unsupported'].includes(helper.state) && ( + + )} + {(installingHelper || helper?.state === 'installing') && ( + + )} +
+
+ {['metadata', 'thumbnail', 'transcript', 'index'].map((capability) => ( +
+ {label(capability)} +

+ {count(capability, 'complete')} complete · {count(capability, 'partial')} partial + · {count(capability, 'queued') + count(capability, 'running')} pending +

+

+ {count(capability, 'blocked') + count(capability, 'retry')} blocked/retry ·{' '} + {count(capability, 'unavailable')} unavailable ·{' '} + {count(capability, 'not-applicable')} not applicable +

+
+ ))} +
+ + {status?.recent.map((job) => ( +

+ {job.title} · {job.capability}: {job.reason} +

+ ))} +
+ )} + {section === 'graph' ? ( + Loading 3D view…

}> + setSection('resources')} + onOpenResource={(id) => { + setSection('resources') + setSelectedId(id) + setSelectedCue(null) + }} + /> +
+ ) : ( +
+
+ {section === 'collections' ? ( + { + setSelectedId(id) + setSelectedCue(null) + }} + /> + ) : ( + <> + {!results.length && ( +

+ {query + ? 'No matching source text yet. Check enrichment coverage for unresolved sources.' + : 'Import an archive to begin, then find its links here.'} +

+ )} +
+ {results.map((resource, index) => ( + { + setSelectedId(resource.id) + setSelectedCue(resource.startMs ?? null) + }} + /> + ))} +
+
+ + +
+ + )} +
+ {selected && ( + + )} +
+ )} +
+ ) +} diff --git a/apps/electron/src/renderer/components/PageView.tsx b/apps/electron/src/renderer/components/PageView.tsx index c3992bea6..5a9ecb178 100644 --- a/apps/electron/src/renderer/components/PageView.tsx +++ b/apps/electron/src/renderer/components/PageView.tsx @@ -8,7 +8,7 @@ * - Real-time presence indicators */ -import type { SyncStatus } from '@xnetjs/react' +import type { SyncStatus, PageTaskInput } from '@xnetjs/react' import type { CommentThreadData } from '@xnetjs/ui' import { PageSchema } from '@xnetjs/data' import { @@ -26,8 +26,7 @@ import { useComments, useNode, useIdentity, - usePageTaskSync, - type PageTaskInput + usePageTaskSync } from '@xnetjs/react' import { CommentsSidebar } from '@xnetjs/ui' import React, { useState, useCallback, useMemo, useRef } from 'react' @@ -79,6 +78,9 @@ export function PageView({ docId, minimalChrome = false }: PageViewProps) { data: page, doc, loading, + isDirty, + error, + save, update, syncStatus, peerCount, @@ -292,6 +294,23 @@ export function PageView({ docId, minimalChrome = false }: PageViewProps) { onTitleSubmit={handleTitleSubmit} titleInputRef={titleInputRef} > + + {error ? ( + + ) : isDirty ? ( + 'Saving text…' + ) : ( + 'Text saved on this Mac' + )} + {!minimalChrome && } {!minimalChrome && unresolvedCount > 0 && ( - - - - - +
+ )} +

+ xNet copies changed data about every 15 minutes, and verifies a copy before quitting or + installing an update. Recent copies cover a day, daily copies a week, and weekly copies a + month. The latest two and pinned copies stay available. Copies include private workspace + content; keep the recovery folder private. +

+
+ +
+ {status?.checkpoints.length === 0 && ( +

No verified local recovery copy yet.

+ )} + {Boolean(status?.unreadable.length) && ( +
+

+ {status!.unreadable.length} recovery point(s) could not be read. Their files have been + kept for inspection. New recovery copies can still be made. +

+
    + {status!.unreadable.map((point) => ( +
  • + {point.id}: {point.reason} +
  • + ))} +
+
+ )} +
+

Encrypted backup

+

+ Save your workspace and recovery keys to a folder you choose. For protection against + losing this Mac, copy the entire .xnetbackup folder to another disk or device. xNet + verifies the exported files; it cannot confirm that your destination is off this Mac. +

+ + setRecoveryPassword(event.target.value)} + placeholder="At least 12 characters; several random words work well" + disabled={busy} + /> +

+ Keep this password somewhere safe and separate from the backup. xNet does not save it and + cannot recover a forgotten password. Browser settings and sign-in sessions are excluded. +

+
+ + +
+ {exported && ( +

+ Backup written and verified: {exported} +

+ )} +
+
    + {status?.checkpoints.map((point) => ( +
  • +
    +
    {new Date(point.createdAt).toLocaleString()}
    +
    + Created by xNet {point.appVersion} + {point.storageVersion ? ` · Storage ${point.storageVersion}` : ''} + {point.pinned ? ' · Pinned before upgrade' : ''} · {point.files.length} files ·{' '} + {(point.files.reduce((total, file) => total + file.size, 0) / 1048576).toFixed(1)}{' '} + MB +
    +
    + +
  • + ))} +
) } diff --git a/apps/electron/src/renderer/components/SocialImportView.tsx b/apps/electron/src/renderer/components/SocialImportView.tsx index ea8288c94..ab5cb88ba 100644 --- a/apps/electron/src/renderer/components/SocialImportView.tsx +++ b/apps/electron/src/renderer/components/SocialImportView.tsx @@ -6,7 +6,7 @@ import type { SocialImportArchivePreview, SocialImportCommitJobSnapshot, SocialImportStageResult -} from '../../main/social-import-ipc' +} from '../../shared/social-import' import type { SocialImporterRegistryEntry } from '@xnetjs/social/importers' import { useMutate, useXNet } from '@xnetjs/react' import { useXNetInternal } from '@xnetjs/react/internal' @@ -74,6 +74,7 @@ export function SocialImportView({ const [commitSummary, setCommitSummary] = useState(null) const [commitProgress, setCommitProgress] = useState(null) const [commitJobId, setCommitJobId] = useState(null) + const [savedJobs, setSavedJobs] = useState([]) const [workspaceSummary, setWorkspaceSummary] = useState(null) const [workspaceSeeding, setWorkspaceSeeding] = useState(false) const activeCommitJobIdRef = useRef(null) @@ -89,7 +90,7 @@ export function SocialImportView({ return 2 + (includeSourceRecords ? stagedRecordCount : canonicalRecordCount) }, [canonicalRecordCount, includeSourceRecords, stageResult, stagedRecordCount]) - const handlePickArchive = useCallback(async () => { + const handlePickArchive = useCallback(async (file?: File) => { setError(null) setCommitSummary(null) setCommitProgress(null) @@ -98,7 +99,9 @@ export function SocialImportView({ setWorkspaceSummary(null) try { - const preview = await window.xnetSocialImport.pickArchive() + const preview = file + ? await window.xnetSocialImport.previewArchiveFile(file) + : await window.xnetSocialImport.pickArchive() if (!preview) return setArchive(preview) @@ -157,6 +160,7 @@ export function SocialImportView({ }, [archive, includeSensitive, selectedBuckets]) const applyCommitJobSnapshot = useCallback((job: SocialImportCommitJobSnapshot) => { + setSavedJobs((jobs) => [job, ...jobs.filter((item) => item.jobId !== job.jobId)]) upsertSocialImportJobProgress(job) if (job.jobId !== activeCommitJobIdRef.current) return @@ -183,12 +187,12 @@ export function SocialImportView({ return } - if (job.status === 'cancelled') { + if (job.status === 'cancelled' || job.status === 'paused') { setStatus('staged') activeCommitJobIdRef.current = null setCommitJobId(null) setCommitProgress(null) - setError('Import cancelled.') + setError(job.error ?? 'Import paused. You can resume it below.') } }, []) @@ -196,6 +200,36 @@ export function SocialImportView({ () => window.xnetSocialImport.onCommitJob(applyCommitJobSnapshot), [applyCommitJobSnapshot] ) + useEffect(() => { + void window.xnetSocialImport + .listCommitJobs() + .then(setSavedJobs) + .catch((error: unknown) => setError(toErrorMessage(error))) + }, []) + + const handleResume = async (jobId: string) => { + if (!authorDID || !signingKey || !nodeStoreReady) return + setError(null) + setStatus('committing') + activeCommitJobIdRef.current = jobId + setCommitJobId(jobId) + try { + applyCommitJobSnapshot( + await window.xnetSocialImport.resumeCommitJob({ + jobId, + authorDID, + signingKey: Array.from(signingKey) + }) + ) + const latest = await window.xnetSocialImport.getCommitJob(jobId) + if (latest) applyCommitJobSnapshot(latest) + } catch (error) { + activeCommitJobIdRef.current = null + setCommitJobId(null) + setStatus('idle') + setError(toErrorMessage(error)) + } + } const handleCommit = useCallback(async () => { if (!stageResult || !nodeStoreReady || !authorDID || !signingKey) return @@ -296,7 +330,15 @@ export function SocialImportView({
@@ -405,7 +465,7 @@ export function SocialImportView({ className="flex items-center gap-2 rounded-md border border-border px-3 py-1.5 text-sm transition-colors hover:bg-accent" > - Cancel + Pause ) : null}
-
+
+ {savedJobs.length > 0 && ( +
+

Saved import progress

+

+ Resume from retained source files, even after a restart. Completed batches stay + in your library. +

+ {savedJobs.map((job) => ( +
+
+

+ {job.archiveName} · {job.status} · {job.processedRecords.toLocaleString()}{' '} + / {job.totalRecords?.toLocaleString() ?? '?'} records +

+ {job.error &&

{job.error}

} +
+ {['paused', 'failed', 'cancelled'].includes(job.status) && ( + + )} +
+ ))} +
+ )} {error ? : null} {status === 'committed' && commitSummary ? (
{archive.probe.buckets.map((bucket) => { - const sensitive = bucket.privacyClass === 'private-message' + const sensitive = bucket.privacyClass !== 'public' && !bucket.defaultSelected const disabled = sensitive && !includeSensitive const checked = selectedBuckets.includes(bucket.id) && !disabled diff --git a/apps/electron/src/renderer/components/library-graph/GraphCanvas.tsx b/apps/electron/src/renderer/components/library-graph/GraphCanvas.tsx new file mode 100644 index 000000000..d7e7cd2f2 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/GraphCanvas.tsx @@ -0,0 +1,110 @@ +import type { LibraryGraph } from '../../../shared/library-graph' +import type { MutableRefObject } from 'react' +import { useEffect, useRef } from 'react' +import { createGraphScene } from './scene' + +export type GraphControls = Pick, 'fit' | 'focus' | 'zoom'> + +export function GraphCanvas({ + graph, + active, + paused, + controller, + onHover, + onSelect, + onProgress, + onError +}: { + graph: LibraryGraph + active: number | null + paused: boolean + controller: MutableRefObject + onHover: (index: number) => void + onSelect: (index: number) => void + onProgress: (value: number) => void + onError: (message: string) => void +}) { + const host = useRef(null) + const scene = useRef | null>(null) + const worker = useRef(null) + const touched = useRef(false) + const latest = useRef({ onHover, onSelect, onProgress, onError, paused }) + latest.current = { onHover, onSelect, onProgress, onError, paused } + useEffect(() => { + if (!host.current) return + let layout: Worker | null = null + touched.current = false + const visibility = () => + layout?.postMessage({ paused: latest.current.paused || document.hidden }) + try { + scene.current = createGraphScene(host.current, graph, { + hover: (index) => { + if (index !== null) latest.current.onHover(index) + }, + select: (index) => latest.current.onSelect(index), + failure: (message) => latest.current.onError(message) + }) + controller.current = { + fit: () => { + touched.current = true + scene.current?.fit() + }, + focus: (index) => { + touched.current = true + scene.current?.focus(index) + }, + zoom: (factor) => { + touched.current = true + scene.current?.zoom(factor) + } + } + layout = new Worker(new URL('./layout.worker.ts', import.meta.url), { type: 'module' }) + worker.current = layout + layout.onmessage = ({ + data + }: MessageEvent<{ positions: Float32Array; progress: number }>) => { + scene.current?.positions(data.positions) + latest.current.onProgress(data.progress) + if (data.progress === 1 && !touched.current) scene.current?.fit() + } + layout.onerror = (event) => { + event.preventDefault() + latest.current.onError('The background layout failed. Reload the graph to try again.') + layout?.terminate() + } + latest.current.onProgress(0) + layout.postMessage({ graph, paused: latest.current.paused || document.hidden }) + document.addEventListener('visibilitychange', visibility) + } catch (error) { + latest.current.onError( + `Could not start the 3D view: ${error instanceof Error ? error.message : String(error)}` + ) + } + return () => { + layout?.terminate() + worker.current = null + document.removeEventListener('visibilitychange', visibility) + scene.current?.dispose() + scene.current = null + controller.current = null + } + }, [graph, controller]) + useEffect(() => { + scene.current?.active(active) + }, [active, graph]) + useEffect(() => { + worker.current?.postMessage({ paused: paused || document.hidden }) + }, [paused]) + return ( +
{ + touched.current = true + }} + onWheel={() => { + touched.current = true + }} + /> + ) +} diff --git a/apps/electron/src/renderer/components/library-graph/GraphGroups.tsx b/apps/electron/src/renderer/components/library-graph/GraphGroups.tsx new file mode 100644 index 000000000..bbbdfdb19 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/GraphGroups.tsx @@ -0,0 +1,126 @@ +import type { LibraryGraph, LibraryGraphRelation } from '../../../shared/library-graph' +import { useDeferredValue, useMemo, useState } from 'react' +import { graphColor } from './model' +import { graphGroups, groupNames, normalizeSearch } from './navigation' + +const kinds: LibraryGraphRelation[] = ['category', 'tag', 'collection', 'creator'] +const labels = { category: 'Categories', tag: 'Tags', collection: 'Playlists', creator: 'Creators' } + +export function GraphGroups({ + graph, + platform, + selected, + onToggle +}: { + graph: LibraryGraph + platform: string + selected: string[] + onToggle: (id: string) => void +}) { + const [kind, setKind] = useState('category') + const [query, setQuery] = useState('') + const [limit, setLimit] = useState(30) + const search = useDeferredValue(normalizeSearch(query)) + const groups = useMemo(() => graphGroups(graph, platform), [graph, platform]) + const filtered = useMemo( + () => + groups.filter( + ({ node }) => + node.kind === kind && + search.split(/\s+/u).every((term) => normalizeSearch(node.label).includes(term)) + ), + [groups, kind, search] + ) + return ( + + ) +} diff --git a/apps/electron/src/renderer/components/library-graph/GraphSearch.tsx b/apps/electron/src/renderer/components/library-graph/GraphSearch.tsx new file mode 100644 index 000000000..1d644da52 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/GraphSearch.tsx @@ -0,0 +1,161 @@ +import type { LibraryGraph, LibraryGraphNode } from '../../../shared/library-graph' +import type { RefObject } from 'react' +import { useDeferredValue, useEffect, useId, useMemo, useRef, useState } from 'react' +import { graphColor } from './model' +import { graphSearchIndex, groupNames, searchGraph } from './navigation' + +export function GraphSearch({ + graph, + inputRef, + onChoose +}: { + graph: LibraryGraph | null + inputRef: RefObject + onChoose: (node: LibraryGraphNode) => void +}) { + const [query, setQuery] = useState('') + const [open, setOpen] = useState(false) + const [active, setActive] = useState(0) + const search = useDeferredValue(query) + const index = useMemo(() => (graph ? graphSearchIndex(graph) : []), [graph]) + const results = useMemo(() => searchGraph(index, search), [index, search]) + const pending = query !== search + const listId = useId() + const list = useRef(null) + const expanded = open && graph !== null + const selected = Math.min(active, Math.max(0, results.items.length - 1)) + useEffect(() => { + list.current?.querySelector('[aria-selected="true"]')?.scrollIntoView({ block: 'nearest' }) + }, [selected, expanded, results]) + const choose = (node: LibraryGraphNode) => { + onChoose(node) + setOpen(false) + } + return ( +
{ + if (!event.currentTarget.contains(event.relatedTarget)) setOpen(false) + }} + > + setOpen(true)} + onChange={(event) => { + setQuery(event.target.value) + setActive(0) + setOpen(true) + }} + onKeyDown={(event) => { + if (event.key === 'Escape' && expanded) { + event.preventDefault() + event.stopPropagation() + setOpen(false) + } else if (event.key === 'ArrowDown' || event.key === 'ArrowUp') { + event.preventDefault() + setOpen(true) + if (!pending) + setActive( + expanded + ? (selected + (event.key === 'ArrowDown' ? 1 : -1) + results.items.length) % + Math.max(1, results.items.length) + : event.key === 'ArrowDown' + ? 0 + : Math.max(0, results.items.length - 1) + ) + } else if (event.key === 'Enter' && expanded && !pending && results.items[selected]) { + event.preventDefault() + choose(results.items[selected].node) + } + }} + /> + {query && ( + + )} + {expanded && ( +
+
+ {search.trim() + ? `${results.total.toLocaleString()} matches in this view` + : 'Popular groups in this view'} + {' · ↑↓ to choose · Enter to navigate'} +
+
+ {results.items.map(({ node, count }, i) => ( + + ))} +
+ {!results.items.length && ( +

+ No matches in this view. Try fewer words or clear the graph filters. +

+ )} + {results.total > results.items.length && ( +

+ Showing the best {results.items.length} matches. Keep typing to narrow the list. +

+ )} +
+ )} +
+ ) +} diff --git a/apps/electron/src/renderer/components/library-graph/d3-force-3d.d.ts b/apps/electron/src/renderer/components/library-graph/d3-force-3d.d.ts new file mode 100644 index 000000000..381e2e8bf --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/d3-force-3d.d.ts @@ -0,0 +1,31 @@ +/** The upstream package has no declarations. This is the small API used by the worker. */ +declare module 'd3-force-3d' { + export interface SimulationNode { + x: number + y: number + z: number + } + export interface Force { + (alpha: number): void + } + export interface Simulation { + stop(): this + tick(iterations?: number): this + alphaDecay(value: number): this + velocityDecay(value: number): this + force(name: string, force: Force): this + } + export function forceSimulation(nodes: SimulationNode[], dimensions: number): Simulation + export function forceManyBody(): Force & { + strength(value: number): ReturnType + theta(value: number): ReturnType + distanceMax(value: number): ReturnType + } + export function forceLink(links: { source: number; target: number }[]): Force & { + distance(value: number): ReturnType + strength(value: number): ReturnType + } + export function forceX(value: number): Force & { strength(value: number): Force } + export function forceY(value: number): Force & { strength(value: number): Force } + export function forceZ(value: number): Force & { strength(value: number): Force } +} diff --git a/apps/electron/src/renderer/components/library-graph/layout.worker.ts b/apps/electron/src/renderer/components/library-graph/layout.worker.ts new file mode 100644 index 000000000..7d72df443 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/layout.worker.ts @@ -0,0 +1,53 @@ +import type { LibraryGraph } from '../../../shared/library-graph' +import { forceSimulation, forceManyBody, forceLink, forceX, forceY, forceZ } from 'd3-force-3d' +import { initialPositions } from './model' + +type LayoutRequest = { graph: LibraryGraph; paused: boolean } | { paused: boolean } +let paused = false +let step: (() => void) | null = null +let timer: ReturnType | undefined + +self.onmessage = ({ data }: MessageEvent) => { + paused = data.paused + clearTimeout(timer) + if ('graph' in data) { + const initial = initialPositions(data.graph) + const nodes = data.graph.nodes.map((_node, i) => ({ + x: initial[i * 3], + y: initial[i * 3 + 1], + z: initial[i * 3 + 2] + })) + const simulation = forceSimulation(nodes, 3) + .stop() + .alphaDecay(0.035) + .velocityDecay(0.45) + .force('charge', forceManyBody().strength(-18).theta(1.2).distanceMax(400)) + .force( + 'links', + forceLink(data.graph.edges.map(({ source, target }) => ({ source, target }))) + .distance(65) + .strength(0.18) + ) + .force('x', forceX(0).strength(0.006)) + .force('y', forceY(0).strength(0.006)) + .force('z', forceZ(0).strength(0.006)) + let ticks = 0 + const publish = () => { + const positions = new Float32Array(nodes.length * 3) + nodes.forEach((node, i) => positions.set([node.x, node.y, node.z], i * 3)) + self.postMessage({ positions, progress: ticks / 160 }, { transfer: [positions.buffer] }) + } + publish() + step = () => { + if (paused || ticks >= 160) return + const start = performance.now() + do { + simulation.tick() + ticks++ + } while (ticks < 160 && performance.now() - start < 50) + publish() + if (ticks < 160) timer = setTimeout(step!, 40) + } + } + if (!paused) step?.() +} diff --git a/apps/electron/src/renderer/components/library-graph/model.test.ts b/apps/electron/src/renderer/components/library-graph/model.test.ts new file mode 100644 index 000000000..88e4df923 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/model.test.ts @@ -0,0 +1,81 @@ +import type { LibraryGraph } from '../../../shared/library-graph' +import { expect, it } from 'vitest' +import { graphAdjacency, initialPositions, relationKinds, selectGraph } from './model' + +const graph: LibraryGraph = { + nodes: [ + { id: 'a', kind: 'link', label: 'A', platform: 'youtube' }, + { id: 'b', kind: 'link', label: 'B', platform: 'github' }, + { id: 'c', kind: 'link', label: 'C', platform: 'youtube' }, + { id: 'tag', kind: 'tag', label: '#learning', platform: '' }, + { id: 'list', kind: 'collection', label: 'Playlist', platform: 'youtube' } + ], + edges: [ + { source: 0, target: 3, kind: 'tag', evidence: 'import' }, + { source: 1, target: 3, kind: 'tag', evidence: 'import' }, + { source: 0, target: 4, kind: 'collection', evidence: 'import' } + ], + linkCount: 3, + resourceCount: 4, + warnings: [], + builtAt: 1 +} + +it('retains isolated links when relationship filters are disabled', () => { + const result = selectGraph(graph, '', [], null) + expect(result.nodes.map((node) => node.id)).toEqual(['a', 'b', 'c']) + expect(result.edges).toEqual([]) +}) +it('remaps edges after a platform filter and does not include other platforms', () => { + const result = selectGraph(graph, 'github', relationKinds, null) + expect(result.nodes.map((node) => node.id)).toEqual(['b', 'tag']) + expect(result.edges).toEqual([{ source: 0, target: 1, kind: 'tag', evidence: 'import' }]) +}) +it('isolates links sharing a known relationship without pulling in unrelated nodes', () => { + const result = selectGraph(graph, '', relationKinds, 'a') + expect(result.nodes.map((node) => node.id)).toEqual(['a', 'b', 'tag', 'list']) + expect(result.linkCount).toBe(2) + expect(graphAdjacency(result)[0]).toEqual([2, 3]) + expect(selectGraph(graph, '', relationKinds, 'tag').nodes.map((node) => node.id)).toEqual([ + 'a', + 'b', + 'tag' + ]) +}) +it('seeds a deterministic, finite 3D layout, including empty and disconnected graphs', () => { + const positions = initialPositions(graph) + expect(positions).toEqual(initialPositions(graph)) + expect(positions).toHaveLength(graph.nodes.length * 3) + expect(Array.from(positions).every(Number.isFinite)).toBe(true) + expect(new Set(Array.from(positions).filter((_value, i) => i % 3 === 2)).size).toBeGreaterThan(1) + expect(initialPositions({ nodes: [], edges: [] })).toHaveLength(0) +}) + +it('combines group memberships with any/all semantics and preserves other relationships', () => { + const any = selectGraph(graph, '', relationKinds, null, ['tag', 'list']) + expect(any.nodes.map((node) => node.id)).toEqual(['a', 'b', 'tag', 'list']) + const all = selectGraph(graph, '', relationKinds, null, ['tag', 'list'], 'all') + expect(all.nodes.map((node) => node.id)).toEqual(['a', 'tag', 'list']) + expect( + all.edges.map(({ source, target }) => [all.nodes[source].id, all.nodes[target].id]) + ).toEqual([ + ['a', 'tag'], + ['a', 'list'] + ]) + expect(selectGraph(graph, 'github', relationKinds, null, ['list']).linkCount).toBe(0) +}) + +it('filters memberships even when relationship lines are hidden', () => { + const result = selectGraph(graph, '', [], null, ['list']) + expect(result.nodes.map((node) => node.id)).toEqual(['a']) + expect(result.edges).toEqual([]) + expect(selectGraph(graph, '', relationKinds, 'tag', ['list']).linkCount).toBe(1) +}) + +it('does not mistake duplicate edges for membership in another selected group', () => { + const duplicate = { ...graph, edges: [...graph.edges, graph.edges[1]] } + expect(selectGraph(duplicate, '', relationKinds, null, ['tag', 'list'], 'all').linkCount).toBe(1) + expect(selectGraph(graph, '', relationKinds, null, ['missing']).linkCount).toBe(0) + expect(selectGraph(graph, '', relationKinds, null, ['tag', 'missing'], 'all').linkCount).toBe(0) + expect(selectGraph(graph, '', relationKinds, null, []).linkCount).toBe(graph.linkCount) +}) diff --git a/apps/electron/src/renderer/components/library-graph/model.ts b/apps/electron/src/renderer/components/library-graph/model.ts new file mode 100644 index 000000000..a822f43f1 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/model.ts @@ -0,0 +1,136 @@ +import type { LibraryGraph, LibraryGraphRelation } from '../../../shared/library-graph' + +export const relationKinds: LibraryGraphRelation[] = ['collection', 'tag', 'category', 'creator'] +export const graphColors: Record = { + youtube: '#fb7185', + instagram: '#e879f9', + github: '#c4b5fd', + x: '#7dd3fc', + tiktok: '#2dd4bf', + reddit: '#fb923c', + generic: '#a3e635', + openai: '#6ee7b7', + claude: '#fcd34d', + grok: '#94a3b8', + collection: '#fbbf24', + tag: '#a78bfa', + category: '#67e8f9', + creator: '#fda4af' +} +export const graphColor = (kind: string, platform: string) => + graphColors[kind === 'link' ? platform : kind] ?? '#cbd5e1' + +export function graphAdjacency(graph: LibraryGraph): number[][] { + const adjacency = graph.nodes.map(() => [] as number[]) + graph.edges.forEach(({ source, target }) => { + adjacency[source].push(target) + adjacency[target].push(source) + }) + return adjacency +} + +export function selectGraph( + graph: LibraryGraph, + platform: string, + kinds: LibraryGraphRelation[], + focus: string | null, + groups: string[] = [], + match: 'any' | 'all' = 'any' +): LibraryGraph { + const accepted = new Set(kinds) + const selected = new Set(groups) + const memberships = new Map>() + if (selected.size) { + // Group membership filters do not depend on which relationship lines are shown. + graph.edges.forEach(({ source, target }) => { + const group = graph.nodes[target].id + if (!selected.has(group)) return + const values = memberships.get(source) ?? new Set() + values.add(group) + memberships.set(source, values) + }) + } + const included = new Set( + graph.nodes.flatMap((node, index) => + node.kind === 'link' && + (!platform || node.platform === platform) && + (!selected.size || + (match === 'all' ? memberships.get(index)?.size === selected.size : memberships.has(index))) + ? [index] + : [] + ) + ) + const edges = graph.edges.filter((edge) => accepted.has(edge.kind) && included.has(edge.source)) + let scope: Set | null = null + if (focus) { + const center = graph.nodes.findIndex((node) => node.id === focus) + scope = new Set([center]) + // Two hops from a link includes other links with the same known relationship. + for (let hop = 0; hop < (graph.nodes[center]?.kind === 'link' ? 2 : 1); hop++) { + const previous = new Set(scope) + edges.forEach((edge) => { + if (previous.has(edge.source)) scope!.add(edge.target) + if (previous.has(edge.target)) scope!.add(edge.source) + }) + } + } + const chosenEdges = edges.filter( + (edge) => !scope || (scope.has(edge.source) && scope.has(edge.target)) + ) + const hubs = new Set(chosenEdges.map((edge) => edge.target)) + const indices = new Map() + const nodes = graph.nodes.filter((node, i) => { + const visible = node.kind === 'link' ? included.has(i) && (!scope || scope.has(i)) : hubs.has(i) + if (visible) indices.set(i, indices.size) + return visible + }) + return { + ...graph, + nodes, + linkCount: nodes.filter((node) => node.kind === 'link').length, + edges: chosenEdges + .filter((edge) => indices.has(edge.source) && indices.has(edge.target)) + .map((edge) => ({ + ...edge, + source: indices.get(edge.source)!, + target: indices.get(edge.target)! + })) + } +} + +export function initialPositions(graph: Pick): Float32Array { + const result = new Float32Array(graph.nodes.length * 3) + const hubs = graph.nodes.flatMap((node, i) => (node.kind !== 'link' ? [i] : [])) + const radius = Math.max(80, Math.cbrt(graph.nodes.length) * 28) + graph.nodes.forEach((node, i) => { + let hash = 2166136261 + for (const character of node.id) hash = Math.imul(hash ^ character.charCodeAt(0), 16777619) + for (let axis = 0; axis < 3; axis++) { + hash = Math.imul(hash ^ (hash >>> 16), 2246822507) + result[i * 3 + axis] = ((hash >>> 0) / 4294967296 - 0.5) * radius * 2 + } + const length = Math.hypot(result[i * 3], result[i * 3 + 1], result[i * 3 + 2]) || 1 + const scale = (radius * Math.cbrt((hash >>> 0) / 4294967296)) / length + for (let axis = 0; axis < 3; axis++) result[i * 3 + axis] *= scale + }) + hubs.forEach((index, i) => { + const y = 1 - (2 * (i + 0.5)) / hubs.length + const ring = Math.sqrt(1 - y * y) + result.set( + [Math.cos(i * 2.399963) * ring * radius, y * radius, Math.sin(i * 2.399963) * ring * radius], + index * 3 + ) + }) + const weights = new Float32Array(graph.nodes.length) + const centers = new Float32Array(result.length) + graph.edges.forEach(({ source, target }) => { + weights[source]++ + for (let axis = 0; axis < 3; axis++) centers[source * 3 + axis] += result[target * 3 + axis] + }) + graph.nodes.forEach((_node, i) => { + if (!weights[i]) return + for (let axis = 0; axis < 3; axis++) + result[i * 3 + axis] = centers[i * 3 + axis] / weights[i] + result[i * 3 + axis] * 0.12 + }) + return result +} diff --git a/apps/electron/src/renderer/components/library-graph/navigation.test.ts b/apps/electron/src/renderer/components/library-graph/navigation.test.ts new file mode 100644 index 000000000..755f93d42 --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/navigation.test.ts @@ -0,0 +1,77 @@ +import type { LibraryGraph } from '../../../shared/library-graph' +import { expect, it } from 'vitest' +import { graphGroups, graphSearchIndex, searchGraph } from './navigation' + +const graph: LibraryGraph = { + nodes: [ + { + id: 'a', + kind: 'link', + label: 'Building better learning tools', + platform: 'youtube', + url: 'https://youtube.com/watch?v=abc123' + }, + { + id: 'b', + kind: 'link', + label: 'Learning', + platform: 'github', + url: 'https://github.com/example/learning' + }, + { id: 'c', kind: 'link', label: 'Café 道', platform: 'generic' }, + { id: 'tag', kind: 'tag', label: 'learning', platform: '' }, + { id: 'list', kind: 'collection', label: 'Learning tools', platform: 'youtube' }, + { id: 'category', kind: 'category', label: 'Education', platform: 'youtube' } + ], + edges: [ + { source: 0, target: 3, kind: 'tag', evidence: 'metadata' }, + { source: 1, target: 3, kind: 'tag', evidence: 'metadata' }, + { source: 0, target: 4, kind: 'collection', evidence: 'import' }, + { source: 0, target: 5, kind: 'category', evidence: 'metadata' } + ], + resourceCount: 3, + linkCount: 3, + builtAt: 0, + warnings: [] +} + +it('counts unique links for every group within the source, sorted by size', () => { + const duplicate = { ...graph, edges: [...graph.edges, graph.edges[0]] } + expect(graphGroups(duplicate, '').map(({ node, count }) => [node.id, count])).toEqual([ + ['tag', 2], + ['category', 1], + ['list', 1] + ]) + expect(graphGroups(graph, 'github').map(({ node, count }) => [node.id, count])).toEqual([ + ['tag', 1] + ]) + expect(graphGroups(graph, 'instagram')).toEqual([]) +}) + +it('ranks exact titles and groups above prefixes and partial words', () => { + const result = searchGraph(graphSearchIndex(graph), 'learning') + expect(result.items.map(({ node }) => node.id)).toEqual(['tag', 'b', 'list', 'a']) + expect(result.total).toBe(4) +}) + +it('finds nonadjacent title words, URL ids, accents, and Unicode labels', () => { + const index = graphSearchIndex(graph) + expect(searchGraph(index, 'tools building').items.map(({ node }) => node.id)).toEqual(['a']) + expect(searchGraph(index, 'ABC123').items[0].node.id).toBe('a') + expect(searchGraph(index, 'cafe 道').items[0].node.id).toBe('c') + expect(searchGraph(index, 'learning abc123').items[0].node.id).toBe('a') +}) + +it('suggests popular groups for an empty query and bounds rendering without losing match totals', () => { + const index = graphSearchIndex(graph) + expect(searchGraph(index, ' ').items.map(({ node }) => node.id)).toEqual([ + 'tag', + 'category', + 'list' + ]) + const bounded = searchGraph(index, 'learning', 2) + expect(bounded.total).toBe(4) + expect(bounded.items).toHaveLength(2) + expect(searchGraph(index, 'no such match')).toEqual({ total: 0, items: [] }) + expect(searchGraph([], '')).toEqual({ total: 0, items: [] }) +}) diff --git a/apps/electron/src/renderer/components/library-graph/navigation.ts b/apps/electron/src/renderer/components/library-graph/navigation.ts new file mode 100644 index 000000000..d9c62ac4a --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/navigation.ts @@ -0,0 +1,71 @@ +import type { LibraryGraph, LibraryGraphNode } from '../../../shared/library-graph' + +export const groupNames = { + collection: 'Playlists & collections', + tag: 'Tags & topics', + category: 'Categories', + creator: 'Creator names' +} + +export const normalizeSearch = (value: string) => + value.normalize('NFKD').replace(/\p{M}/gu, '').toLocaleLowerCase().trim() + +export type GraphGroup = { node: LibraryGraphNode; count: number } + +export function graphGroups(graph: LibraryGraph, platform: string): GraphGroup[] { + const members = new Map>() + graph.edges.forEach(({ source, target }) => { + if (platform && graph.nodes[source].platform !== platform) return + const links = members.get(target) ?? new Set() + links.add(source) + members.set(target, links) + }) + return [...members] + .map(([index, links]) => ({ node: graph.nodes[index], count: links.size })) + .sort((a, b) => b.count - a.count || a.node.label.localeCompare(b.node.label)) +} + +export function graphSearchIndex(graph: LibraryGraph) { + const degrees = new Uint32Array(graph.nodes.length) + graph.edges.forEach(({ source, target }) => { + degrees[source]++ + degrees[target]++ + }) + return graph.nodes.map((node, index) => ({ + node, + label: normalizeSearch(node.label), + url: normalizeSearch(node.url ?? ''), + count: degrees[index] + })) +} + +export function searchGraph(index: ReturnType, query: string, limit = 12) { + const text = normalizeSearch(query) + const terms = text.split(/\s+/u) + const matches = index.flatMap((entry) => { + if (!text) return entry.node.kind === 'link' ? [] : [{ ...entry, rank: 0 }] + if (!terms.every((term) => entry.label.includes(term) || entry.url.includes(term))) return [] + const rank = + entry.label === text + ? 0 + : entry.label.startsWith(text) + ? 1 + : terms.every((term) => + entry.label.split(/[^\p{L}\p{N}]+/u).some((word) => word.startsWith(term)) + ) + ? 2 + : terms.every((term) => entry.label.includes(term)) + ? 3 + : 4 + return [{ ...entry, rank }] + }) + matches.sort( + (a, b) => + a.rank - b.rank || + Number(a.node.kind === 'link') - Number(b.node.kind === 'link') || + (a.node.kind === 'link' ? 0 : b.count - a.count) || + a.label.localeCompare(b.label) || + a.node.id.localeCompare(b.node.id) + ) + return { total: matches.length, items: matches.slice(0, limit) } +} diff --git a/apps/electron/src/renderer/components/library-graph/scene.ts b/apps/electron/src/renderer/components/library-graph/scene.ts new file mode 100644 index 000000000..145edfbae --- /dev/null +++ b/apps/electron/src/renderer/components/library-graph/scene.ts @@ -0,0 +1,326 @@ +import type { LibraryGraph } from '../../../shared/library-graph' +import { + Scene, + Color, + PerspectiveCamera, + WebGLRenderer, + BufferGeometry, + BufferAttribute, + ShaderMaterial, + Points, + LineBasicMaterial, + LineSegments, + Vector2, + Vector3, + Raycaster, + Box3 +} from 'three' +import { OrbitControls } from 'three/addons/controls/OrbitControls.js' +import { graphColor, graphAdjacency, initialPositions } from './model' + +export function createGraphScene( + host: HTMLElement, + graph: LibraryGraph, + callbacks: { + hover: (index: number | null) => void + select: (index: number) => void + failure: (message: string) => void + } +) { + const renderer = new WebGLRenderer({ + antialias: true, + alpha: false, + powerPreference: 'low-power' + }) + renderer.setPixelRatio(Math.min(window.devicePixelRatio, 2)) + renderer.setClearColor('#080f1d') + const canvas = renderer.domElement + canvas.className = + 'absolute inset-0 h-full w-full outline-none focus-visible:ring-2 focus-visible:ring-violet-400' + canvas.setAttribute( + 'aria-label', + '3D link graph. Drag to orbit, scroll to zoom. Use the search and connection list to select links with the keyboard.' + ) + canvas.tabIndex = 0 + host.appendChild(canvas) + const scene = new Scene() + const camera = new PerspectiveCamera(50, 1, 0.1, 100000) + const controls = new OrbitControls(camera, canvas) + controls.minDistance = 12 + controls.maxDistance = 30000 + controls.listenToKeyEvents(canvas) + const positions = initialPositions(graph) + const colors = new Float32Array(positions.length) + const sizes = new Float32Array(graph.nodes.length) + const emphasis = new Float32Array(graph.nodes.length).fill(1) + graph.nodes.forEach((node, i) => { + new Color(graphColor(node.kind, node.platform)).toArray(colors, i * 3) + sizes[i] = + graph.nodes.length > 10000 ? (node.kind === 'link' ? 2 : 4) : node.kind === 'link' ? 5 : 9 + }) + const geometry = new BufferGeometry() + geometry.setAttribute('position', new BufferAttribute(positions, 3)) + geometry.setAttribute('color', new BufferAttribute(colors, 3)) + geometry.setAttribute('size', new BufferAttribute(sizes, 1)) + geometry.setAttribute('emphasis', new BufferAttribute(emphasis, 1)) + const material = new ShaderMaterial({ + uniforms: { pixelRatio: { value: renderer.getPixelRatio() } }, + vertexShader: `attribute vec3 color; attribute float size; attribute float emphasis; + uniform float pixelRatio; varying vec3 vColor; varying float vAlpha; + void main() { vColor = color; vAlpha = emphasis; + gl_Position = projectionMatrix * modelViewMatrix * vec4(position, 1.0); + gl_PointSize = size * pixelRatio * (emphasis > 0.99 ? 1.0 : 0.8); }`, + fragmentShader: `varying vec3 vColor; varying float vAlpha; + void main() { float d = length(gl_PointCoord - vec2(0.5)); if (d > 0.5) discard; + gl_FragColor = vec4(vColor, vAlpha * smoothstep(0.5, 0.28, d)); }`, + transparent: true, + depthWrite: false + }) + const points = new Points(geometry, material) + points.frustumCulled = false + scene.add(points) + const edgeGeometry = new BufferGeometry() + const edgePositions = new Float32Array(graph.edges.length * 6) + edgeGeometry.setAttribute('position', new BufferAttribute(edgePositions, 3)) + const edgeMaterial = new LineBasicMaterial({ + color: '#6682aa', + transparent: true, + opacity: graph.edges.length > 10000 ? 0.025 : 0.15, + depthWrite: false + }) + const lines = new LineSegments(edgeGeometry, edgeMaterial) + lines.frustumCulled = false + scene.add(lines) + const highlightGeometry = new BufferGeometry() + const highlightPositions = new Float32Array(graph.edges.length * 6) + highlightGeometry.setAttribute('position', new BufferAttribute(highlightPositions, 3)) + highlightGeometry.setDrawRange(0, 0) + const highlightMaterial = new LineBasicMaterial({ + color: '#d8b4fe', + transparent: true, + opacity: 0.7, + depthWrite: false + }) + const highlight = new LineSegments(highlightGeometry, highlightMaterial) + highlight.frustumCulled = false + scene.add(highlight) + const adjacency = graphAdjacency(graph) + let active: number | null = null + let hover: number | null = null + let disposed = false + let frame = 0 + const point = new Vector3() + const labels = graph.nodes + .map((node, index) => ({ node, index, count: adjacency[index].length })) + .filter(({ node }) => node.kind !== 'link') + .sort((a, b) => b.count - a.count) + .slice(0, 12) + .map(({ node, index }) => { + const element = document.createElement('div') + element.className = + 'pointer-events-none absolute max-w-36 truncate rounded bg-slate-950/75 px-1.5 py-0.5 text-[10px] text-slate-300' + element.textContent = node.label + host.appendChild(element) + return { index, element } + }) + const render = () => { + frame = 0 + if (disposed) return + renderer.render(scene, camera) + const occupied: { x: number; y: number }[] = [] + labels.forEach(({ index, element }) => { + point.fromArray(positions, index * 3).project(camera) + const x = ((point.x + 1) * host.clientWidth) / 2 + const y = ((1 - point.y) * host.clientHeight) / 2 + const hidden = + Math.abs(point.z) > 1 || + x < 0 || + x > host.clientWidth - 100 || + y < 0 || + y > host.clientHeight - 20 || + occupied.some((other) => Math.abs(other.x - x) < 100 && Math.abs(other.y - y) < 22) + element.style.display = hidden ? 'none' : 'block' + if (!hidden) { + occupied.push({ x, y }) + element.style.transform = `translate(${x + 8}px, ${y - 10}px)` + } + }) + } + const invalidate = () => { + if (!frame && !disposed) frame = requestAnimationFrame(render) + } + const updateEdges = () => { + graph.edges.forEach((edge, i) => { + for (let axis = 0; axis < 3; axis++) { + edgePositions[i * 6 + axis] = positions[edge.source * 3 + axis] + edgePositions[i * 6 + axis + 3] = positions[edge.target * 3 + axis] + } + }) + edgeGeometry.attributes.position.needsUpdate = true + const selectedEdges = + active === null + ? [] + : graph.edges.filter((edge) => edge.source === active || edge.target === active) + selectedEdges.forEach((edge, i) => { + highlightPositions.set(positions.subarray(edge.source * 3, edge.source * 3 + 3), i * 6) + highlightPositions.set(positions.subarray(edge.target * 3, edge.target * 3 + 3), i * 6 + 3) + }) + highlightGeometry.setDrawRange(0, selectedEdges.length * 2) + highlightGeometry.attributes.position.needsUpdate = true + } + const fit = () => { + const box = new Box3().setFromBufferAttribute(geometry.attributes.position as BufferAttribute) + if (box.isEmpty()) box.set(new Vector3(-50, -50, -50), new Vector3(50, 50, 50)) + const center = box.getCenter(new Vector3()) + const radius = Math.max(40, box.getSize(new Vector3()).length() / 2) + const fov = 2 * Math.atan(Math.tan((camera.fov * Math.PI) / 360) * Math.min(1, camera.aspect)) + const distance = (radius / Math.sin(fov / 2)) * 1.05 + controls.target.copy(center) + camera.position + .copy(center) + .add(new Vector3(0.18, 0.12, 1).normalize().multiplyScalar(distance)) + controls.update() + invalidate() + } + const resize = () => { + const width = Math.max(1, host.clientWidth), + height = Math.max(1, host.clientHeight) + renderer.setSize(width, height, false) + camera.aspect = width / height + camera.updateProjectionMatrix() + invalidate() + } + const observer = new ResizeObserver(resize) + observer.observe(host) + controls.addEventListener('change', invalidate) + resize() + updateEdges() + fit() + const ray = new Raycaster() + const pointer = new Vector2() + let down: { x: number; y: number } | null = null + let lastPick = 0 + const pick = (event: PointerEvent): number | null => { + const rect = canvas.getBoundingClientRect() + pointer.set( + ((event.clientX - rect.left) / rect.width) * 2 - 1, + (-(event.clientY - rect.top) / rect.height) * 2 + 1 + ) + ray.setFromCamera(pointer, camera) + ray.params.Points.threshold = + (camera.position.distanceTo(controls.target) * Math.tan((camera.fov * Math.PI) / 360) * 16) / + rect.height + let best: number | null = null, + score = 65 + for (const intersection of ray.intersectObject(points)) { + const index = intersection.index + if (index === undefined) continue + point.fromArray(positions, index * 3).project(camera) + if (Math.abs(point.z) > 1) continue + const dx = ((point.x - pointer.x) * rect.width) / 2 + const dy = ((point.y - pointer.y) * rect.height) / 2 + const distance = dx * dx + dy * dy + if (distance < score) { + best = index + score = distance + } + } + return best + } + const move = (event: PointerEvent) => { + if (event.buttons || performance.now() - lastPick < 60) return + lastPick = performance.now() + const next = pick(event) + if (next !== hover) { + hover = next + callbacks.hover(next) + canvas.style.cursor = next === null ? 'grab' : 'pointer' + } + } + const leave = () => { + hover = null + callbacks.hover(null) + } + const pointerDown = (event: PointerEvent) => { + down = { x: event.clientX, y: event.clientY } + } + const pointerUp = (event: PointerEvent) => { + if (down && Math.hypot(event.clientX - down.x, event.clientY - down.y) < 5) { + const index = pick(event) + if (index !== null) callbacks.select(index) + } + down = null + } + const contextLost = (event: Event) => { + event.preventDefault() + callbacks.failure( + 'The 3D graphics context was lost. Reload the graph to recover. Your Library is unchanged.' + ) + } + canvas.addEventListener('pointermove', move) + canvas.addEventListener('pointerleave', leave) + canvas.addEventListener('pointerdown', pointerDown) + canvas.addEventListener('pointerup', pointerUp) + canvas.addEventListener('webglcontextlost', contextLost) + return { + fit, + positions(next: Float32Array) { + if (next.length !== positions.length || next.some((value) => !Number.isFinite(value))) { + callbacks.failure('The layout returned invalid positions. Reload the graph to try again.') + return + } + positions.set(next) + geometry.attributes.position.needsUpdate = true + geometry.computeBoundingSphere() + updateEdges() + invalidate() + }, + active(index: number | null) { + active = index + emphasis.fill(index === null ? 1 : 0.13) + if (index !== null) { + emphasis[index] = 1 + adjacency[index].forEach((neighbor) => { + emphasis[neighbor] = 1 + }) + } + geometry.attributes.emphasis.needsUpdate = true + updateEdges() + invalidate() + }, + focus(index: number) { + const center = new Vector3().fromArray(positions, index * 3) + const direction = camera.position.clone().sub(controls.target).normalize() + controls.target.copy(center) + camera.position.copy(center).add(direction.multiplyScalar(240)) + controls.update() + invalidate() + }, + zoom(factor: number) { + camera.position.sub(controls.target).multiplyScalar(factor).add(controls.target) + controls.update() + invalidate() + }, + dispose() { + disposed = true + cancelAnimationFrame(frame) + observer.disconnect() + controls.dispose() + canvas.removeEventListener('pointermove', move) + canvas.removeEventListener('pointerleave', leave) + canvas.removeEventListener('pointerdown', pointerDown) + canvas.removeEventListener('pointerup', pointerUp) + canvas.removeEventListener('webglcontextlost', contextLost) + geometry.dispose() + material.dispose() + edgeGeometry.dispose() + edgeMaterial.dispose() + highlightGeometry.dispose() + highlightMaterial.dispose() + renderer.dispose() + renderer.forceContextLoss() + canvas.remove() + labels.forEach(({ element }) => element.remove()) + } + } +} diff --git a/apps/electron/src/renderer/index.html b/apps/electron/src/renderer/index.html index d25be4f49..7389d268a 100644 --- a/apps/electron/src/renderer/index.html +++ b/apps/electron/src/renderer/index.html @@ -43,6 +43,6 @@
- + diff --git a/apps/electron/src/renderer/lib/ipc-node-storage.ts b/apps/electron/src/renderer/lib/ipc-node-storage.ts index 31a9bcc16..636ef3281 100644 --- a/apps/electron/src/renderer/lib/ipc-node-storage.ts +++ b/apps/electron/src/renderer/lib/ipc-node-storage.ts @@ -9,6 +9,8 @@ import type { ContentId, DID } from '@xnetjs/core' import type { + ApplyNodeBatchInput, + ApplyNodeBatchResult, NodeStorageAdapter, NodeState, NodeChange, @@ -19,6 +21,7 @@ import type { SchemaIRI, PropertyTimestamp } from '@xnetjs/data' +import { serializeNodeBatch, type SerializedNodeBatch } from '../../shared/node-batch' // Debug logging - controlled by localStorage flag (same as sync debug) function log(...args: unknown[]): void { @@ -52,6 +55,10 @@ export class IPCNodeStorageAdapter implements NodeStorageAdapter { // Change Log Operations // ========================================================================== + async applyNodeBatch(input: ApplyNodeBatchInput): Promise { + return window.xnetNodes.applyNodeBatch(serializeNodeBatch(input)) + } + async appendChange(change: NodeChange): Promise { log('appendChange()', change.payload.nodeId) await window.xnetNodes.appendChange(serializeChange(change)) @@ -328,6 +335,7 @@ declare global { } export interface XNetNodesAPI { + applyNodeBatch(input: SerializedNodeBatch): Promise // Change log operations appendChange(change: unknown): Promise getChanges(nodeId: string): Promise diff --git a/apps/electron/src/renderer/lib/ipc-sync-manager.ts b/apps/electron/src/renderer/lib/ipc-sync-manager.ts index 8c420bbce..be8f1db6d 100644 --- a/apps/electron/src/renderer/lib/ipc-sync-manager.ts +++ b/apps/electron/src/renderer/lib/ipc-sync-manager.ts @@ -52,6 +52,7 @@ interface DevToolsEventBus { } export interface IPCSyncManager extends SyncManager { + flushDocuments(): Promise /** Instrument with devtools event bus for sync monitoring */ instrument(eventBus: DevToolsEventBus): () => void /** Set identity for signing outgoing updates */ @@ -398,6 +399,13 @@ export function createIPCSyncManager(): IPCSyncManager { window.xnetBSM.untrack(nodeId) }, + async flushDocuments(): Promise { + await Promise.all([...pendingAcquires.values()]) + for (const [nodeId, doc] of docs) { + await window.xnetNodes.setDocumentContent(nodeId, Array.from(Y.encodeStateAsUpdate(doc))) + } + }, + async acquire(nodeId: string): Promise { // Reuse existing mirror if already fully acquired const existing = docs.get(nodeId) diff --git a/apps/electron/src/renderer/lib/use-native-node-changes.ts b/apps/electron/src/renderer/lib/use-native-node-changes.ts new file mode 100644 index 000000000..e5a28c144 --- /dev/null +++ b/apps/electron/src/renderer/lib/use-native-node-changes.ts @@ -0,0 +1,53 @@ +import { useNodeStore } from '@xnetjs/react/internal' +import { useEffect, useState } from 'react' + +/** Native import/enrichment writes share storage with the renderer's live store. */ +export function useNativeNodeChanges(): string | null { + const { store, isReady } = useNodeStore() + const [error, setError] = useState(null) + useEffect(() => { + if (!store || !isReady) return + const pending = new Set() + let active = true + let running = false + const drain = async () => { + if (running || !active) return + running = true + try { + while (active && pending.size) { + const ids = Array.from(pending).slice(0, 100) + ids.forEach((id) => pending.delete(id)) + try { + await store.refreshPersistedNodes(ids) + } catch (error) { + ids.forEach((id) => pending.add(id)) + throw error + } + } + if (active) setError(null) + } catch (error) { + if (active) setError(`Could not refresh saved library changes: ${String(error)}`) + } finally { + running = false + } + } + const unsubscribe = window.xnetNodes.onChange(({ changes }) => { + for (const change of changes) { + const payload = (change as { payload?: { nodeId?: unknown } } | null)?.payload + if (typeof payload?.nodeId !== 'string') { + setError('A saved change notification was unreadable. Restart to reload the workspace.') + return + } + pending.add(payload.nodeId) + } + void drain() + }) + const retryAfterBarrier = window.xnet.onResumeEditing(() => void drain()) + return () => { + active = false + unsubscribe() + retryAfterBarrier() + } + }, [store, isReady]) + return error +} diff --git a/apps/electron/src/renderer/main.tsx b/apps/electron/src/renderer/main.tsx index 5e7d4d721..1cfab5753 100644 --- a/apps/electron/src/renderer/main.tsx +++ b/apps/electron/src/renderer/main.tsx @@ -12,6 +12,7 @@ import { XNetDevToolsProvider, useDevTools } from '@xnetjs/devtools' import { BlobProvider } from '@xnetjs/editor/react' import { identityFromPrivateKey } from '@xnetjs/identity' import { XNetProvider } from '@xnetjs/react' +import { flushDocumentWrites } from '@xnetjs/react/internal' import { ChunkManager } from '@xnetjs/storage' import { ConsentManager, @@ -24,6 +25,7 @@ import React, { useEffect } from 'react' import { createRoot, type Root } from 'react-dom/client' import { Awareness, applyAwarenessUpdate, encodeAwarenessUpdate } from 'y-protocols/awareness' import * as Y from 'yjs' +import { captureDesktopSettings } from '../shared/desktop-settings' import { App } from './App' import { ShellErrorBoundary } from './components/ShellErrorBoundary' import { configuredHubUrl } from './lib/hub-url' @@ -889,6 +891,22 @@ async function init() { // Thumbnails are generated at attach time in the renderer (0385 W4). const blobService = new BlobService(chunkManager, { generateThumbnails: true }) + const removeFlush = window.xnet.onFlushDocuments(async () => { + const root = document.getElementById('root') + if (root) root.inert = true + await flushDocumentWrites() + await ipcSyncManager.flushDocuments() + await window.xnet.saveDesktopSettings(captureDesktopSettings(localStorage)) + }) + const removeResume = window.xnet.onResumeEditing(() => { + const root = document.getElementById('root') + if (root) root.inert = false + }) + import.meta.hot?.dispose(() => { + removeFlush() + removeResume() + }) + // Listen for devtools toggle from main process menu. // Keep only one active listener across HMR reloads. window.__xnetDevToolsToggleCleanup?.() diff --git a/apps/electron/src/renderer/shell/desktop-platform.test.ts b/apps/electron/src/renderer/shell/desktop-platform.test.ts index 55009a948..0743a6457 100644 --- a/apps/electron/src/renderer/shell/desktop-platform.test.ts +++ b/apps/electron/src/renderer/shell/desktop-platform.test.ts @@ -16,7 +16,7 @@ function makeDeps(): DesktopNavDeps & { calls: string[] } { calls, shellState: { kind: 'canvas-home' }, returnHome: () => void calls.push('shell:return-home'), - openDocument: (id) => void calls.push(`doc:${id}`), + openDocument: (id, type) => void calls.push(`doc:${id}:${type}`), openAssistant: () => void calls.push('assistant'), openSettings: () => void calls.push('settings'), openMeetings: () => void calls.push('meetings'), @@ -53,7 +53,7 @@ describe('navigateShell', () => { navigateShell({ kind: 'node', nodeType: 'page', nodeId: 'p1' }, deps) navigateShell({ kind: 'node', nodeType: 'settings', nodeId: '' }, deps) navigateShell({ kind: 'home' }, deps) - expect(deps.calls).toEqual(['doc:p1', 'settings', 'shell:return-home']) + expect(deps.calls).toEqual(['doc:p1:page', 'settings', 'shell:return-home']) }) it('maps the path escape hatch onto desktop surfaces', () => { diff --git a/apps/electron/src/renderer/shell/desktop-platform.ts b/apps/electron/src/renderer/shell/desktop-platform.ts index 98171b0a8..5fccfd569 100644 --- a/apps/electron/src/renderer/shell/desktop-platform.ts +++ b/apps/electron/src/renderer/shell/desktop-platform.ts @@ -27,7 +27,7 @@ export interface DesktopNavDeps { /** The shell's own home transition (viewport/timer-aware). */ returnHome: () => void /** Open a document by id through the shell's own resolution (type lookup + canvas glide). */ - openDocument: (docId: string) => void + openDocument: (docId: string, type?: 'page' | 'database' | 'canvas') => void openAssistant: () => void openSettings: () => void openMeetings: () => void @@ -84,7 +84,7 @@ export function navigateShell(target: NavTarget, deps: DesktopNavDeps): boolean case 'page': case 'database': case 'canvas': - deps.openDocument(target.nodeId) + deps.openDocument(target.nodeId, target.nodeType) return true case 'settings': deps.openSettings() diff --git a/apps/electron/src/renderer/shell/use-document-shell.ts b/apps/electron/src/renderer/shell/use-document-shell.ts index 983ad54ef..dcaf81153 100644 --- a/apps/electron/src/renderer/shell/use-document-shell.ts +++ b/apps/electron/src/renderer/shell/use-document-shell.ts @@ -73,7 +73,7 @@ export interface DocumentShell { docType: Exclude, animateFromCanvas: boolean ) => void - handleOpenDocument: (docId: string) => void + handleOpenDocument: (docId: string, type?: DocType) => void handleCreateLinkedDocument: (type: Exclude) => Promise handleCreateCanvasNote: () => void handleReturnHome: () => void @@ -250,18 +250,21 @@ export function useDocumentShell(): DocumentShell { ) const handleOpenDocument = useCallback( - (docId: string) => { + (docId: string, type?: DocType) => { const document = documents.find((entry) => entry.id === docId) - if (!document) return + // Typed links can target a newly imported note or a Page outside the + // recent-document query's 100-row window. + const documentType = document?.type ?? type + if (!documentType) return - if (document.type === 'canvas') { - setHomeCanvasId(document.id) + if (documentType === 'canvas') { + setHomeCanvasId(docId) transitionShell({ type: 'return-home' }) - setActiveNodeId(document.id) + setActiveNodeId(docId) return } - focusDocument(document.id, document.type, true) + focusDocument(docId, documentType, true) }, [documents, focusDocument, setActiveNodeId, transitionShell] ) @@ -271,9 +274,11 @@ export function useDocumentShell(): DocumentShell { clearTransitionTimer() try { - const schema = type === 'page' ? PageSchema : DatabaseSchema const title = type === 'page' ? 'Untitled Page' : 'Untitled Database' - const newDocument = await create(schema, { title }) + const newDocument = + type === 'page' + ? await create(PageSchema, { title }) + : await create(DatabaseSchema, { title }) if (!newDocument) return setPendingCanvasInsert({ diff --git a/apps/electron/src/renderer/shell/workbench-host.tsx b/apps/electron/src/renderer/shell/workbench-host.tsx index 7ff494fcc..e6b1e0d32 100644 --- a/apps/electron/src/renderer/shell/workbench-host.tsx +++ b/apps/electron/src/renderer/shell/workbench-host.tsx @@ -142,6 +142,11 @@ export function useDesktopWorkbenchHost( return useMemo( () => ({ + library: { + capture: window.xnet.libraryCapture, + lookup: window.xnet.libraryLookup, + closed: window.xnet.closeLibraryCapture + }, logout: async () => { // Desktop has no lock/relock flow yet (identity lives in the OS keychain); // loud so a dead menu item is diagnosable, not mysterious. diff --git a/apps/electron/src/shared/desktop-settings.ts b/apps/electron/src/shared/desktop-settings.ts new file mode 100644 index 000000000..5915e7b94 --- /dev/null +++ b/apps/electron/src/shared/desktop-settings.ts @@ -0,0 +1,87 @@ +/** Logical settings only: Chromium files and device-bound sessions are not portable. */ +export const DESKTOP_SETTING_KEYS = [ + 'xnet.library.capture-draft.v1', + 'xnet:workbench:v1', + 'xnet:hub-url', + 'xnet-electron-theme', + 'xnet-electron-theme-variant', + 'xnet-electron-theme-density', + 'xnet-electron-theme-tokens', + 'xnet:ai-api-key', + 'xnet:ai-cloud-provider', + 'xnet:ai-model', + 'xnet:ai-local-base-url', + 'xnet:ai-tier', + 'xnet:ai-semantic-search', + 'xnet:ai-writes', + 'xnet:ai:assist-mode', + 'xnet:experiment:quiet-default', + 'xnet:experiment:desk-radial', + 'xnet:meetings:consent', + 'xnet:meetings:engine', + 'xnet:meetings:byo-endpoint', + 'xnet:data-workspace:dismissed-patterns', + 'xnet:telemetry:consent' +] as const + +export type DesktopSettings = { + version: 1 + values: Record<(typeof DESKTOP_SETTING_KEYS)[number], string | null> +} + +export type SettingsRecovery = { settings: DesktopSettings | null; restoreId: string | null } +export const SETTINGS_LIMIT = 16 * 1024 * 1024 + +export function validateDesktopSettings(value: unknown): DesktopSettings { + if (!value || typeof value !== 'object' || !('version' in value) || value.version !== 1) + throw new Error('Unsupported desktop settings version') + if (!('values' in value) || !value.values || typeof value.values !== 'object') + throw new Error('Invalid desktop settings') + const entries = Object.entries(value.values) + if ( + entries.length !== DESKTOP_SETTING_KEYS.length || + entries.some( + ([key, entry]) => + !(DESKTOP_SETTING_KEYS as readonly string[]).includes(key) || + (entry !== null && typeof entry !== 'string') + ) || + new TextEncoder().encode(JSON.stringify(value)).byteLength > SETTINGS_LIMIT + ) + throw new Error('Incomplete, unsupported, or oversized desktop settings') + return value as DesktopSettings +} + +export function captureDesktopSettings(storage: Pick): DesktopSettings { + return validateDesktopSettings({ + version: 1, + values: Object.fromEntries(DESKTOP_SETTING_KEYS.map((key) => [key, storage.getItem(key)])) + }) +} + +const INITIALIZED = 'xnet:recovery:settings-initialized' +const APPLIED = 'xnet:recovery:settings-applied' + +/** Run before importing consumers. A failed application is retried in full on next boot. */ +export function restoreDesktopSettings( + storage: Pick, + recovery: SettingsRecovery +): void { + const restoring = + !storage.getItem(INITIALIZED) || + (recovery.restoreId !== null && storage.getItem(APPLIED) !== recovery.restoreId) + if (restoring && recovery.settings) { + const settings = validateDesktopSettings(recovery.settings) + for (const key of DESKTOP_SETTING_KEYS) { + const value = settings.values[key] + if (value === null) storage.removeItem(key) + else storage.setItem(key, value) + } + } + if (restoring && recovery.restoreId) { + // These authorize a session on this installation; they must be paired again. + storage.removeItem('xnet:ai-bridge-token') + storage.removeItem('xnet:ai-openrouter-verifier') + storage.setItem(APPLIED, recovery.restoreId) + } + storage.setItem(INITIALIZED, '1') +} diff --git a/apps/electron/src/shared/library-graph.ts b/apps/electron/src/shared/library-graph.ts new file mode 100644 index 000000000..cce245585 --- /dev/null +++ b/apps/electron/src/shared/library-graph.ts @@ -0,0 +1,29 @@ +import type { LibraryResource } from './library' + +export type LibraryGraphKind = 'link' | 'collection' | 'creator' | 'tag' | 'category' +export type LibraryGraphRelation = Exclude +export type LibraryGraphNode = { + id: string + kind: LibraryGraphKind + label: string + platform: string + url?: string +} +export type LibraryGraphEdge = { + source: number + target: number + kind: LibraryGraphRelation + evidence: 'import' | 'metadata' | 'hashtag' +} +export type LibraryGraph = { + nodes: LibraryGraphNode[] + edges: LibraryGraphEdge[] + resourceCount: number + linkCount: number + warnings: string[] + builtAt: number +} +export type LibraryGraphDetail = { + resource: LibraryResource | null + source: Record | null +} diff --git a/apps/electron/src/shared/library.ts b/apps/electron/src/shared/library.ts new file mode 100644 index 000000000..a649704bc --- /dev/null +++ b/apps/electron/src/shared/library.ts @@ -0,0 +1,106 @@ +export const LIBRARY_PROVIDER_VERSION = 'desktop-4/public-pages-1/captions-2/yt-dlp-2026.07.04' +export const CAPABILITIES = ['metadata', 'thumbnail', 'transcript', 'index'] as const +export type Capability = (typeof CAPABILITIES)[number] +export type WorkState = + | 'queued' + | 'running' + | 'complete' + | 'partial' + | 'retry' + | 'blocked' + | 'unavailable' + | 'not-applicable' +export type FieldCoverage = { state: 'complete' | 'unavailable' | 'partial'; reason?: string } +export type CaptionTrack = { + url: string + language: string + format: 'json3' | 'vtt' + autoGenerated: boolean +} +export type Cue = { startMs: number; durationMs: number; text: string } +export type LibraryMetadata = { + title?: string + description?: string + author?: string + thumbnailUrl?: string + durationSeconds?: number + language?: string + tracks?: CaptionTrack[] + fields: Record + provider: string + fetchedAt: number + evidence: unknown +} +export type LibraryResource = { + id: string + kind?: 'conversation' | 'archive-text' + platform: string + /** Queue routing derived from the source URL; archive provenance stays in platform. */ + networkPlatform?: string + platformContentId: string + url: string + title: string + sourceText: string + actor: string + privacy: string + notes?: { + id: string + title: string + text: string + url: string + author: string + pageId?: string + }[] + metadata?: LibraryMetadata + thumbnail?: { cid: string; contentType: string; bytes: number } + transcript?: { + cues: Cue[] + language: string + autoGenerated: boolean + source: 'captions' | 'local-asr' + fetchedAt: number + provider: string + evidence: string + } + addedAt: number +} +export type LibraryJob = { + resourceId: string + capability: Capability + version: string + language: string + state: WorkState + attempts: number + nextAt: number + reason: string | null +} +export type LibrarySearchResult = Omit & { + metadata?: Omit + snippet?: string + startMs?: number +} +export type LibraryStatus = { + paused: boolean + running: (LibraryJob & { title: string })[] + nextAt: number | null + resources: number + counts: { capability: Capability; state: WorkState; count: number }[] + recent: (LibraryJob & { title: string })[] + providerVersion: string +} + +export type CaptureInput = { + requestId: string + url: string + title: string + note: string + excerpt: string +} +export type CaptureResult = { resourceId: string; pageId: string; reusedResource: boolean } + +export type LibraryHelperStatus = { + state: 'missing' | 'ready' | 'damaged' | 'installing' | 'unsupported' + version: string + bytes: number + reason?: string +} diff --git a/apps/electron/src/shared/node-batch.ts b/apps/electron/src/shared/node-batch.ts new file mode 100644 index 000000000..bafa2d942 --- /dev/null +++ b/apps/electron/src/shared/node-batch.ts @@ -0,0 +1,32 @@ +import type { ApplyNodeBatchInput, NodeChange, NodeState } from '@xnetjs/data' + +/** Keep byte arrays explicit across the renderer, main, and utility boundaries. */ +export type SerializedNodeBatch = Omit & { + nodes: Array & { documentContent?: number[] }> + changes: Array & { signature: number[] }> +} + +export function serializeNodeBatch(input: ApplyNodeBatchInput): SerializedNodeBatch { + return { + ...input, + nodes: input.nodes.map((node) => ({ + ...node, + documentContent: node.documentContent ? Array.from(node.documentContent) : undefined + })), + changes: input.changes.map((change) => ({ ...change, signature: Array.from(change.signature) })) + } +} + +export function deserializeNodeBatch(input: SerializedNodeBatch): ApplyNodeBatchInput { + return { + ...input, + nodes: input.nodes.map((node) => ({ + ...node, + documentContent: node.documentContent ? new Uint8Array(node.documentContent) : undefined + })), + changes: input.changes.map((change) => ({ + ...change, + signature: new Uint8Array(change.signature) + })) + } +} diff --git a/apps/electron/src/shared/recovery.ts b/apps/electron/src/shared/recovery.ts new file mode 100644 index 000000000..8fc489a0a --- /dev/null +++ b/apps/electron/src/shared/recovery.ts @@ -0,0 +1,17 @@ +export type CheckpointManifest = { + format: 'xnet-desktop-checkpoint/1' + id: string + createdAt: string + appVersion: string + profile: string + identity: 'stored' | 'test' + sourceFingerprint?: string + pinned?: boolean + storageVersion?: number + files: { path: string; size: number; sha256: string }[] +} + +export type CheckpointListing = { + checkpoints: CheckpointManifest[] + unreadable: { id: string; reason: string }[] +} diff --git a/apps/electron/src/shared/social-import.ts b/apps/electron/src/shared/social-import.ts new file mode 100644 index 000000000..85dcdc7c5 --- /dev/null +++ b/apps/electron/src/shared/social-import.ts @@ -0,0 +1,48 @@ +import type { + ArchiveManifest, + SocialImportArchivePreview as SharedSocialImportArchivePreview, + SocialImportNodeDraft as SharedSocialImportNodeDraft, + SocialImportNodeDraftStreamResult, + SocialImportJobProgress +} from '@xnetjs/social/import/core' + +export type SocialImportArchivePreview = Omit & { + archivePath: string +} + +export type SocialImportNodeDraft = SharedSocialImportNodeDraft + +export type SocialImportStageRequest = { + archivePath: string + buckets?: string[] + includeSensitive?: boolean +} + +export type SocialImportStageResult = Omit & { + archive: SocialImportArchivePreview + stageId: string +} + +export type SocialImportCommitJobRequest = { + stageId: string + includeSourceRecords: boolean + authorDID: string + signingKey: number[] +} + +export type SocialImportCommitJobSummary = { + created: number + updated: number + batches: number +} + +export type SocialImportCommitJobSnapshot = SocialImportJobProgress & { + summary?: SocialImportCommitJobSummary +} +export type ElectronStagedSocialImport = Omit & { + archive: SocialImportArchivePreview + archivePath: string + manifest: ArchiveManifest + stageRequest: SocialImportStageRequest + importedAt: string +} diff --git a/apps/electron/src/storage/checkpoint-policy.test.ts b/apps/electron/src/storage/checkpoint-policy.test.ts new file mode 100644 index 000000000..7a12f878c --- /dev/null +++ b/apps/electron/src/storage/checkpoint-policy.test.ts @@ -0,0 +1,90 @@ +import { expect, it, vi } from 'vitest' +import { + CHECKPOINT_INTERVAL_MS, + createCheckpointSchedule, + retainedCheckpointIds +} from './checkpoint-policy' + +const now = Date.parse('2026-09-29T12:00:00Z') +const at = (minutesAgo: number) => new Date(now - minutesAgo * 60_000).toISOString() + +it('keeps recent, daily, weekly, pinned, and future-dated points', () => { + const points = [ + { id: 'newest', createdAt: at(1) }, + { id: 'second', createdAt: at(2) }, + { id: 'duplicate', createdAt: at(3) }, + { id: 'previous-quarter', createdAt: at(16) }, + { id: 'daily', createdAt: at(2 * 1440) }, + { id: 'same-day', createdAt: at(2 * 1440 + 1) }, + { id: 'weekly', createdAt: at(14 * 1440) }, + { id: 'expired', createdAt: at(40 * 1440) }, + { id: 'pinned', createdAt: at(50 * 1440), pinned: true } + ] + expect([...retainedCheckpointIds(points, now)].sort()).toEqual([ + 'daily', + 'newest', + 'pinned', + 'previous-quarter', + 'second', + 'weekly' + ]) + expect(retainedCheckpointIds([{ id: 'future', createdAt: at(-5) }], now).has('future')).toBe(true) +}) + +it('checks changed data only when overdue and retries failed copies', async () => { + const latest = { createdAt: at(5), sourceFingerprint: 'old' } + const fingerprint = vi.fn(async () => 'changed') + const create = vi.fn(async () => {}) + const failed = vi.fn() + const tick = createCheckpointSchedule({ + now: () => now, + busy: () => false, + latest: async () => latest, + fingerprint, + create, + failed + }) + await tick() + expect(fingerprint).not.toHaveBeenCalled() + latest.createdAt = at(16) + latest.sourceFingerprint = 'changed' + await tick() + expect(create).not.toHaveBeenCalled() + latest.sourceFingerprint = 'old' + create.mockRejectedValueOnce(new Error('disk full')) + await tick() + expect(failed).toHaveBeenCalledWith(expect.objectContaining({ message: 'disk full' })) + await tick() + expect(create).toHaveBeenCalledTimes(2) +}) + +it('does not overlap checks or run while import or recovery is busy', async () => { + let release!: () => void + let busy = true + const create = vi.fn(async () => {}) + const latest = vi.fn( + () => + new Promise((resolve) => { + release = () => resolve(null) + }) + ) + const tick = createCheckpointSchedule({ + now: () => now, + busy: () => busy, + latest, + fingerprint: async () => 'first', + create, + failed: vi.fn() + }) + await tick() + expect(latest).not.toHaveBeenCalled() + busy = false + const checking = tick() + await tick() + expect(latest).toHaveBeenCalledTimes(1) + busy = true + release() + await checking + expect(create).not.toHaveBeenCalled() + expect(CHECKPOINT_INTERVAL_MS).toBe(900_000) +}) diff --git a/apps/electron/src/storage/checkpoint-policy.ts b/apps/electron/src/storage/checkpoint-policy.ts new file mode 100644 index 000000000..633e40982 --- /dev/null +++ b/apps/electron/src/storage/checkpoint-policy.ts @@ -0,0 +1,60 @@ +export const CHECKPOINT_INTERVAL_MS = 15 * 60 * 1000 +const DAY = 24 * 60 * 60 * 1000 + +/** Keep one point per interval/day/week, plus the latest two and explicitly pinned points. */ +export function retainedCheckpointIds( + points: readonly { id: string; createdAt: string; pinned?: boolean }[], + now: number +): Set { + const sorted = [...points].sort((a, b) => Date.parse(b.createdAt) - Date.parse(a.createdAt)) + const retained = new Set(sorted.slice(0, 2).map((point) => point.id)) + const buckets = new Set() + for (const point of sorted) { + const time = Date.parse(point.createdAt) + if (!Number.isFinite(time)) throw new Error('Invalid checkpoint date') + const age = now - time + if (point.pinned || age < 0) retained.add(point.id) + const bucket = + age <= DAY + ? `recent:${Math.floor(time / CHECKPOINT_INTERVAL_MS)}` + : age <= 7 * DAY + ? `daily:${Math.floor(time / DAY)}` + : age <= 28 * DAY + ? `weekly:${Math.floor(time / (7 * DAY))}` + : null + if (bucket && !buckets.has(bucket)) { + buckets.add(bucket) + retained.add(point.id) + } + } + return retained +} + +/** Level-triggered: every tick rechecks durable evidence, so missed timers need no replay. */ +export function createCheckpointSchedule(options: { + now: () => number + busy: () => boolean + latest: () => Promise<{ createdAt: string; sourceFingerprint?: string } | null> + fingerprint: () => Promise + create: () => Promise + failed: (error: unknown) => void +}) { + let checking = false + return async (): Promise => { + if (checking || options.busy()) return + checking = true + try { + const latest = await options.latest() + const age = latest ? options.now() - Date.parse(latest.createdAt) : Infinity + // A clock correction should not suppress protection indefinitely. + if (age >= 0 && age < CHECKPOINT_INTERVAL_MS) return + const fingerprint = await options.fingerprint() + if (latest?.sourceFingerprint === fingerprint) return + if (!options.busy()) await options.create() + } catch (error) { + options.failed(error) + } finally { + checking = false + } + } +} diff --git a/apps/electron/src/storage/checkpoints.test.ts b/apps/electron/src/storage/checkpoints.test.ts new file mode 100644 index 000000000..ee22651d6 --- /dev/null +++ b/apps/electron/src/storage/checkpoints.test.ts @@ -0,0 +1,236 @@ +import { mkdtemp, mkdir, readFile, rm, writeFile, readdir, symlink } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import Database from 'better-sqlite3' +import { beforeEach, afterEach, describe, expect, it } from 'vitest' +import { + createCheckpoint, + listCheckpoints, + inspectCheckpoints, + retainCheckpoints, + verifyCheckpoint +} from './checkpoints' + +let root: string +let dataPath: string +let recoveryPath: string +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-checkpoint-')) + dataPath = join(root, 'workspace') + recoveryPath = join(root, 'recovery') + await mkdir(dataPath) + for (const file of ['data.db', 'xnet.db']) { + const db = new Database(join(dataPath, file)) + db.exec("CREATE TABLE content (value TEXT); INSERT INTO content VALUES ('keep me')") + db.close() + } + await writeFile( + join(dataPath, 'identity-seed.json'), + '{"testFixture":"opaque encrypted identity"}' + ) +}) +afterEach(() => rm(root, { recursive: true, force: true })) + +const create = () => + createCheckpoint({ dataPath, recoveryPath, appVersion: '3.0.0', profile: 'test' }) + +describe('complete native recovery copies', () => { + it('refuses an optional database whose recovery sidecars remain without its base file', async () => { + await writeFile(join(dataPath, 'library.db-wal'), 'not a complete database') + await expect(create()).rejects.toThrow('missing database with remaining sidecars') + }) + it('verifies both stores, key files, and nested imported evidence', async () => { + await mkdir(join(dataPath, 'sources')) + await writeFile(join(dataPath, 'sources', 'archive.zip'), 'retained archive bytes') + const manifest = await create() + const path = join(recoveryPath, manifest.id) + expect((await verifyCheckpoint(path)).files).toHaveLength(4) + expect((await listCheckpoints(recoveryPath)).map((point) => point.id)).toEqual([manifest.id]) + await rm(dataPath, { recursive: true }) + expect((await verifyCheckpoint(path)).id).toBe(manifest.id) + expect(await readFile(join(path, 'workspace/sources/archive.zip'), 'utf8')).toBe( + 'retained archive bytes' + ) + }) + + it('prunes only after verification and preserves originals outside retention', async () => { + const first = await create() + await create() + const newest = await create() + await mkdir(join(recoveryPath, 'preserved', 'original'), { recursive: true }) + await writeFile(join(recoveryPath, 'preserved', 'original', 'data'), 'newer work') + await retainCheckpoints(recoveryPath, newest.id, { keep: 2 }) + expect(await listCheckpoints(recoveryPath)).toHaveLength(2) + expect(await readdir(recoveryPath)).not.toContain(first.id) + expect(await readFile(join(recoveryPath, 'preserved', 'original', 'data'), 'utf8')).toBe( + 'newer work' + ) + const broken = await create() + await writeFile(join(recoveryPath, broken.id, 'workspace/data.db'), 'broken') + await expect(retainCheckpoints(recoveryPath, broken.id, { keep: 2 })).rejects.toThrow() + expect(await listCheckpoints(recoveryPath)).toHaveLength(3) + await verifyCheckpoint(join(recoveryPath, newest.id)) + }) + + it('rejects a backup destination inside the source', async () => { + await expect( + createCheckpoint({ + dataPath, + recoveryPath: join(dataPath, 'copies'), + appVersion: '1', + profile: 'test' + }) + ).rejects.toThrow('outside') + }) + + it('includes committed WAL content in a standalone database', async () => { + const writer = new Database(join(dataPath, 'data.db')) + writer.pragma('journal_mode = WAL') + writer.pragma('wal_autocheckpoint = 0') + writer.exec("INSERT INTO content VALUES ('from WAL')") + try { + const manifest = await create() + const path = join(recoveryPath, manifest.id) + const db = new Database(join(path, 'workspace/data.db'), { readonly: true }) + try { + expect(db.prepare('SELECT value FROM content ORDER BY rowid').all()).toEqual([ + { value: 'keep me' }, + { value: 'from WAL' } + ]) + } finally { + db.close() + } + await verifyCheckpoint(path) + expect(manifest.files.some((file) => file.path.endsWith('-wal'))).toBe(false) + } finally { + writer.close() + } + }) + + it('keeps WAL copies independent of live writes and later recovery points', async () => { + const source = join(dataPath, 'data.db') + const writer = new Database(source) + writer.pragma('journal_mode = WAL') + writer.pragma('wal_autocheckpoint = 0') + writer.exec("INSERT INTO content VALUES ('first saved value')") + try { + const before = await Promise.all([readFile(source), readFile(`${source}-wal`)]) + const first = await create() + expect(await Promise.all([readFile(source), readFile(`${source}-wal`)])).toEqual(before) + expect(writer.pragma('journal_mode', { simple: true })).toBe('wal') + + writer.exec("UPDATE content SET value = 'later live value' WHERE rowid = 2") + const second = await create() + const firstPath = join(recoveryPath, first.id) + const secondPath = join(recoveryPath, second.id) + const firstCopy = new Database(join(firstPath, 'workspace/data.db'), { readonly: true }) + const secondCopy = new Database(join(secondPath, 'workspace/data.db')) + try { + expect(firstCopy.pragma('journal_mode', { simple: true })).toBe('delete') + expect(firstCopy.prepare('SELECT value FROM content WHERE rowid = 2').get()).toEqual({ + value: 'first saved value' + }) + expect(secondCopy.prepare('SELECT value FROM content WHERE rowid = 2').get()).toEqual({ + value: 'later live value' + }) + secondCopy.exec("UPDATE content SET value = 'edited restored copy' WHERE rowid = 2") + expect(writer.prepare('SELECT value FROM content WHERE rowid = 2').get()).toEqual({ + value: 'later live value' + }) + } finally { + firstCopy.close() + secondCopy.close() + } + await verifyCheckpoint(firstPath) + } finally { + writer.close() + } + }) + + it('rejects an incomplete source and keeps the previous good copy', async () => { + const good = await create() + await rm(join(dataPath, 'identity-seed.json')) + await expect(create()).rejects.toThrow('missing required content') + expect((await verifyCheckpoint(join(recoveryPath, good.id))).id).toBe(good.id) + expect(await readdir(recoveryPath)).toEqual([good.id]) + }) + + it('rejects missing blobs, altered content, and forged incomplete manifests', async () => { + const manifest = await create() + const path = join(recoveryPath, manifest.id) + await writeFile(join(path, 'workspace/identity-seed.json'), 'changed') + await expect(verifyCheckpoint(path)).rejects.toThrow('verification failed') + manifest.files = manifest.files.filter((file) => file.path !== 'xnet.db') + await writeFile(join(path, 'manifest.json'), JSON.stringify(manifest)) + await expect(verifyCheckpoint(path)).rejects.toThrow('Incomplete recovery copy') + }) + + it('never follows a source symlink into unrelated user files', async () => { + await symlink(root, join(dataPath, 'outside')) + await expect(create()).rejects.toThrow('symbolic link') + }) + + it('does not label a failed or corrupt source as a recovery point', async () => { + await writeFile(join(dataPath, 'data.db'), 'broken database') + await expect(create()).rejects.toThrow() + expect(await listCheckpoints(recoveryPath)).toEqual([]) + expect(await readdir(recoveryPath)).toEqual([]) + }) + + it('does not allow a test identity to stand in for daily ownership', async () => { + await rm(join(dataPath, 'identity-seed.json')) + const point = await createCheckpoint({ + dataPath, + recoveryPath, + appVersion: 'test', + profile: 'test', + testIdentity: true + }) + await expect(verifyCheckpoint(join(recoveryPath, point.id))).rejects.toThrow('test identity') + }) +}) + +it('preserves damaged older manifests while creating and retaining new verified copies', async () => { + const broken = await create() + await writeFile(join(recoveryPath, broken.id, 'manifest.json'), '{truncated') + await create() + await create() + const newest = await create() + await retainCheckpoints(recoveryPath, newest.id, { keep: 2 }) + const listing = await inspectCheckpoints(recoveryPath) + expect(listing.checkpoints).toHaveLength(2) + expect(listing.checkpoints[0].id).toBe(newest.id) + expect(listing.unreadable).toEqual([{ id: broken.id, reason: expect.any(String) }]) + expect(await readFile(join(recoveryPath, broken.id, 'manifest.json'), 'utf8')).toBe('{truncated') + await expect(listCheckpoints(recoveryPath)).rejects.toThrow('Unreadable recovery points') + await verifyCheckpoint(join(recoveryPath, newest.id)) +}) + +it('reports recovery symlinks and unsafe manifests without following or pruning them', async () => { + const point = await create() + const alias = '1-11111111-1111-1111-1111-111111111111' + await symlink(join(recoveryPath, point.id), join(recoveryPath, alias)) + const corrupt = await create() + const manifestPath = join(recoveryPath, corrupt.id, 'manifest.json') + const manifest = JSON.parse(await readFile(manifestPath, 'utf8')) + manifest.files.push({ path: '../outside', size: 0, sha256: '0'.repeat(64) }) + await writeFile(manifestPath, JSON.stringify(manifest)) + const listing = await inspectCheckpoints(recoveryPath) + expect(listing.checkpoints.map((value) => value.id)).toEqual([point.id]) + expect(listing.unreadable).toHaveLength(2) + expect(listing.unreadable.map((value) => value.reason)).toEqual( + expect.arrayContaining([ + 'Recovery point must be a real directory', + 'Invalid recovery file path' + ]) + ) + const newest = await create() + await retainCheckpoints(recoveryPath, newest.id, { keep: 2 }) + expect(await readdir(recoveryPath)).toEqual(expect.arrayContaining([alias, corrupt.id])) +}) + +it('distinguishes absent recovery storage from an unreadable recovery root', async () => { + expect(await inspectCheckpoints(recoveryPath)).toEqual({ checkpoints: [], unreadable: [] }) + await writeFile(recoveryPath, 'not a directory') + await expect(inspectCheckpoints(recoveryPath)).rejects.toThrow() +}) diff --git a/apps/electron/src/storage/checkpoints.ts b/apps/electron/src/storage/checkpoints.ts new file mode 100644 index 000000000..2216bef82 --- /dev/null +++ b/apps/electron/src/storage/checkpoints.ts @@ -0,0 +1,358 @@ +import type { CheckpointManifest, CheckpointListing } from '../shared/recovery' +import { execFile } from 'node:child_process' +import { createHash, randomUUID } from 'node:crypto' +import { createReadStream, constants } from 'node:fs' +import { copyFile, lstat, mkdir, readdir, readFile, rename, rm, open } from 'node:fs/promises' +import { dirname, join, relative, resolve, sep } from 'node:path' +import { promisify } from 'node:util' +import Database from 'better-sqlite3' +import { retainedCheckpointIds } from './checkpoint-policy' + +const FORMAT = 'xnet-desktop-checkpoint/1' +const DATABASES = ['data.db', 'xnet.db'] +const OPTIONAL_DATABASES = ['library.db'] +const REQUIRED = [...DATABASES, 'identity-seed.json'] +const runFile = promisify(execFile) + +export type { CheckpointManifest } from '../shared/recovery' + +async function digest(path: string): Promise { + const hash = createHash('sha256') + for await (const chunk of createReadStream(path)) hash.update(chunk) + return hash.digest('hex') +} + +async function inventory(root: string, at = root): Promise<{ path: string; stamp: string }[]> { + if ((await lstat(at)).isSymbolicLink()) + throw new Error('Recovery directory cannot be a symbolic link') + const entries = await readdir(at, { withFileTypes: true }) + const files: { path: string; stamp: string }[] = [] + for (const entry of entries.sort((a, b) => a.name.localeCompare(b.name))) { + const path = join(at, entry.name) + if (entry.isSymbolicLink()) + throw new Error(`Recovery copy refuses a symbolic link: ${relative(root, path)}`) + if (entry.isDirectory()) files.push(...(await inventory(root, path))) + else if (entry.isFile()) { + const stat = await lstat(path, { bigint: true }) + files.push({ + path: relative(root, path).split(sep).join('/'), + stamp: `${stat.ino}:${stat.size}:${stat.mtimeNs}:${stat.ctimeNs}` + }) + } else throw new Error(`Unsupported workspace file: ${relative(root, path)}`) + } + return files +} + +function inventoryFingerprint(files: { path: string; stamp: string }[]): string { + // Shared-memory reader locks can change without any saved workspace data changing. + return createHash('sha256') + .update(JSON.stringify(files.filter((file) => !file.path.endsWith('-shm')))) + .digest('hex') +} + +export async function workspaceFingerprint(dataPath: string): Promise { + return inventoryFingerprint(await inventory(dataPath)) +} + +function safePath(root: string, path: string): string { + if ( + !path || + path.includes('\\') || + path.split('/').some((part) => part === '..' || part === '.' || part === '') + ) + throw new Error('Invalid recovery file path') + const result = resolve(root, path) + if (!result.startsWith(resolve(root) + sep)) + throw new Error('Recovery file escapes its directory') + return result +} + +async function syncFile(path: string): Promise { + const handle = await open(path, 'r') + try { + await handle.sync() + } finally { + await handle.close() + } +} + +async function copyCheckpointFile(source: string, destination: string): Promise { + if (process.platform === 'darwin') { + // Electron's Node 20/libuv cannot clone APFS files through copyFile. Native + // cp uses independent copy-on-write files, with a full-copy fallback off APFS. + await runFile('/bin/cp', ['-c', source, destination]) + } else { + await copyFile(source, destination, constants.COPYFILE_FICLONE) + } +} + +async function normalizeDatabase(path: string): Promise { + const db = new Database(path, { fileMustExist: true }) + try { + if (db.pragma('quick_check', { simple: true }) !== 'ok') + throw new Error('Recovery database failed integrity verification') + // This is the isolated copy, never the live database. Leaving WAL mode folds + // committed pages into it without rewriting every unchanged cloned page. + if (db.pragma('journal_mode = DELETE', { simple: true }) !== 'delete') + throw new Error('Recovery database could not become a standalone copy') + } finally { + db.close() + } + for (const suffix of ['-wal', '-shm']) await rm(path + suffix, { force: true }) +} + +function databaseVersion(path: string): number | undefined { + const db = new Database(path, { readonly: true, fileMustExist: true }) + try { + if (!db.prepare("SELECT 1 FROM sqlite_master WHERE name = '_schema_version'").get()) + return undefined + const row = db.prepare('SELECT MAX(version) AS version FROM _schema_version').get() as { + version: unknown + } + if (!Number.isSafeInteger(row.version) || Number(row.version) < 1) + throw new Error('Recovery source has an invalid storage version') + return Number(row.version) + } finally { + db.close() + } +} + +/** The caller flushes editors first. A changing source fails instead of producing a mixed copy. */ +export async function createCheckpoint(options: { + dataPath: string + recoveryPath: string + appVersion: string + profile: string + testIdentity?: boolean + pinned?: boolean +}): Promise { + if ( + resolve(options.recoveryPath).startsWith(resolve(options.dataPath) + sep) || + resolve(options.recoveryPath) === resolve(options.dataPath) + ) + throw new Error('Recovery copies must live outside the workspace directory') + const before = await inventory(options.dataPath) + for (const name of OPTIONAL_DATABASES) { + if ( + !before.some((entry) => entry.path === name) && + before.some((entry) => + ['-wal', '-shm', '-journal'].some((suffix) => entry.path === name + suffix) + ) + ) + throw new Error(`Recovery copy is missing database with remaining sidecars: ${name}`) + } + const required = options.testIdentity ? DATABASES : REQUIRED + for (const path of required) { + if (!before.some((entry) => entry.path === path)) + throw new Error(`Recovery copy is missing required content: ${path}`) + } + const id = `${Date.now()}-${randomUUID()}` + const temporary = join(options.recoveryPath, `.incomplete-${id}`) + const payload = join(temporary, 'workspace') + await mkdir(payload, { recursive: true, mode: 0o700 }) + try { + for (const entry of before) { + const destination = safePath(payload, entry.path) + await mkdir(dirname(destination), { recursive: true, mode: 0o700 }) + await copyCheckpointFile(safePath(options.dataPath, entry.path), destination) + } + if (JSON.stringify(before) !== JSON.stringify(await inventory(options.dataPath))) { + throw new Error( + 'Workspace changed while making a recovery copy. Retry after current work finishes.' + ) + } + for (const name of [ + ...DATABASES, + ...OPTIONAL_DATABASES.filter((name) => before.some((entry) => entry.path === name)) + ]) + await normalizeDatabase(join(payload, name)) + const files: CheckpointManifest['files'] = [] + for (const entry of await inventory(payload)) { + const path = safePath(payload, entry.path) + files.push({ path: entry.path, size: (await lstat(path)).size, sha256: await digest(path) }) + await syncFile(path) + await syncFile(dirname(path)) + } + const manifest: CheckpointManifest = { + format: FORMAT, + id, + createdAt: new Date().toISOString(), + appVersion: options.appVersion, + profile: options.profile, + identity: options.testIdentity ? 'test' : 'stored', + sourceFingerprint: inventoryFingerprint(before), + ...(options.pinned ? { pinned: true } : {}), + storageVersion: databaseVersion(join(payload, 'data.db')), + files + } + const file = await open(join(temporary, 'manifest.json'), 'wx', 0o600) + try { + await file.writeFile(JSON.stringify(manifest, null, 2)) + await file.sync() + } finally { + await file.close() + } + await verifyCheckpoint(temporary, { allowTestIdentity: options.testIdentity }) + await syncFile(payload) + await syncFile(temporary) + await rename(temporary, join(options.recoveryPath, id)) + await syncFile(options.recoveryPath) + return manifest + } catch (error) { + await rm(temporary, { recursive: true, force: true }) + throw error + } +} + +async function readManifest( + path: string, + options: { allowTestIdentity?: boolean } = {} +): Promise { + const directory = await lstat(path) + if (!directory.isDirectory() || directory.isSymbolicLink()) + throw new Error('Recovery point must be a real directory') + const metadata = await lstat(join(path, 'manifest.json')) + if (!metadata.isFile() || metadata.isSymbolicLink()) + throw new Error('Recovery manifest must be a regular file') + const parsed: unknown = JSON.parse(await readFile(join(path, 'manifest.json'), 'utf8')) + if (!parsed || typeof parsed !== 'object') throw new Error('Invalid recovery manifest') + const manifest = parsed as CheckpointManifest + if ( + manifest.format !== FORMAT || + typeof manifest.id !== 'string' || + !/^\d+-[a-f0-9-]{36}$/.test(manifest.id) || + !Array.isArray(manifest.files) || + typeof manifest.appVersion !== 'string' || + typeof manifest.profile !== 'string' || + typeof manifest.createdAt !== 'string' || + !Number.isFinite(Date.parse(manifest.createdAt)) || + !['stored', 'test'].includes(manifest.identity) + ) + throw new Error('Unsupported recovery manifest') + if ( + (manifest.sourceFingerprint !== undefined && + (typeof manifest.sourceFingerprint !== 'string' || + !/^[a-f0-9]{64}$/.test(manifest.sourceFingerprint))) || + (manifest.pinned !== undefined && typeof manifest.pinned !== 'boolean') || + (manifest.storageVersion !== undefined && + (!Number.isSafeInteger(manifest.storageVersion) || manifest.storageVersion < 1)) + ) + throw new Error('Invalid recovery policy metadata') + if (manifest.identity === 'test' && !options.allowTestIdentity) + throw new Error('A test identity cannot restore a daily workspace') + const files = manifest.files + if ( + files.some( + (file) => + !file || + typeof file.path !== 'string' || + !Number.isSafeInteger(file.size) || + file.size < 0 || + typeof file.sha256 !== 'string' || + !/^[a-f0-9]{64}$/.test(file.sha256) + ) + ) + throw new Error('Invalid recovery file inventory') + if (new Set(files.map((file) => file.path)).size !== files.length) + throw new Error('Duplicate recovery files') + for (const file of files) safePath(join(path, 'workspace'), file.path) + const required = manifest.identity === 'test' ? DATABASES : REQUIRED + for (const name of required) + if (!files.some((file) => file.path === name)) + throw new Error(`Incomplete recovery copy: ${name}`) + return manifest +} + +export async function verifyCheckpoint( + path: string, + options: { allowTestIdentity?: boolean } = {} +): Promise { + const manifest = await readManifest(path, options) + const files = manifest.files + const payload = join(path, 'workspace') + const actual = (await inventory(payload)).map((file) => file.path).sort() + if (JSON.stringify(actual) !== JSON.stringify(files.map((file) => file.path).sort())) + throw new Error('Recovery file inventory does not match') + for (const file of files) { + const stored = safePath(payload, file.path) + if ((await lstat(stored)).size !== file.size || (await digest(stored)) !== file.sha256) + throw new Error(`Recovery file verification failed: ${file.path}`) + } + for (const name of [ + ...DATABASES, + ...OPTIONAL_DATABASES.filter((name) => manifest.files.some((entry) => entry.path === name)) + ]) { + const db = new Database(join(payload, name), { readonly: true, fileMustExist: true }) + try { + if (db.pragma('quick_check', { simple: true }) !== 'ok') + throw new Error(`Recovery database is unreadable: ${name}`) + } finally { + db.close() + } + } + return manifest +} + +/** Manifest inspection is cheap; restore still verifies every byte before changing anything. */ +export async function inspectCheckpoints(recoveryPath: string): Promise { + let names: string[] + try { + names = await readdir(recoveryPath) + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') + return { checkpoints: [], unreadable: [] } + throw error + } + const result: CheckpointListing = { checkpoints: [], unreadable: [] } + for (const name of names.filter((name) => /^\d+-[a-f0-9-]{36}$/.test(name))) { + try { + const value = await readManifest(join(recoveryPath, name), { allowTestIdentity: true }) + if (value.id !== name) throw new Error('Recovery directory and manifest IDs differ') + result.checkpoints.push(value) + } catch (error) { + result.unreadable.push({ + id: name, + reason: error instanceof Error ? error.message : String(error) + }) + } + } + result.checkpoints.sort((a, b) => Date.parse(b.createdAt) - Date.parse(a.createdAt)) + result.unreadable.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0)) + return result +} + +/** Strict caller convenience: incomplete listing can never masquerade as an empty workspace. */ +export async function listCheckpoints(recoveryPath: string): Promise { + const result = await inspectCheckpoints(recoveryPath) + if (result.unreadable.length) + throw new Error( + `Unreadable recovery points: ${result.unreadable.map((point) => point.id).join(', ')}` + ) + return result.checkpoints +} + +/** Prune only after a new, complete point has passed verification. Preserved originals are separate. */ +export async function retainCheckpoints( + recoveryPath: string, + verifiedId: string, + options: { keep?: number; allowTestIdentity?: boolean } = {} +): Promise { + const keep = options.keep + if (keep !== undefined && (!Number.isSafeInteger(keep) || keep < 2)) + throw new Error('Keep at least two recovery copies') + if (!/^\d+-[a-f0-9-]{36}$/.test(verifiedId)) throw new Error('Invalid recovery point') + await verifyCheckpoint(join(recoveryPath, verifiedId), options) + // Unknown/corrupt points remain on disk for inspection. They must neither block + // a fresh verified copy nor become candidates for automatic deletion. + const { checkpoints: points } = await inspectCheckpoints(recoveryPath) + const retained = + keep === undefined + ? retainedCheckpointIds(points, Date.now()) + : new Set(points.slice(0, keep).map((point) => point.id)) + retained.add(verifiedId) + for (const point of points) { + if (point.pinned) retained.add(point.id) + if (!retained.has(point.id)) await rm(join(recoveryPath, point.id), { recursive: true }) + } + await syncFile(recoveryPath) +} diff --git a/apps/electron/src/storage/compatibility.test.ts b/apps/electron/src/storage/compatibility.test.ts new file mode 100644 index 000000000..a3cc3d82b --- /dev/null +++ b/apps/electron/src/storage/compatibility.test.ts @@ -0,0 +1,108 @@ +import { mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { SCHEMA_VERSION } from '@xnetjs/sqlite' +import Database from 'better-sqlite3' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + inspectDatabase, + requireCompatibleDatabase, + WorkspaceRecoveryRequired +} from './compatibility' + +let root: string +let path: string +beforeEach(() => { + root = mkdtempSync(join(tmpdir(), 'xnet-compatibility-')) + path = join(root, 'data.db') +}) +afterEach(() => rmSync(root, { recursive: true, force: true })) + +function fixture(version?: number | string): void { + const db = new Database(path) + db.exec("CREATE TABLE notes (body TEXT); INSERT INTO notes VALUES ('Keep this note')") + if (version !== undefined) { + db.exec('CREATE TABLE _schema_version (version, applied_at INTEGER)') + db.prepare('INSERT INTO _schema_version VALUES (?, 1)').run(version) + } + db.close() +} + +describe('read-only workspace compatibility', () => { + it('distinguishes an absent file without creating a database', () => { + expect(inspectDatabase(path)).toEqual({ status: 'missing' }) + expect(readdirSync(root)).toEqual([]) + }) + + it('does not create a new database over orphaned recovery files', () => { + writeFileSync(`${path}-wal`, 'previous workspace') + expect(inspectDatabase(path).status).toBe('unreadable') + expect(readdirSync(root)).toEqual(['data.db-wal']) + }) + + it.each([ + [undefined, 'unversioned'], + [1, 'old'], + [SCHEMA_VERSION - 1, 'old'], + [SCHEMA_VERSION, 'supported'], + [SCHEMA_VERSION + 1, 'future'], + ['broken', 'unreadable'], + [0, 'unreadable'] + ] as const)('preserves a database with version %s (%s)', (version, status) => { + fixture(version) + const before = readFileSync(path) + expect(inspectDatabase(path).status).toBe(status) + if (status === 'supported') expect(() => requireCompatibleDatabase(path)).not.toThrow() + else expect(() => requireCompatibleDatabase(path)).toThrow(WorkspaceRecoveryRequired) + expect(readFileSync(path)).toEqual(before) + const db = new Database(path, { readonly: true }) + expect(db.prepare('SELECT body FROM notes').get()).toEqual({ body: 'Keep this note' }) + db.close() + }) + + it('preserves corrupt files and their sidecars', () => { + for (const suffix of ['', '-wal', '-shm']) writeFileSync(path + suffix, `preserve ${suffix}`) + expect(inspectDatabase(path).status).toBe('unreadable') + expect(() => requireCompatibleDatabase(path)).toThrow(WorkspaceRecoveryRequired) + for (const suffix of ['', '-wal', '-shm']) { + expect(readFileSync(path + suffix, 'utf8')).toBe(`preserve ${suffix}`) + } + }) + + it('reads committed versions in a live WAL without checkpointing it', () => { + const writer = new Database(path) + try { + writer.pragma('journal_mode = WAL') + writer.pragma('wal_autocheckpoint = 0') + writer.exec('CREATE TABLE _schema_version (version INTEGER)') + writer.prepare('INSERT INTO _schema_version VALUES (?)').run(SCHEMA_VERSION + 1) + const database = readFileSync(path) + const wal = readFileSync(`${path}-wal`) + expect(inspectDatabase(path).status).toBe('future') + expect(readFileSync(path)).toEqual(database) + expect(readFileSync(`${path}-wal`)).toEqual(wal) + } finally { + writer.close() + } + }) + + it('does not mistake invalid filesystem paths for new workspaces', () => { + expect(inspectDatabase(root).status).toBe('unreadable') + fixture() + expect(inspectDatabase(join(path, 'child')).status).toBe('unreadable') + }) + + it('validates the separate legacy blob store without writing to it', () => { + const db = new Database(path) + db.exec('CREATE TABLE blobs (cid TEXT PRIMARY KEY, data BLOB)') + db.close() + const before = readFileSync(path) + expect(inspectDatabase(path, 'blobs')).toEqual({ status: 'supported', version: 1 }) + expect(readFileSync(path)).toEqual(before) + }) + + it('rejects an unreadable blob schema before normal initialization', () => { + fixture() + expect(() => requireCompatibleDatabase(path, 'blobs')).toThrow(WorkspaceRecoveryRequired) + }) +}) diff --git a/apps/electron/src/storage/compatibility.ts b/apps/electron/src/storage/compatibility.ts new file mode 100644 index 000000000..96f953909 --- /dev/null +++ b/apps/electron/src/storage/compatibility.ts @@ -0,0 +1,141 @@ +import { constants, copyFileSync, mkdtempSync, rmSync, statSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { TaggedError } from '@xnetjs/core' +import { SCHEMA_VERSION } from '@xnetjs/sqlite' +import Database from 'better-sqlite3' + +export type StorageCompatibility = + | { status: 'missing' } + | { status: 'supported'; version: number } + | { status: 'old' | 'future'; version: number; expected: number } + | { status: 'unversioned' } + | { status: 'unreadable'; reason: string } + +function fingerprint(path: string): string | null { + try { + const stat = statSync(path, { bigint: true }) + return `${stat.ino}:${stat.size}:${stat.mtimeNs}:${stat.ctimeNs}` + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return null + throw error + } +} + +/** SQLite can write the SHM even in readonly mode. Inspect a disposable copy instead. */ +export function inspectDatabase( + path: string, + kind: 'workspace' | 'blobs' | 'library' = 'workspace' +): StorageCompatibility { + let db: Database.Database | undefined + let scratch: string | undefined + try { + try { + const file = statSync(path) + if (!file.isFile()) return { status: 'unreadable', reason: 'Database path is not a file.' } + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') { + if (['-wal', '-shm', '-journal'].some((suffix) => fingerprint(path + suffix) !== null)) { + return { + status: 'unreadable', + reason: 'Database is missing but recovery sidecars remain.' + } + } + return { status: 'missing' } + } + throw error + } + // The app holds the profile lock. Still refuse a source changed by another + // writer during the copy; it is not evidence of a compatible workspace. + if (fingerprint(`${path}-journal`) !== null) { + return { status: 'unreadable', reason: 'A rollback journal needs recovery on a copy.' } + } + const before = [fingerprint(path), fingerprint(`${path}-wal`)] + scratch = mkdtempSync(join(tmpdir(), 'xnet-inspect-')) + const candidate = join(scratch, 'data.db') + copyFileSync(path, candidate, constants.COPYFILE_FICLONE) + if (before[1] !== null) + copyFileSync(`${path}-wal`, `${candidate}-wal`, constants.COPYFILE_FICLONE) + const after = [fingerprint(path), fingerprint(`${path}-wal`)] + if (before.some((value, index) => value !== after[index])) { + return { + status: 'unreadable', + reason: 'Database changed during inspection. Close other writers and retry.' + } + } + db = new Database(candidate, { readonly: true, fileMustExist: true }) + const integrity = db.pragma('quick_check') as { quick_check: string }[] + if (integrity.length !== 1 || integrity[0]?.quick_check !== 'ok') { + return { status: 'unreadable', reason: 'SQLite integrity check failed.' } + } + if (kind === 'blobs') { + // This older store has no version table; validate its actual contract. + db.prepare('SELECT cid, data FROM blobs LIMIT 0').all() + return { status: 'supported', version: 1 } + } + if (kind === 'library') { + const version = db.pragma('user_version', { simple: true }) + if (version !== 1) + return { status: 'unreadable', reason: 'Unsupported library storage version.' } + db.prepare( + 'SELECT resource_id, capability, version, language, state, attempts, next_at, reason FROM work LIMIT 0' + ).all() + db.prepare('SELECT id, url, platform, title, payload, added_at FROM resources LIMIT 0').all() + db.prepare('SELECT key, value FROM settings LIMIT 0').all() + db.prepare('SELECT platform, until_ms FROM provider_pause LIMIT 0').all() + db.prepare( + 'SELECT resource_id, capability, version, at_ms, state, reason FROM attempts LIMIT 0' + ).all() + db.prepare('SELECT resource_id, title, body, start_ms FROM search LIMIT 0').all() + return { status: 'supported', version: 1 } + } + const table = db + .prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = '_schema_version'") + .get() + if (!table) return { status: 'unversioned' } + const rows = db.prepare('SELECT version FROM _schema_version').all() as { version: unknown }[] + if (rows.length === 0) return { status: 'unversioned' } + if (rows.some(({ version }) => !Number.isSafeInteger(version) || Number(version) < 1)) { + return { status: 'unreadable', reason: 'Invalid storage version record.' } + } + const version = Math.max(...rows.map((row) => Number(row.version))) + if (version === SCHEMA_VERSION) return { status: 'supported', version } + return { + status: version < SCHEMA_VERSION ? 'old' : 'future', + version, + expected: SCHEMA_VERSION + } + } catch (error) { + return { status: 'unreadable', reason: error instanceof Error ? error.message : String(error) } + } finally { + try { + db?.close() + } finally { + if (scratch) rmSync(scratch, { recursive: true, force: true }) + } + } +} + +export class WorkspaceRecoveryRequired extends TaggedError { + readonly _tag = 'WorkspaceRecoveryRequired' + + constructor( + readonly path: string, + readonly compatibility: StorageCompatibility + ) { + const detail = compatibility.status === 'unreadable' ? ` ${compatibility.reason}` : '' + super( + `This workspace needs recovery (${compatibility.status}).${detail} Original files were preserved.` + ) + } +} + +export function requireCompatibleDatabase( + path: string, + kind: 'workspace' | 'blobs' | 'library' = 'workspace' +): void { + const result = inspectDatabase(path, kind) + if (result.status !== 'missing' && result.status !== 'supported') { + throw new WorkspaceRecoveryRequired(path, result) + } +} diff --git a/apps/electron/src/storage/desktop-settings.test.ts b/apps/electron/src/storage/desktop-settings.test.ts new file mode 100644 index 000000000..091d0b22b --- /dev/null +++ b/apps/electron/src/storage/desktop-settings.test.ts @@ -0,0 +1,155 @@ +import type { SafeStorageLike } from '../main/secure-seed' +import { mkdtemp, readFile, readdir, rm, stat, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { + captureDesktopSettings, + restoreDesktopSettings, + validateDesktopSettings +} from '../shared/desktop-settings' +import { readDesktopSettings, SETTINGS_FILE, writeDesktopSettings } from './desktop-settings' + +const safe: SafeStorageLike = { + isEncryptionAvailable: () => true, + encryptString: (value) => Buffer.from(`fixture:${value}`), + decryptString: (bytes) => { + if (!bytes.toString().startsWith('fixture:')) throw new Error('Cannot decrypt') + return bytes.toString().slice(8) + } +} +const memoryStorage = () => { + const values = new Map() + return { + getItem: (key: string) => values.get(key) ?? null, + setItem: (key: string, value: string) => { + values.set(key, value) + }, + removeItem: (key: string) => { + values.delete(key) + } + } +} +const fixture = () => + captureDesktopSettings({ + getItem: (key) => (key === 'xnet:ai-api-key' ? 'test-only-secret' : null) + }) +let root: string +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-settings-')) +}) +afterEach(() => rm(root, { recursive: true, force: true })) + +it('distinguishes absent settings from damaged settings and preserves damaged bytes', async () => { + expect(await readDesktopSettings(root, safe)).toBeNull() + await writeFile(join(root, SETTINGS_FILE), '{broken') + await expect(readDesktopSettings(root, safe)).rejects.toThrow() + await expect(writeDesktopSettings(root, fixture(), safe)).rejects.toThrow() + expect(await readFile(join(root, SETTINGS_FILE), 'utf8')).toBe('{broken') +}) + +it('writes privately and leaves an unchanged snapshot untouched', async () => { + await writeDesktopSettings(root, fixture(), safe) + const before = await stat(join(root, SETTINGS_FILE)) + expect(before.mode & 0o777).toBe(0o600) + expect(await readDesktopSettings(root, safe)).toEqual(fixture()) + await writeDesktopSettings(root, fixture(), safe) + expect((await stat(join(root, SETTINGS_FILE))).mtimeMs).toBe(before.mtimeMs) + expect(await readdir(root)).toEqual([SETTINGS_FILE]) +}) + +it('does not include decrypted credentials in a parse failure', async () => { + await writeFile( + join(root, SETTINGS_FILE), + JSON.stringify({ + version: 1, + payload: safe.encryptString('{"secret":"do-not-print-this-key", broken').toString('base64') + }) + ) + const error = await readDesktopSettings(root, safe).catch((failure: unknown) => failure) + expect(error).toBeInstanceOf(Error) + expect((error as Error).message).toContain('invalid JSON') + expect((error as Error).message).not.toContain('do-not-print') +}) + +it('refuses a locked key store, failed encryption, and a symlink without replacing data', async () => { + await writeDesktopSettings(root, fixture(), safe) + const before = await readFile(join(root, SETTINGS_FILE)) + const locked = { ...safe, isEncryptionAvailable: () => false } + await expect(readDesktopSettings(root, locked)).rejects.toThrow('Unlock') + await expect(writeDesktopSettings(root, fixture(), locked)).rejects.toThrow('Unlock') + const changed = fixture() + changed.values['xnet-electron-theme'] = 'light' + await expect( + writeDesktopSettings(root, changed, { + ...safe, + encryptString: () => { + throw new Error('key store failed') + } + }) + ).rejects.toThrow('key store failed') + expect(await readFile(join(root, SETTINGS_FILE))).toEqual(before) + await rm(join(root, SETTINGS_FILE)) + await writeFile(join(root, 'original'), before) + await symlink(join(root, 'original'), join(root, SETTINGS_FILE)) + await expect(writeDesktopSettings(root, fixture(), safe)).rejects.toThrow('Invalid') + expect(await readFile(join(root, 'original'))).toEqual(before) +}) + +it('rejects incomplete and future contracts rather than silently losing keys', () => { + expect(() => validateDesktopSettings({ version: 2, values: {} })).toThrow('version') + expect(() => validateDesktopSettings({ version: 1, values: {} })).toThrow('Incomplete') + expect(() => + validateDesktopSettings({ ...fixture(), values: { ...fixture().values, unknown: 'value' } }) + ).toThrow('unsupported') +}) + +it('restores before first use, removes absent values, and excludes device authorizations', () => { + const storage = memoryStorage() + storage.setItem('xnet-electron-theme', 'stale') + storage.setItem('xnet:ai-bridge-token', 'old-device') + storage.setItem('xnet:ai-openrouter-verifier', 'old-request') + restoreDesktopSettings(storage, { settings: fixture(), restoreId: 'restore-one' }) + expect(storage.getItem('xnet:ai-api-key')).toBe('test-only-secret') + expect(storage.getItem('xnet-electron-theme')).toBeNull() + expect(storage.getItem('xnet:ai-bridge-token')).toBeNull() + expect(storage.getItem('xnet:ai-openrouter-verifier')).toBeNull() +}) + +it('preserves newer edits on normal restarts and applies each explicit restore once', () => { + const storage = memoryStorage() + const recovery = { settings: fixture(), restoreId: 'restore-one' } + restoreDesktopSettings(storage, recovery) + storage.setItem('xnet:ai-api-key', 'newer-setting') + restoreDesktopSettings(storage, recovery) + restoreDesktopSettings(storage, { ...recovery, restoreId: null }) + expect(storage.getItem('xnet:ai-api-key')).toBe('newer-setting') + restoreDesktopSettings(storage, { ...recovery, restoreId: 'restore-two' }) + expect(storage.getItem('xnet:ai-api-key')).toBe('test-only-secret') +}) + +it('does not mark an interrupted application complete and retries all settings', () => { + const storage = memoryStorage() + const recovery = { settings: fixture(), restoreId: 'restore-one' } + expect(() => + restoreDesktopSettings( + { + ...storage, + setItem: () => { + throw new Error('quota') + } + }, + recovery + ) + ).toThrow('quota') + restoreDesktopSettings(storage, recovery) + expect(storage.getItem('xnet:ai-api-key')).toBe('test-only-secret') +}) + +it('does not collect arbitrary browser keys or authorization tokens', () => { + const storage = memoryStorage() + storage.setItem('xnet:test:bypass', 'true') + storage.setItem('xnet:ai-openrouter-verifier', 'private') + storage.setItem('xnet:ai-bridge-token', 'private') + expect(JSON.stringify(captureDesktopSettings(storage))).not.toMatch(/private|bypass/) +}) diff --git a/apps/electron/src/storage/desktop-settings.ts b/apps/electron/src/storage/desktop-settings.ts new file mode 100644 index 000000000..8ac886c89 --- /dev/null +++ b/apps/electron/src/storage/desktop-settings.ts @@ -0,0 +1,85 @@ +import type { SafeStorageLike } from '../main/secure-seed' +import type { DesktopSettings } from '../shared/desktop-settings' +import { randomUUID } from 'node:crypto' +import { lstat, mkdir, open, readFile, rename, rm } from 'node:fs/promises' +import { join } from 'node:path' +import { SETTINGS_LIMIT, validateDesktopSettings } from '../shared/desktop-settings' + +export const SETTINGS_FILE = 'desktop-settings.json' + +function parseSettingsJson(text: string): unknown { + try { + return JSON.parse(text) + } catch { + // JSON parser messages may include the input, which can contain a provider key. + throw new Error('Desktop settings contain invalid JSON. The existing file has been preserved.') + } +} + +export async function readDesktopSettings( + dataPath: string, + safeStorage: SafeStorageLike +): Promise { + const path = join(dataPath, SETTINGS_FILE) + try { + const info = await lstat(path) + if (!info.isFile() || info.isSymbolicLink() || info.size > SETTINGS_LIMIT * 2) + throw new Error('Invalid desktop settings file') + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return null + throw error + } + if (!safeStorage.isEncryptionAvailable()) + throw new Error('Unlock the Mac key store to read desktop settings.') + const file = parseSettingsJson(await readFile(path, 'utf8')) + if ( + !file || + typeof file !== 'object' || + !('version' in file) || + file.version !== 1 || + !('payload' in file) || + typeof file.payload !== 'string' + ) + throw new Error('Unsupported desktop settings file') + const bytes = Buffer.from(file.payload, 'base64') + if (!bytes.length || bytes.toString('base64') !== file.payload) + throw new Error('Invalid desktop settings ciphertext') + return validateDesktopSettings(parseSettingsJson(safeStorage.decryptString(bytes))) +} + +/** Caller serializes writes. Refuse to overwrite an unreadable previous copy. */ +export async function writeDesktopSettings( + dataPath: string, + settings: DesktopSettings, + safeStorage: SafeStorageLike, + options: { rewrap?: boolean } = {} +): Promise { + const value = validateDesktopSettings(settings) + if (!safeStorage.isEncryptionAvailable()) + throw new Error('Unlock the Mac key store to save desktop settings.') + // Rewrapping is only for a disposable, already authenticated portable restore. + const previous = options.rewrap ? null : await readDesktopSettings(dataPath, safeStorage) + const serialized = JSON.stringify(value) + if (previous && JSON.stringify(previous) === serialized) return + const payload = safeStorage.encryptString(serialized).toString('base64') + await mkdir(dataPath, { recursive: true, mode: 0o700 }) + const temporary = join(dataPath, `.${SETTINGS_FILE}.${randomUUID()}.tmp`) + try { + const file = await open(temporary, 'wx', 0o600) + try { + await file.writeFile(JSON.stringify({ version: 1, payload })) + await file.sync() + } finally { + await file.close() + } + await rename(temporary, join(dataPath, SETTINGS_FILE)) + const directory = await open(dataPath, 'r') + try { + await directory.sync() + } finally { + await directory.close() + } + } finally { + await rm(temporary, { force: true }) + } +} diff --git a/apps/electron/src/storage/fixtures/schema-v8.sql b/apps/electron/src/storage/fixtures/schema-v8.sql new file mode 100644 index 000000000..4e2c65d0d --- /dev/null +++ b/apps/electron/src/storage/fixtures/schema-v8.sql @@ -0,0 +1,279 @@ +-- Historical schema from 6650c1f39^; synthetic data is added by migration tests. + +-- ============================================ +-- Schema Version Tracking +-- ============================================ + +CREATE TABLE IF NOT EXISTS _schema_version ( + version INTEGER PRIMARY KEY, + applied_at INTEGER NOT NULL +); + +-- ============================================ +-- Core Tables +-- ============================================ + +-- All nodes (Pages, Databases, Rows, Comments, etc.) +CREATE TABLE IF NOT EXISTS nodes ( + id TEXT PRIMARY KEY, + schema_id TEXT NOT NULL, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL, + created_by TEXT NOT NULL, + deleted_at INTEGER +); + +-- Node properties (LWW per-property) +CREATE TABLE IF NOT EXISTS node_properties ( + node_id TEXT NOT NULL, + property_key TEXT NOT NULL, + value BLOB, + lamport_time INTEGER NOT NULL, + updated_by TEXT NOT NULL, + updated_at INTEGER NOT NULL, + -- Grinding-resistant LWW final tiebreak key (exploration 0305): blake3 of + -- (author ‖ property ‖ value), present only for protocol v4+ writes. NULL + -- for legacy rows, which fall back to the author-DID tiebreak. + tiebreak_key TEXT, + + PRIMARY KEY (node_id, property_key), + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Rebuildable scalar property index for query planning. +CREATE TABLE IF NOT EXISTS node_property_scalars ( + node_id TEXT NOT NULL, + schema_id TEXT NOT NULL, + property_key TEXT NOT NULL, + value_type TEXT NOT NULL, + value_text TEXT, + value_number REAL, + value_boolean INTEGER, + value_hash TEXT, + updated_at INTEGER NOT NULL, + lamport_time INTEGER NOT NULL, + + PRIMARY KEY (schema_id, property_key, node_id), + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Query planner telemetry for adaptive read indexes. +CREATE TABLE IF NOT EXISTS query_descriptor_stats ( + descriptor_hash TEXT PRIMARY KEY, + schema_id TEXT NOT NULL, + descriptor_json TEXT NOT NULL, + hits INTEGER NOT NULL, + total_duration_ms REAL NOT NULL, + avg_duration_ms REAL NOT NULL, + avg_candidates REAL NOT NULL, + last_seen_at INTEGER NOT NULL +); + +CREATE TABLE IF NOT EXISTS query_index_candidates ( + index_name TEXT PRIMARY KEY, + descriptor_hash TEXT NOT NULL, + schema_id TEXT NOT NULL, + property_key TEXT NOT NULL, + value_type TEXT NOT NULL, + ddl TEXT NOT NULL, + created_at INTEGER NOT NULL, + last_used_at INTEGER NOT NULL, + estimated_bytes INTEGER NOT NULL DEFAULT 0, + estimated_rows INTEGER NOT NULL DEFAULT 0, + + FOREIGN KEY (descriptor_hash) REFERENCES query_descriptor_stats(descriptor_hash) + ON DELETE CASCADE +); + +CREATE TABLE IF NOT EXISTS node_query_materializations ( + view_id TEXT PRIMARY KEY, + descriptor_hash TEXT NOT NULL, + schema_id TEXT NOT NULL, + descriptor_json TEXT NOT NULL, + generated_at INTEGER NOT NULL, + invalidated_at INTEGER, + row_count INTEGER NOT NULL, + -- Authorization fingerprint the view was materialized under (exploration + -- 0226). NULL when authz is off; a mismatch forces an 'authz-changed' + -- refresh so a cached id list can never serve rows the viewer can no + -- longer read. + auth_fingerprint TEXT +); + +CREATE TABLE IF NOT EXISTS node_query_materialized_ids ( + view_id TEXT NOT NULL, + ordinal INTEGER NOT NULL, + node_id TEXT NOT NULL, + + PRIMARY KEY (view_id, ordinal), + UNIQUE (view_id, node_id), + FOREIGN KEY (view_id) REFERENCES node_query_materializations(view_id) + ON DELETE CASCADE, + FOREIGN KEY (node_id) REFERENCES nodes(id) + ON DELETE CASCADE +); + +-- Change log (event sourcing) +CREATE TABLE IF NOT EXISTS changes ( + hash TEXT PRIMARY KEY, + node_id TEXT NOT NULL, + payload BLOB NOT NULL, + lamport_time INTEGER NOT NULL, + lamport_peer TEXT NOT NULL, + wall_time INTEGER NOT NULL, + author TEXT NOT NULL, + parent_hash TEXT, + batch_id TEXT, + signature BLOB NOT NULL, + + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Y.Doc binary state (for nodes with collaborative content) +CREATE TABLE IF NOT EXISTS yjs_state ( + node_id TEXT PRIMARY KEY, + state BLOB NOT NULL, + updated_at INTEGER NOT NULL, + + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Y.Doc incremental updates (for sync) +CREATE TABLE IF NOT EXISTS yjs_updates ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + node_id TEXT NOT NULL, + update_data BLOB NOT NULL, + timestamp INTEGER NOT NULL, + origin TEXT, + + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Yjs snapshots (for document time travel) +CREATE TABLE IF NOT EXISTS yjs_snapshots ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + node_id TEXT NOT NULL, + timestamp INTEGER NOT NULL, + snapshot BLOB NOT NULL, + doc_state BLOB NOT NULL, + byte_size INTEGER NOT NULL, + + FOREIGN KEY (node_id) REFERENCES nodes(id) ON DELETE CASCADE +); + +-- Blobs (content-addressed) +CREATE TABLE IF NOT EXISTS blobs ( + cid TEXT PRIMARY KEY, + data BLOB NOT NULL, + mime_type TEXT, + size INTEGER NOT NULL, + created_at INTEGER NOT NULL, + reference_count INTEGER DEFAULT 1 +); + +-- Documents (for @xnetjs/storage compatibility) +CREATE TABLE IF NOT EXISTS documents ( + id TEXT PRIMARY KEY, + content BLOB NOT NULL, + metadata TEXT NOT NULL, + version INTEGER NOT NULL DEFAULT 1 +); + +-- Signed updates (for @xnetjs/storage compatibility) +CREATE TABLE IF NOT EXISTS updates ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + doc_id TEXT NOT NULL, + update_hash TEXT NOT NULL, + update_data TEXT NOT NULL, + created_at INTEGER DEFAULT (strftime('%s', 'now') * 1000), + UNIQUE(doc_id, update_hash) +); + +-- Snapshots (for @xnetjs/storage compatibility) +CREATE TABLE IF NOT EXISTS snapshots ( + doc_id TEXT PRIMARY KEY, + snapshot_data TEXT NOT NULL, + created_at INTEGER DEFAULT (strftime('%s', 'now') * 1000) +); + +-- Sync metadata +CREATE TABLE IF NOT EXISTS sync_state ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL +); + +-- ============================================ +-- Indexes +-- ============================================ + +CREATE INDEX IF NOT EXISTS idx_nodes_schema ON nodes(schema_id); +CREATE INDEX IF NOT EXISTS idx_nodes_updated ON nodes(updated_at); +CREATE INDEX IF NOT EXISTS idx_nodes_created_by ON nodes(created_by); +CREATE INDEX IF NOT EXISTS idx_nodes_deleted ON nodes(deleted_at); +CREATE INDEX IF NOT EXISTS idx_nodes_live_schema_updated + ON nodes(schema_id, updated_at DESC, id) + WHERE deleted_at IS NULL; +CREATE INDEX IF NOT EXISTS idx_nodes_all_schema_updated + ON nodes(schema_id, updated_at DESC, id); +CREATE INDEX IF NOT EXISTS idx_nodes_live_schema_created + ON nodes(schema_id, created_at DESC, id) + WHERE deleted_at IS NULL; + +CREATE INDEX IF NOT EXISTS idx_properties_node ON node_properties(node_id); +CREATE INDEX IF NOT EXISTS idx_properties_lamport ON node_properties(lamport_time); + +CREATE INDEX IF NOT EXISTS idx_prop_scalars_text + ON node_property_scalars(schema_id, property_key, value_text, node_id) + WHERE value_type = 'text'; +CREATE INDEX IF NOT EXISTS idx_prop_scalars_number + ON node_property_scalars(schema_id, property_key, value_number, node_id) + WHERE value_type = 'number'; +CREATE INDEX IF NOT EXISTS idx_prop_scalars_boolean + ON node_property_scalars(schema_id, property_key, value_boolean, node_id) + WHERE value_type = 'boolean'; +CREATE INDEX IF NOT EXISTS idx_prop_scalars_null + ON node_property_scalars(schema_id, property_key, node_id) + WHERE value_type = 'null'; +CREATE INDEX IF NOT EXISTS idx_prop_scalars_node + ON node_property_scalars(node_id); + +CREATE INDEX IF NOT EXISTS idx_query_stats_schema_seen + ON query_descriptor_stats(schema_id, last_seen_at DESC); +CREATE INDEX IF NOT EXISTS idx_query_indexes_schema_property + ON query_index_candidates(schema_id, property_key, value_type); +CREATE INDEX IF NOT EXISTS idx_query_materializations_schema + ON node_query_materializations(schema_id, invalidated_at); +CREATE INDEX IF NOT EXISTS idx_query_materialized_ids_node + ON node_query_materialized_ids(node_id); + +CREATE INDEX IF NOT EXISTS idx_changes_node ON changes(node_id); +CREATE INDEX IF NOT EXISTS idx_changes_lamport ON changes(lamport_time); +CREATE INDEX IF NOT EXISTS idx_changes_wall_time ON changes(wall_time); +CREATE INDEX IF NOT EXISTS idx_changes_batch ON changes(batch_id); +CREATE INDEX IF NOT EXISTS idx_changes_node_lamport + ON changes(node_id, lamport_time DESC, hash); + +CREATE INDEX IF NOT EXISTS idx_yjs_state_updated ON yjs_state(updated_at); +CREATE INDEX IF NOT EXISTS idx_yjs_updates_node ON yjs_updates(node_id); +CREATE INDEX IF NOT EXISTS idx_yjs_snapshots_node ON yjs_snapshots(node_id); +CREATE INDEX IF NOT EXISTS idx_yjs_snapshots_timestamp ON yjs_snapshots(node_id, timestamp); + +CREATE INDEX IF NOT EXISTS idx_updates_doc ON updates(doc_id); +CREATE INDEX IF NOT EXISTS idx_updates_created ON updates(created_at); + +-- ============================================ +-- Full-Text Search (FTS5) +-- ============================================ + +-- FTS index for searchable node content +CREATE VIRTUAL TABLE IF NOT EXISTS nodes_fts USING fts5( + node_id, + title, + content, + tokenize='porter unicode61' +); + +-- Triggers to keep FTS in sync will be managed by application layer +-- since the searchable content is derived from node properties + +INSERT INTO _schema_version (version, applied_at) VALUES (8, 1); diff --git a/apps/electron/src/storage/import-journal.test.ts b/apps/electron/src/storage/import-journal.test.ts new file mode 100644 index 000000000..ee825dd54 --- /dev/null +++ b/apps/electron/src/storage/import-journal.test.ts @@ -0,0 +1,140 @@ +import type { SocialImportCommitJobSnapshot } from '../shared/social-import' +import { mkdtemp, mkdir, readFile, readdir, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + streamSocialImportNodeDrafts, + type SocialImportNodeDraftStreamResult +} from '@xnetjs/social/import/core' +import { openSocialImportSource } from '@xnetjs/social/import/node' +import { builtInSocialImportAdapters } from '@xnetjs/social/importers' +import { beforeEach, afterEach, expect, it } from 'vitest' +import { + loadImportJournals, + retainedJournalSource, + saveImportJournal, + type ImportJournal +} from './import-journal' + +let root: string +let journal: ImportJournal +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-import-journal-')) + const path = join(root, 'stars.json') + await writeFile( + path, + JSON.stringify([ + { + starred_at: '2025-01-01T00:00:00Z', + repo: { + id: 123, + full_name: 'fixture/library', + html_url: 'https://github.com/fixture/library', + private: false, + owner: { id: 456, login: 'fixture' } + } + } + ]) + ) + const source = await openSocialImportSource(path) + const importedAt = '2026-09-29T12:00:00Z' + let result: SocialImportNodeDraftStreamResult | undefined + for await (const draft of streamSocialImportNodeDrafts({ + ...source, + adapters: builtInSocialImportAdapters, + importedAt, + onComplete: (value) => { + result = value + } + })) + void draft + if (!result) throw new Error('Fixture did not stage') + const totalRecords = result.canonicalRecordCount + 2 + const job: SocialImportCommitJobSnapshot = { + jobId: 'electron-social-import:fixture', + status: 'running', + phase: 'writing', + platform: 'github', + archiveName: 'stars.json', + totalRecords, + processedRecords: 2, + created: 2, + updated: 0, + skipped: 0, + warnings: 0, + currentBucketId: null, + currentChunk: 1, + totalChunks: 2, + startedAt: 1, + updatedAt: 2, + completedAt: null, + error: null, + metrics: null, + checkpoint: null, + bucketCheckpoints: [] + } + journal = { + version: 1, + job, + authorDID: 'did:key:fixture', + sourceHash: source.manifest.archiveHash!, + sourceExtension: '.json', + includeSourceRecords: false, + stage: { + ...result, + archive: { ...result.archive, archivePath: path }, + archivePath: path, + manifest: source.manifest, + stageRequest: { archivePath: path, includeSensitive: false }, + importedAt + } + } +}) +afterEach(() => rm(root, { recursive: true, force: true })) + +it('restores an interrupted cursor as paused with its exact selection and adapter version', () => { + saveImportJournal(root, journal) + const [restored] = loadImportJournals(root) + expect(restored.job.status).toBe('paused') + expect(restored.job.processedRecords).toBe(2) + expect(restored.stage).toEqual(journal.stage) + expect(restored.authorDID).toBe('did:key:fixture') +}) + +it('resolves retained evidence relative to the restored workspace, never the old path', () => { + expect(retainedJournalSource('/new/profile', journal)).toBe( + `/new/profile/import-sources/${journal.sourceHash}/source.json` + ) +}) + +it('preserves the last published cursor when an incomplete next write exists', async () => { + saveImportJournal(root, journal) + await writeFile(join(root, 'import-jobs', 'interrupted.tmp'), '{') + expect(loadImportJournals(root)[0].job.processedRecords).toBe(2) + saveImportJournal(root, { + ...journal, + job: { ...journal.job, status: 'completed', processedRecords: journal.job.totalRecords! } + }) + expect(loadImportJournals(root)[0].job.status).toBe('completed') +}) + +it('distinguishes absent journals from damaged progress and rejects invalid cursors', async () => { + expect(loadImportJournals(root)).toEqual([]) + expect(() => + saveImportJournal(root, { ...journal, job: { ...journal.job, processedRecords: 999999 } }) + ).toThrow('checkpoint') + await mkdir(join(root, 'import-jobs')) + await writeFile(join(root, 'import-jobs', 'broken.json'), '{') + expect(() => loadImportJournals(root)).toThrow() +}) + +it('persists no signing private key and rejects unsupported formats without replacing valid progress', async () => { + saveImportJournal(root, journal) + const [name] = await readdir(join(root, 'import-jobs')) + const before = await readFile(join(root, 'import-jobs', name), 'utf8') + expect(before).not.toContain('signingKey') + expect(() => + saveImportJournal(root, { ...journal, version: 2 } as unknown as ImportJournal) + ).toThrow('Invalid import journal') + expect(await readFile(join(root, 'import-jobs', name), 'utf8')).toBe(before) +}) diff --git a/apps/electron/src/storage/import-journal.ts b/apps/electron/src/storage/import-journal.ts new file mode 100644 index 000000000..6259a856b --- /dev/null +++ b/apps/electron/src/storage/import-journal.ts @@ -0,0 +1,143 @@ +import type { + ElectronStagedSocialImport, + SocialImportCommitJobSnapshot +} from '../shared/social-import' +import { createHash, randomUUID } from 'node:crypto' +import { + closeSync, + fsyncSync, + lstatSync, + mkdirSync, + openSync, + readFileSync, + readdirSync, + renameSync, + rmSync, + writeFileSync +} from 'node:fs' +import { join } from 'node:path' + +export type ImportJournal = { + version: 1 + job: SocialImportCommitJobSnapshot + stage: ElectronStagedSocialImport + authorDID: string + sourceHash: string + sourceExtension: '.zip' | '.json' + includeSourceRecords: boolean +} +const filename = (id: string) => `${createHash('sha256').update(id).digest('hex')}.json` +const record = (value: unknown): value is Record => + value !== null && typeof value === 'object' && !Array.isArray(value) + +function validate(value: unknown): asserts value is ImportJournal { + if ( + !record(value) || + value.version !== 1 || + !record(value.job) || + !record(value.stage) || + !record(value.stage.archive) || + !record(value.stage.archive.adapter) || + !record(value.stage.manifest) || + !record(value.stage.stageRequest) || + typeof value.authorDID !== 'string' || + typeof value.sourceHash !== 'string' || + !/^[a-f0-9]{64}$/.test(value.sourceHash) || + !['.zip', '.json'].includes(String(value.sourceExtension)) || + typeof value.includeSourceRecords !== 'boolean' + ) + throw new Error('Invalid import journal') + const job = value.job + if ( + typeof job.jobId !== 'string' || + !job.jobId.startsWith('electron-social-import:') || + !['queued', 'running', 'paused', 'completed', 'failed', 'cancelled'].includes( + String(job.status) + ) || + !Number.isSafeInteger(job.totalRecords) || + Number(job.totalRecords) < 2 || + !Number.isSafeInteger(job.processedRecords) || + Number(job.processedRecords) < 0 || + Number(job.processedRecords) > Number(job.totalRecords) || + !Number.isSafeInteger(job.currentChunk) || + Number(job.currentChunk) < 0 || + typeof value.stage.archive.adapter.id !== 'string' || + typeof value.stage.archive.adapter.version !== 'string' || + typeof value.stage.importedAt !== 'string' || + !Number.isFinite(Date.parse(value.stage.importedAt)) || + value.stage.manifest.archiveHash !== value.sourceHash || + !Array.isArray(value.stage.manifest.entries) + ) + throw new Error('Invalid import journal checkpoint') +} + +/** The durable cursor advances only after an acknowledged database batch. */ +export function saveImportJournal(dataPath: string, journal: ImportJournal): void { + validate(journal) + const directory = join(dataPath, 'import-jobs') + mkdirSync(directory, { recursive: true, mode: 0o700 }) + if (lstatSync(directory).isSymbolicLink()) + throw new Error('Import jobs cannot use a symbolic link') + const target = join(directory, filename(journal.job.jobId)) + const temporary = `${target}.${randomUUID()}.tmp` + try { + const file = openSync(temporary, 'wx', 0o600) + try { + writeFileSync(file, JSON.stringify(journal)) + fsyncSync(file) + } finally { + closeSync(file) + } + renameSync(temporary, target) + for (const path of [directory, dataPath]) { + const parent = openSync(path, 'r') + try { + fsyncSync(parent) + } finally { + closeSync(parent) + } + } + } finally { + rmSync(temporary, { force: true }) + } +} + +export function loadImportJournals(dataPath: string): ImportJournal[] { + const directory = join(dataPath, 'import-jobs') + let names: string[] + try { + if (lstatSync(directory).isSymbolicLink()) + throw new Error('Import jobs cannot use a symbolic link') + names = readdirSync(directory) + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return [] + throw error + } + return names + .filter((name) => name.endsWith('.json')) + .map((name) => { + const path = join(directory, name) + const info = lstatSync(path) + if (!info.isFile() || info.isSymbolicLink() || info.size > 64 * 1024 * 1024) + throw new Error('Unreadable import journal. Its source files were preserved.') + const value: unknown = JSON.parse(readFileSync(path, 'utf8')) + validate(value) + if (filename(value.job.jobId) !== name) + throw new Error('Import journal identity does not match its file') + return { + ...value, + job: ['queued', 'running'].includes(value.job.status) + ? { + ...value.job, + status: 'paused', + error: 'Interrupted import. Resume from the last saved batch.' + } + : value.job + } + }) +} + +export function retainedJournalSource(dataPath: string, journal: ImportJournal): string { + validate(journal) + return join(dataPath, 'import-sources', journal.sourceHash, `source${journal.sourceExtension}`) +} diff --git a/apps/electron/src/storage/import-sources.test.ts b/apps/electron/src/storage/import-sources.test.ts new file mode 100644 index 000000000..1a4eab0ce --- /dev/null +++ b/apps/electron/src/storage/import-sources.test.ts @@ -0,0 +1,59 @@ +import { createHash } from 'node:crypto' +import { mkdtemp, mkdir, readFile, readdir, rm, stat, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { retainImportSource } from './import-sources' + +let root: string +let dataPath: string +let sourcePath: string +const bytes = '{"private":"source evidence"}' +const expectedHash = createHash('sha256').update(bytes).digest('hex') +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-source-custody-')) + dataPath = join(root, 'workspace') + sourcePath = join(root, 'export.json') + await mkdir(dataPath) + await writeFile(sourcePath, bytes) +}) +afterEach(() => rm(root, { recursive: true, force: true })) +const retain = () => retainImportSource({ dataPath, sourcePath, expectedHash }) + +it('keeps private exact bytes usable after the original export is removed', async () => { + const path = await retain() + expect(await readFile(sourcePath, 'utf8')).toBe(bytes) + expect((await stat(path)).mode & 0o777).toBe(0o600) + await rm(sourcePath) + expect(await readFile(path, 'utf8')).toBe(bytes) + expect(await retain()).toBe(path) +}) + +it('rejects a changed archive before promoting it', async () => { + await writeFile(sourcePath, 'changed after review') + await expect(retain()).rejects.toThrow('differs from the reviewed archive') + expect(await readdir(join(dataPath, 'import-sources'))).toEqual([]) +}) + +it('detects damaged retained evidence without overwriting it', async () => { + const path = await retain() + await writeFile(path, 'damaged') + await expect(retain()).rejects.toThrow('differs') + expect(await readFile(path, 'utf8')).toBe('damaged') +}) + +it('handles concurrent imports of the same archive', async () => { + const paths = await Promise.all([retain(), retain()]) + expect(paths[0]).toBe(paths[1]) + expect(await readdir(join(dataPath, 'import-sources'))).toEqual([expectedHash]) +}) + +it('refuses symbolic links and invalid fingerprints', async () => { + await symlink(sourcePath, join(root, 'link.json')) + await expect( + retainImportSource({ dataPath, sourcePath: join(root, 'link.json'), expectedHash }) + ).rejects.toThrow('regular file') + await expect( + retainImportSource({ dataPath, sourcePath, expectedHash: '../escape' }) + ).rejects.toThrow('fingerprint') +}) diff --git a/apps/electron/src/storage/import-sources.ts b/apps/electron/src/storage/import-sources.ts new file mode 100644 index 000000000..967eb226f --- /dev/null +++ b/apps/electron/src/storage/import-sources.ts @@ -0,0 +1,86 @@ +import { createHash, randomUUID } from 'node:crypto' +import { constants, createReadStream } from 'node:fs' +import { chmod, copyFile, lstat, mkdir, open, rename, rm } from 'node:fs/promises' +import { extname, join } from 'node:path' + +async function hashFile(path: string): Promise { + const hash = createHash('sha256') + for await (const chunk of createReadStream(path)) hash.update(chunk) + return hash.digest('hex') +} + +async function sync(path: string): Promise { + const handle = await open(path, 'r') + try { + await handle.sync() + } finally { + await handle.close() + } +} + +async function requireDirectory(path: string): Promise { + await mkdir(path, { recursive: true, mode: 0o700 }) + const info = await lstat(path) + if (!info.isDirectory() || info.isSymbolicLink()) + throw new Error('Import source storage must be a local directory, not a symbolic link') +} + +async function verify(path: string, expectedHash: string): Promise { + const info = await lstat(path) + if (!info.isFile() || info.isSymbolicLink() || (await hashFile(path)) !== expectedHash) + throw new Error('Import source differs from the reviewed archive. Select and review it again.') +} + +/** Retain exact input bytes before the first database write. Never move the user's original. */ +export async function retainImportSource(options: { + dataPath: string + sourcePath: string + expectedHash: string +}): Promise { + const extension = extname(options.sourcePath).toLowerCase() + if (!/^[a-f0-9]{64}$/.test(options.expectedHash) || !['.zip', '.json'].includes(extension)) + throw new Error('Invalid import source fingerprint or file type') + const root = join(options.dataPath, 'import-sources') + await requireDirectory(root) + const destination = join(root, options.expectedHash) + const filename = `source${extension}` + const retained = join(destination, filename) + try { + const info = await lstat(destination) + if (!info.isDirectory() || info.isSymbolicLink()) + throw new Error('Invalid retained import source directory') + await verify(retained, options.expectedHash) + return retained + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error + } + + const sourceInfo = await lstat(options.sourcePath) + if (!sourceInfo.isFile() || sourceInfo.isSymbolicLink()) + throw new Error('Import source must be a regular file') + const temporary = join(root, `.incomplete-${randomUUID()}`) + await mkdir(temporary, { mode: 0o700 }) + try { + const candidate = join(temporary, filename) + await copyFile(options.sourcePath, candidate, constants.COPYFILE_FICLONE) + await chmod(candidate, 0o600) + await verify(candidate, options.expectedHash) + await sync(candidate) + await sync(temporary) + try { + await rename(temporary, destination) + } catch (error) { + if (!['EEXIST', 'ENOTEMPTY'].includes((error as NodeJS.ErrnoException).code ?? '')) + throw error + // Another import may have retained these same bytes while this copy was in progress. + const info = await lstat(destination) + if (!info.isDirectory() || info.isSymbolicLink()) throw error + await verify(retained, options.expectedHash) + } + await sync(root) + await sync(options.dataPath) + return retained + } finally { + await rm(temporary, { recursive: true, force: true }) + } +} diff --git a/apps/electron/src/storage/migrations.test.ts b/apps/electron/src/storage/migrations.test.ts new file mode 100644 index 000000000..9616f1955 --- /dev/null +++ b/apps/electron/src/storage/migrations.test.ts @@ -0,0 +1,129 @@ +import { mkdtemp, mkdir, readFile, readdir, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { SCHEMA_MIGRATIONS, SCHEMA_VERSION } from '@xnetjs/sqlite' +import Database from 'better-sqlite3' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { listCheckpoints, verifyCheckpoint } from './checkpoints' +import { inspectDatabase } from './compatibility' +import { prepareWorkspaceUpgrade } from './migrations' + +let root: string +let dataPath: string +let recoveryPath: string +const requireIdentity = vi.fn() +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-migration-')) + dataPath = join(root, 'data') + recoveryPath = join(root, 'recovery') + await mkdir(dataPath) + const db = new Database(join(dataPath, 'data.db')) + db.exec(await readFile(new URL('./fixtures/schema-v8.sql', import.meta.url), 'utf8')) + db.prepare( + 'INSERT INTO nodes (id, schema_id, created_at, updated_at, created_by) VALUES (?, ?, ?, ?, ?)' + ).run('note', 'xnet://xnet.fyi/Page@1.0.0', 1, 1, 'did:key:fixture') + db.prepare('INSERT INTO yjs_state (node_id, state, updated_at) VALUES (?, ?, ?)').run( + 'note', + Buffer.from('synthetic document bytes'), + 1 + ) + db.close() + const blobs = new Database(join(dataPath, 'xnet.db')) + blobs.exec('CREATE TABLE blobs (cid TEXT PRIMARY KEY, data BLOB NOT NULL)') + blobs.prepare('INSERT INTO blobs VALUES (?, ?)').run('fixture', Buffer.from('attachment')) + blobs.close() + await writeFile(join(dataPath, 'identity-seed.json'), 'opaque fixture identity') + await writeFile(join(dataPath, 'retained-source.json'), 'source evidence') + requireIdentity.mockReset() +}) +afterEach(() => rm(root, { recursive: true, force: true })) +const upgrade = () => + prepareWorkspaceUpgrade({ + dataPath, + recoveryPath, + profile: 'daily', + appVersion: 'fixture-new', + requireIdentity + }) + +it('upgrades a historical fixture on a copy and keeps identity, documents, blobs, and the original', async () => { + const before = await readFile(join(dataPath, 'data.db')) + expect(await upgrade()).toBe(true) + expect(inspectDatabase(join(dataPath, 'data.db'))).toEqual({ + status: 'supported', + version: SCHEMA_VERSION + }) + const points = await listCheckpoints(recoveryPath) + expect(points).toHaveLength(2) + const pinned = points.find((point) => point.pinned)! + expect(pinned.storageVersion).toBe(8) + expect(points.some((point) => point.storageVersion === SCHEMA_VERSION)).toBe(true) + await verifyCheckpoint(join(recoveryPath, pinned.id)) + expect(inspectDatabase(join(recoveryPath, pinned.id, 'workspace/data.db')).status).toBe('old') + const [generation] = await readdir(join(recoveryPath, 'preserved')) + expect(await readFile(join(recoveryPath, 'preserved', generation, 'data.db'))).toEqual(before) + expect(await readFile(join(dataPath, 'identity-seed.json'), 'utf8')).toBe( + 'opaque fixture identity' + ) + expect(await readFile(join(dataPath, 'retained-source.json'), 'utf8')).toBe('source evidence') + const db = new Database(join(dataPath, 'data.db'), { readonly: true }) + expect(db.prepare('SELECT state FROM yjs_state WHERE node_id = ?').get('note')).toEqual({ + state: Buffer.from('synthetic document bytes') + }) + db.close() + const blobs = new Database(join(dataPath, 'xnet.db'), { readonly: true }) + expect(blobs.prepare('SELECT data FROM blobs WHERE cid = ?').get('fixture')).toEqual({ + data: Buffer.from('attachment') + }) + blobs.close() + expect( + JSON.parse(await readFile(join(recoveryPath, 'review-required.json'), 'utf8')).version + ).toBe(1) + expect(await upgrade()).toBe(false) +}) + +it('keeps originals byte-for-byte when a migration fails', async () => { + const before = await readFile(join(dataPath, 'data.db')) + const sql = SCHEMA_MIGRATIONS[9] + SCHEMA_MIGRATIONS[9] = 'THIS IS NOT VALID SQL' + try { + await expect(upgrade()).rejects.toThrow() + } finally { + SCHEMA_MIGRATIONS[9] = sql + } + expect(await readFile(join(dataPath, 'data.db'))).toEqual(before) + expect((await listCheckpoints(recoveryPath))[0].pinned).toBe(true) + expect(await readdir(join(recoveryPath, 'migrations'))).toEqual([]) +}) + +it('refuses a candidate whose declared version hides a missing column', async () => { + const db = new Database(join(dataPath, 'data.db')) + db.exec('ALTER TABLE node_properties DROP COLUMN tiebreak_key') + db.close() + const before = await readFile(join(dataPath, 'data.db')) + await expect(upgrade()).rejects.toThrow('node_properties.tiebreak_key') + expect(await readFile(join(dataPath, 'data.db'))).toEqual(before) +}) + +it.each([7, SCHEMA_VERSION + 1])( + 'preserves unsupported version %s without an attempted upgrade', + async (version) => { + const db = new Database(join(dataPath, 'data.db')) + db.prepare('UPDATE _schema_version SET version = ?').run(version) + db.close() + const before = await readFile(join(dataPath, 'data.db')) + await expect(upgrade()).rejects.toThrow('needs recovery') + expect(await readFile(join(dataPath, 'data.db'))).toEqual(before) + expect(requireIdentity).not.toHaveBeenCalled() + } +) + +it('stops before copies or database changes when identity validation fails', async () => { + const before = await readFile(join(dataPath, 'data.db')) + requireIdentity.mockImplementation(() => { + throw new Error('key store locked') + }) + await expect(upgrade()).rejects.toThrow('key store locked') + expect(await readFile(join(dataPath, 'data.db'))).toEqual(before) + expect(await listCheckpoints(recoveryPath)).toEqual([]) +}) diff --git a/apps/electron/src/storage/migrations.ts b/apps/electron/src/storage/migrations.ts new file mode 100644 index 000000000..9aa5fc14b --- /dev/null +++ b/apps/electron/src/storage/migrations.ts @@ -0,0 +1,134 @@ +import { randomUUID } from 'node:crypto' +import { constants } from 'node:fs' +import { cp, mkdir, rm } from 'node:fs/promises' +import { join } from 'node:path' +import { SCHEMA_DDL, SCHEMA_MIGRATIONS, SCHEMA_VERSION } from '@xnetjs/sqlite' +import Database from 'better-sqlite3' +import { createCheckpoint } from './checkpoints' +import { + inspectDatabase, + requireCompatibleDatabase, + WorkspaceRecoveryRequired +} from './compatibility' +import { restoreCheckpoint } from './restore' + +// Version 8 is the oldest independently captured fixture verified by this desktop migration path. +export const OLDEST_DESKTOP_STORAGE_VERSION = 8 +const quote = (name: string) => `"${name.replaceAll('"', '""')}"` +const rowCount = (db: Database.Database, name: string) => + (db.prepare(`SELECT COUNT(*) AS count FROM ${quote(name)}`).get() as { count: number }).count + +/** Apply ordered upgrades only to a disposable candidate; validate its actual column contract. */ +export function migrateDatabaseCandidate(path: string, fromVersion: number): void { + if (fromVersion < OLDEST_DESKTOP_STORAGE_VERSION || fromVersion >= SCHEMA_VERSION) + throw new Error('Unsupported candidate migration version') + const db = new Database(path, { fileMustExist: true }) + const expected = new Database(':memory:') + try { + if ( + ( + db.prepare('SELECT MAX(version) AS version FROM _schema_version').get() as { + version: number + } + ).version !== fromVersion + ) + throw new Error('Candidate version differs from the inspected source') + db.pragma('synchronous = FULL') + db.pragma('foreign_keys = ON') + expected.exec(SCHEMA_DDL) + const oldTables = db + .prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'") + .all() as { name: string }[] + const oldCounts = new Map(oldTables.map(({ name }) => [name, rowCount(db, name)])) + db.transaction(() => { + for (let version = fromVersion + 1; version <= SCHEMA_VERSION; version++) { + const sql = SCHEMA_MIGRATIONS[version] + if (!sql) throw new Error(`Missing ordered migration for storage version ${version}`) + db.exec(sql) + db.prepare('INSERT INTO _schema_version (version, applied_at) VALUES (?, ?)').run( + version, + Date.now() + ) + } + const tables = expected + .prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'") + .all() as { name: string }[] + for (const { name } of tables) { + const required = expected.prepare(`PRAGMA table_info(${quote(name)})`).all() as { + name: string + type: string + pk: number + }[] + const actual = db.prepare(`PRAGMA table_info(${quote(name)})`).all() as { + name: string + type: string + pk: number + }[] + for (const column of required) + if ( + !actual.some( + (candidate) => + candidate.name === column.name && + candidate.type === column.type && + candidate.pk === column.pk + ) + ) + throw new Error( + `Candidate is missing the expected column contract: ${name}.${column.name}` + ) + } + // These migrations add structure; deleting even derived rows would violate their contract. + for (const [name, count] of oldCounts) + if (name !== '_schema_version' && rowCount(db, name) !== count) + throw new Error(`Candidate migration changed existing row counts: ${name}`) + if ( + db.pragma('quick_check', { simple: true }) !== 'ok' || + (db.pragma('foreign_key_check') as unknown[]).length > 0 + ) + throw new Error('Candidate failed database integrity validation') + })() + db.pragma('wal_checkpoint(TRUNCATE)') + } finally { + expected.close() + db.close() + } +} + +/** Before any normal writable open. Originals and the pinned pre-upgrade copy both survive. */ +export async function prepareWorkspaceUpgrade(options: { + dataPath: string + recoveryPath: string + profile: string + appVersion: string + testIdentity?: boolean + requireIdentity: () => void +}): Promise { + const dbPath = join(options.dataPath, 'data.db') + const state = inspectDatabase(dbPath) + requireCompatibleDatabase(join(options.dataPath, 'xnet.db'), 'blobs') + if (state.status === 'missing' || state.status === 'supported') return false + if (state.status !== 'old' || state.version < OLDEST_DESKTOP_STORAGE_VERSION) + throw new WorkspaceRecoveryRequired(dbPath, state) + options.requireIdentity() + const original = await createCheckpoint({ ...options, pinned: true }) + const container = join(options.recoveryPath, 'migrations', randomUUID()) + const candidate = join(container, 'workspace') + await mkdir(container, { recursive: true, mode: 0o700 }) + try { + await cp(join(options.recoveryPath, original.id, 'workspace'), candidate, { + recursive: true, + mode: constants.COPYFILE_FICLONE + }) + migrateDatabaseCandidate(join(candidate, 'data.db'), state.version) + const upgraded = await createCheckpoint({ ...options, dataPath: candidate }) + await restoreCheckpoint({ + ...options, + id: upgraded.id, + allowTestIdentity: options.testIdentity + }) + return true + } finally { + // This is only the disposable candidate. Original generations and recovery points are separate. + await rm(container, { recursive: true, force: true }) + } +} diff --git a/apps/electron/src/storage/portable.test.ts b/apps/electron/src/storage/portable.test.ts new file mode 100644 index 000000000..946cc3886 --- /dev/null +++ b/apps/electron/src/storage/portable.test.ts @@ -0,0 +1,238 @@ +import type { SafeStorageLike } from '../main/secure-seed' +import { createCipheriv, createHash, randomBytes, scryptSync } from 'node:crypto' +import { mkdtemp, mkdir, readFile, readdir, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import Database from 'better-sqlite3' +import { beforeEach, afterEach, expect, it, vi } from 'vitest' +import { getOrCreateIdentitySeed } from '../main/identity-seed' +import { loadSeedPhrase, storeSeedPhrase } from '../main/secure-seed' +import { captureDesktopSettings } from '../shared/desktop-settings' +import { createCheckpoint, verifyCheckpoint } from './checkpoints' +import { readDesktopSettings, writeDesktopSettings } from './desktop-settings' +import { exportPortableCheckpoint, unpackPortableCheckpoint } from './portable' + +const safe = (mac: string): SafeStorageLike => ({ + isEncryptionAvailable: () => true, + encryptString: (text) => Buffer.from(`${mac}:${text}`), + decryptString: (bytes) => { + const text = bytes.toString() + if (!text.startsWith(`${mac}:`)) throw new Error('This key belongs to another Mac') + return text.slice(mac.length + 1) + } +}) +// Production-cost scrypt and fsync run repeatedly; shared CI CPUs need a bounded larger budget. +vi.setConfig({ testTimeout: 60_000 }) + +const password = 'test-only four random recovery words' +let root: string +let dataPath: string +let recoveryPath: string +let destination: string +let checkpointPath: string +let seed: Uint8Array + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-portable-')) + dataPath = join(root, 'data') + recoveryPath = join(root, 'recovery') + destination = join(root, 'external') + await mkdir(dataPath) + await mkdir(destination) + seed = getOrCreateIdentitySeed(dataPath, safe('first-Mac'), { profile: 'daily' }).seed + storeSeedPhrase(dataPath, 'fixture recovery mnemonic', safe('first-Mac')) + for (const name of ['data.db', 'xnet.db']) { + const db = new Database(join(dataPath, name)) + db.exec("CREATE TABLE notes(body TEXT); INSERT INTO notes VALUES ('private note body')") + db.close() + } + await mkdir(join(dataPath, 'sources')) + await writeFile(join(dataPath, 'sources', 'export.json'), '{"private":"source evidence"}') + const point = await createCheckpoint({ + dataPath, + recoveryPath, + appVersion: 'fixture', + profile: 'daily' + }) + checkpointPath = join(recoveryPath, point.id) +}) +afterEach(() => rm(root, { recursive: true, force: true })) +const exportPoint = () => + exportPortableCheckpoint({ + checkpointPath, + destination, + password, + safeStorage: safe('first-Mac') + }) +const unpack = (path: string, recoveryPassword = password, output = join(root, 'incoming')) => + unpackPortableCheckpoint({ + path, + output, + password: recoveryPassword, + safeStorage: safe('second-Mac') + }) + +it('still reads a version-1 backup without desktop settings', async () => { + // Independent legacy writer: an old backup must not depend on the current export format. + const format = 'xnet-desktop-portable/1' + const salt = randomBytes(32) + const key = scryptSync(password, salt, 32, { N: 131072, r: 8, p: 1, maxmem: 256 * 1024 * 1024 }) + const sealLegacy = (bytes: Buffer) => { + const nonce = randomBytes(12) + const cipher = createCipheriv('aes-256-gcm', key, nonce) + cipher.setAAD(Buffer.from(format)) + return Buffer.concat([nonce, cipher.update(bytes), cipher.final(), cipher.getAuthTag()]) + } + const path = join(destination, 'legacy.xnetbackup') + await mkdir(join(path, 'objects'), { recursive: true }) + const checkpoint = await verifyCheckpoint(checkpointPath) + const chunks: Record = {} + for (const file of checkpoint.files) { + const encrypted = sealLegacy(await readFile(join(checkpointPath, 'workspace', file.path))) + const hash = createHash('sha256').update(encrypted).digest('hex') + chunks[file.path] = [hash] + await writeFile(join(path, 'objects', hash), encrypted) + } + await writeFile( + join(path, 'header.json'), + JSON.stringify({ format, kdf: 'scrypt-N131072-r8-p1', salt: salt.toString('base64') }) + ) + await writeFile( + join(path, 'manifest.enc'), + sealLegacy( + Buffer.from( + JSON.stringify({ + format, + checkpoint, + chunks, + seedB64: Buffer.from(seed).toString('base64'), + mnemonic: loadSeedPhrase(dataPath, safe('first-Mac')) + }) + ) + ) + ) + key.fill(0) + const restored = await unpack(path) + expect( + getOrCreateIdentitySeed(restored.workspace, safe('second-Mac'), { profile: 'daily' }).seed + ).toEqual(seed) + expect(await readDesktopSettings(restored.workspace, safe('second-Mac'))).toBeNull() +}) + +it('restores exact content and the same identity without the original Mac key store', async () => { + const exported = await exportPoint() + await rm(dataPath, { recursive: true }) + await rm(recoveryPath, { recursive: true }) + const restored = await unpack(exported.path) + expect( + getOrCreateIdentitySeed(restored.workspace, safe('second-Mac'), { profile: 'daily' }).seed + ).toEqual(seed) + expect(loadSeedPhrase(restored.workspace, safe('second-Mac'))).toBe('fixture recovery mnemonic') + expect(await readFile(join(restored.workspace, 'sources/export.json'), 'utf8')).toContain( + 'source evidence' + ) + const db = new Database(join(restored.workspace, 'data.db'), { readonly: true }) + expect(db.prepare('SELECT body FROM notes').get()).toEqual({ body: 'private note body' }) + db.close() + const local = await createCheckpoint({ + dataPath: restored.workspace, + recoveryPath, + profile: 'new-Mac', + appVersion: 'fixture' + }) + await verifyCheckpoint(join(recoveryPath, local.id)) + const clear = Buffer.concat( + await Promise.all( + (await readdir(exported.path)) + .filter((name) => name !== 'objects') + .map((name) => readFile(join(exported.path, name))) + ) + ) + expect(clear.includes(Buffer.from(seed).toString('base64'))).toBe(false) + expect(clear.includes('private note body')).toBe(false) +}) + +it('authenticates a multi-chunk file without losing its tail', async () => { + const bytes = Buffer.alloc(4 * 1024 * 1024 + 53, 42) + bytes.write('important tail', bytes.length - 14) + await writeFile(join(dataPath, 'large.bin'), bytes) + const point = await createCheckpoint({ + dataPath, + recoveryPath, + profile: 'daily', + appVersion: 'fixture' + }) + checkpointPath = join(recoveryPath, point.id) + const restored = await unpack((await exportPoint()).path) + expect(await readFile(join(restored.workspace, 'large.bin'))).toEqual(bytes) +}) + +it('rewraps preferences, a provider key, and a draft for a different Mac key store', async () => { + const settings = captureDesktopSettings({ getItem: () => null }) + settings.values['xnet:ai-api-key'] = 'fixture-provider-secret' + settings.values['xnet.library.capture-draft.v1'] = '{"note":"unfinished private thought"}' + settings.values['xnet-electron-theme'] = 'light' + await writeDesktopSettings(dataPath, settings, safe('first-Mac')) + const point = await createCheckpoint({ + dataPath, + recoveryPath, + profile: 'daily', + appVersion: 'fixture' + }) + checkpointPath = join(recoveryPath, point.id) + const exported = await exportPoint() + await rm(dataPath, { recursive: true }) + await rm(recoveryPath, { recursive: true }) + const restored = await unpack(exported.path) + expect(await readDesktopSettings(restored.workspace, safe('second-Mac'))).toEqual(settings) + await expect(readDesktopSettings(restored.workspace, safe('first-Mac'))).rejects.toThrow( + 'another Mac' + ) + for (const object of await readdir(join(exported.path, 'objects'))) { + const bytes = await readFile(join(exported.path, 'objects', object)) + expect(bytes.includes('fixture-provider-secret')).toBe(false) + expect(bytes.includes('unfinished private thought')).toBe(false) + } +}) + +it('rejects a wrong password before creating any plaintext output', async () => { + const exported = await exportPoint() + await expect(unpack(exported.path, 'wrong recovery password')).rejects.toThrow('Could not unlock') + expect(await readdir(root)).not.toContain('incoming') +}) + +it.each(['damage', 'remove'] as const)( + 'rejects %s to a chunk and cleans its disposable output', + async (action) => { + const exported = await exportPoint() + const [name] = await readdir(join(exported.path, 'objects')) + const path = join(exported.path, 'objects', name) + if (action === 'remove') await rm(path) + else await writeFile(path, 'damaged cipher bytes') + await expect(unpack(exported.path)).rejects.toThrow() + expect(await readdir(root)).not.toContain('incoming') + await verifyCheckpoint(checkpointPath) + } +) + +it('never replaces an existing destination workspace', async () => { + const exported = await exportPoint() + const before = await readFile(join(dataPath, 'data.db')) + await expect(unpack(exported.path, password, dataPath)).rejects.toThrow() + expect(await readFile(join(dataPath, 'data.db'))).toEqual(before) +}) + +it('rejects a damaged manifest and refuses weak recovery passwords', async () => { + await expect( + exportPortableCheckpoint({ + checkpointPath, + destination, + password: 'short', + safeStorage: safe('first-Mac') + }) + ).rejects.toThrow('12 characters') + const exported = await exportPoint() + await writeFile(join(exported.path, 'manifest.enc'), 'damaged') + await expect(unpack(exported.path)).rejects.toThrow('Could not unlock') + expect(await readdir(root)).not.toContain('incoming') +}) diff --git a/apps/electron/src/storage/portable.ts b/apps/electron/src/storage/portable.ts new file mode 100644 index 000000000..e832b3dfa --- /dev/null +++ b/apps/electron/src/storage/portable.ts @@ -0,0 +1,383 @@ +import type { SafeStorageLike } from '../main/secure-seed' +import type { DesktopSettings } from '../shared/desktop-settings' +import type { CheckpointManifest } from '../shared/recovery' +import { + createCipheriv, + createDecipheriv, + createHash, + randomBytes, + randomUUID, + scrypt +} from 'node:crypto' +import { createReadStream } from 'node:fs' +import { lstat, mkdir, open, readFile, readdir, rename, rm, writeFile } from 'node:fs/promises' +import { dirname, join, resolve, sep } from 'node:path' +import { getOrCreateIdentitySeed } from '../main/identity-seed' +import { loadSeedPhrase, storeSeedPhrase } from '../main/secure-seed' +import { validateDesktopSettings } from '../shared/desktop-settings' +import { verifyCheckpoint } from './checkpoints' +import { readDesktopSettings, SETTINGS_FILE, writeDesktopSettings } from './desktop-settings' + +const FORMAT = 'xnet-desktop-portable/2' +const LEGACY_FORMAT = 'xnet-desktop-portable/1' +const NONCE_SIZE = 12 +const TAG_SIZE = 16 +const CHUNK_BYTES = 4 * 1024 * 1024 +const MANIFEST_LIMIT = 64 * 1024 * 1024 +const KDF = 'scrypt-N131072-r8-p1' +const hash = (bytes: Uint8Array) => createHash('sha256').update(bytes).digest('hex') +const isRecord = (value: unknown): value is Record => + value !== null && typeof value === 'object' && !Array.isArray(value) + +type PortableManifest = { + format: typeof FORMAT | typeof LEGACY_FORMAT + checkpoint: CheckpointManifest + seedB64: string + mnemonic: string | null + desktopSettings?: DesktopSettings | null + chunks: Record +} + +function validatePassword(password: string): void { + if (typeof password !== 'string' || password.length < 12 || Buffer.byteLength(password) > 1024) + throw new Error( + 'Use a recovery password of at least 12 characters, preferably several random words.' + ) +} + +async function deriveKey(password: string, salt: Buffer): Promise { + validatePassword(password) + return new Promise((resolveKey, reject) => + scrypt( + password, + salt, + 32, + { N: 131072, r: 8, p: 1, maxmem: 256 * 1024 * 1024 }, + (error, key) => (error ? reject(error) : resolveKey(key)) + ) + ) +} + +function seal(bytes: Uint8Array, key: Uint8Array): Buffer { + const nonce = randomBytes(NONCE_SIZE) + const cipher = createCipheriv('aes-256-gcm', key, nonce) + cipher.setAAD(Buffer.from(FORMAT)) + return Buffer.concat([nonce, cipher.update(bytes), cipher.final(), cipher.getAuthTag()]) +} + +function unseal(bytes: Uint8Array, key: Uint8Array, format: string): Uint8Array { + if (bytes.length < NONCE_SIZE + TAG_SIZE) throw new Error('Incomplete encrypted recovery object') + const decipher = createDecipheriv('aes-256-gcm', key, bytes.subarray(0, NONCE_SIZE)) + decipher.setAAD(Buffer.from(format)) + decipher.setAuthTag(bytes.subarray(bytes.length - TAG_SIZE)) + return Buffer.concat([decipher.update(bytes.subarray(NONCE_SIZE, -TAG_SIZE)), decipher.final()]) +} + +function safePath(root: string, path: string): string { + if ( + !path || + path.includes('\\') || + path.split('/').some((part) => !part || part === '.' || part === '..') + ) + throw new Error('Invalid portable recovery path') + const target = resolve(root, path) + if (!target.startsWith(resolve(root) + sep)) + throw new Error('Portable recovery path escapes its directory') + return target +} + +async function sync(path: string): Promise { + const handle = await open(path, 'r') + try { + await handle.sync() + } finally { + await handle.close() + } +} + +async function readBounded(path: string, limit: number): Promise { + const info = await lstat(path) + if (!info.isFile() || info.isSymbolicLink() || info.size > limit) + throw new Error('Invalid or oversized portable recovery object') + const bytes = await readFile(path) + if (bytes.length !== info.size || bytes.length > limit) + throw new Error('Portable recovery object changed during reading') + return bytes +} + +async function writePrivate(path: string, bytes: Uint8Array): Promise { + const file = await open(path, 'wx', 0o600) + try { + await file.writeFile(bytes) + await file.sync() + } finally { + await file.close() + } +} + +/** Encrypt a verified native point. No plaintext keys or filenames enter the destination. */ +export async function exportPortableCheckpoint(options: { + checkpointPath: string + destination: string + password: string + safeStorage: SafeStorageLike + allowTestIdentity?: boolean +}): Promise<{ path: string; createdAt: string; files: number; bytes: number }> { + const checkpoint = await verifyCheckpoint(options.checkpointPath, options) + const workspace = join(options.checkpointPath, 'workspace') + const destination = resolve(options.destination) + if ( + destination === resolve(options.checkpointPath) || + destination.startsWith(resolve(options.checkpointPath) + sep) + ) + throw new Error('Choose a backup destination outside the recovery point') + const targetInfo = await lstat(destination) + if (!targetInfo.isDirectory() || targetInfo.isSymbolicLink()) + throw new Error('Choose a regular backup directory') + const salt = randomBytes(32) + const key = await deriveKey(options.password, salt) + const id = `xnet-${Date.now()}-${randomUUID()}.xnetbackup` + const temporary = join(destination, `.${id}.incomplete`) + const finalPath = join(destination, id) + let sourceIdentity: ReturnType | undefined + try { + sourceIdentity = getOrCreateIdentitySeed(workspace, options.safeStorage, { + profile: checkpoint.profile, + testMode: checkpoint.identity === 'test' + }) + const mnemonic = loadSeedPhrase(workspace, options.safeStorage) + const desktopSettings = await readDesktopSettings(workspace, options.safeStorage) + await mkdir(join(temporary, 'objects'), { recursive: true, mode: 0o700 }) + const chunks: Record = Object.create(null) as Record + for (const file of checkpoint.files) { + const objects: string[] = [] + for await (const bytes of createReadStream(safePath(workspace, file.path), { + highWaterMark: CHUNK_BYTES + })) { + const encrypted = seal(bytes as Buffer, key) + const objectId = hash(encrypted) + await writePrivate(join(temporary, 'objects', objectId), encrypted) + objects.push(objectId) + } + chunks[file.path] = objects + } + const manifest: PortableManifest = { + format: FORMAT, + checkpoint, + seedB64: Buffer.from(sourceIdentity.seed).toString('base64'), + mnemonic, + desktopSettings, + chunks + } + const encodedManifest = Buffer.from(JSON.stringify(manifest)) + if (encodedManifest.length > MANIFEST_LIMIT) + throw new Error('Recovery manifest exceeds its supported size') + await writePrivate(join(temporary, 'manifest.enc'), seal(encodedManifest, key)) + encodedManifest.fill(0) + await writePrivate( + join(temporary, 'header.json'), + Buffer.from(JSON.stringify({ format: FORMAT, kdf: KDF, salt: salt.toString('base64') })) + ) + // Authentication and every plaintext hash must pass before publishing the encrypted folder. + const verified = await readManifest(temporary, options.password) + try { + await verifyObjects(temporary, verified.manifest, verified.key) + } finally { + verified.key.fill(0) + } + await sync(join(temporary, 'objects')) + await sync(temporary) + await rename(temporary, finalPath) + await sync(destination) + return { + path: finalPath, + createdAt: checkpoint.createdAt, + files: checkpoint.files.length, + bytes: checkpoint.files.reduce((sum, file) => sum + file.size, 0) + } + } finally { + key.fill(0) + sourceIdentity?.seed.fill(0) + await rm(temporary, { recursive: true, force: true }) + } +} + +async function readManifest( + path: string, + password: string +): Promise<{ manifest: PortableManifest; key: Buffer }> { + const info = await lstat(path) + if (!info.isDirectory() || info.isSymbolicLink()) + throw new Error('Choose a regular encrypted backup directory') + const header: unknown = JSON.parse( + (await readBounded(join(path, 'header.json'), 4096)).toString('utf8') + ) + if ( + !isRecord(header) || + (header.format !== FORMAT && header.format !== LEGACY_FORMAT) || + header.kdf !== KDF || + typeof header.salt !== 'string' + ) + throw new Error('Unsupported portable recovery format') + const salt = Buffer.from(header.salt, 'base64') + if (salt.length !== 32 || salt.toString('base64') !== header.salt) + throw new Error('Invalid recovery salt') + const key = await deriveKey(password, salt) + try { + const encoded = await readBounded( + join(path, 'manifest.enc'), + MANIFEST_LIMIT + NONCE_SIZE + TAG_SIZE + ) + const bytes = unseal(encoded, key, header.format) + const value: unknown = JSON.parse(Buffer.from(bytes).toString('utf8')) + bytes.fill(0) + if ( + !isRecord(value) || + value.format !== header.format || + !isRecord(value.checkpoint) || + !Array.isArray(value.checkpoint.files) || + !isRecord(value.chunks) || + typeof value.seedB64 !== 'string' || + (value.mnemonic !== null && typeof value.mnemonic !== 'string') + ) + throw new Error('Invalid portable recovery manifest') + const seed = Buffer.from(value.seedB64, 'base64') + if (seed.length !== 32 || seed.toString('base64') !== value.seedB64) + throw new Error('Invalid portable identity') + seed.fill(0) + const manifest = value as unknown as PortableManifest + const hasSettings = manifest.checkpoint.files.some((file) => file?.path === SETTINGS_FILE) + if (manifest.format === FORMAT) { + if (manifest.desktopSettings !== null) validateDesktopSettings(manifest.desktopSettings) + if (hasSettings !== (manifest.desktopSettings !== null)) + throw new Error('Portable desktop settings do not match the checkpoint inventory') + } else if (hasSettings) { + throw new Error('This legacy backup cannot rewrap desktop settings') + } + const paths = new Set() + for (const file of manifest.checkpoint.files) { + if ( + !file || + typeof file.path !== 'string' || + !Number.isSafeInteger(file.size) || + file.size < 0 || + typeof file.sha256 !== 'string' || + !/^[a-f0-9]{64}$/.test(file.sha256) + ) + throw new Error('Invalid portable file inventory') + safePath('/workspace', file.path) + if (paths.has(file.path)) throw new Error('Duplicate portable file path') + paths.add(file.path) + const objects = manifest.chunks[file.path] + if ( + !Array.isArray(objects) || + objects.some((id) => typeof id !== 'string' || !/^[a-f0-9]{64}$/.test(id)) + ) + throw new Error('Invalid encrypted recovery object list') + } + if (paths.size !== Object.keys(manifest.chunks).length) + throw new Error('Portable inventory does not match') + return { manifest, key } + } catch (error) { + key.fill(0) + throw new Error( + 'Could not unlock this backup. Check the password and that the copy is complete.', + { cause: error } + ) + } +} + +async function verifyObjects( + path: string, + manifest: PortableManifest, + key: Buffer, + output?: string +): Promise { + const expectedObjects = new Set(Object.values(manifest.chunks).flat()) + const directoryInfo = await lstat(join(path, 'objects')) + if (!directoryInfo.isDirectory() || directoryInfo.isSymbolicLink()) + throw new Error('Invalid recovery objects directory') + const actual = await readdir(join(path, 'objects')) + if (actual.length !== expectedObjects.size || actual.some((name) => !expectedObjects.has(name))) + throw new Error('Encrypted recovery objects are missing or unexpected') + for (const file of manifest.checkpoint.files) { + const digest = createHash('sha256') + let size = 0 + const target = output ? safePath(output, file.path) : null + if (target) await mkdir(dirname(target), { recursive: true, mode: 0o700 }) + const handle = target ? await open(target, 'wx', 0o600) : null + try { + for (const id of manifest.chunks[file.path]) { + const encrypted = await readBounded( + join(path, 'objects', id), + CHUNK_BYTES + NONCE_SIZE + TAG_SIZE + ) + if (hash(encrypted) !== id) throw new Error('Encrypted recovery object is damaged') + const bytes = unseal(encrypted, key, manifest.format) + size += bytes.byteLength + if (size > file.size) throw new Error('Recovered file exceeds the recorded size') + digest.update(bytes) + if (handle) await handle.writeFile(bytes) + bytes.fill(0) + } + if (size !== file.size || digest.digest('hex') !== file.sha256) + throw new Error(`Recovered file is incomplete or damaged: ${file.path}`) + if (handle) await handle.sync() + } finally { + await handle?.close() + } + } +} + +/** Recover into a NEW private directory only. Live workspace replacement is a separate step. */ +export async function unpackPortableCheckpoint(options: { + path: string + output: string + password: string + safeStorage: SafeStorageLike + allowTestIdentity?: boolean +}): Promise<{ source: CheckpointManifest; workspace: string }> { + const { manifest, key } = await readManifest(options.path, options.password) + let created = false + try { + if (!options.safeStorage.isEncryptionAvailable()) + throw new Error('Unlock the destination Mac key store before restoring the identity.') + await mkdir(options.output, { mode: 0o700 }) + created = true + const workspace = join(options.output, 'workspace') + await mkdir(workspace, { mode: 0o700 }) + await verifyObjects(options.path, manifest, key, workspace) + await writePrivate( + join(options.output, 'manifest.json'), + Buffer.from(JSON.stringify(manifest.checkpoint)) + ) + await verifyCheckpoint(options.output, options) + // The following rewrapping changes native file hashes; the caller creates a fresh local point. + await rm(join(options.output, 'manifest.json')) + // The authenticated export carries the actual seed, so the old Mac's key store is unnecessary. + await writeFile( + join(workspace, 'identity-seed.json'), + JSON.stringify({ + version: 1, + payload: options.safeStorage.encryptString(manifest.seedB64).toString('base64'), + updatedAt: Date.now() + }), + { mode: 0o600 } + ) + if (manifest.mnemonic !== null) + storeSeedPhrase(workspace, manifest.mnemonic, options.safeStorage) + if (manifest.desktopSettings) + await writeDesktopSettings(workspace, manifest.desktopSettings, options.safeStorage, { + rewrap: true + }) + await sync(join(workspace, 'identity-seed.json')) + if (manifest.mnemonic !== null) await sync(join(workspace, 'seed-recovery.json')) + await sync(workspace) + return { source: manifest.checkpoint, workspace } + } catch (error) { + if (created) await rm(options.output, { recursive: true, force: true }) + throw error + } finally { + key.fill(0) + } +} diff --git a/apps/electron/src/storage/restore.test.ts b/apps/electron/src/storage/restore.test.ts new file mode 100644 index 000000000..a4cf08b30 --- /dev/null +++ b/apps/electron/src/storage/restore.test.ts @@ -0,0 +1,87 @@ +import { mkdtemp, mkdir, writeFile, readFile, readdir, rm, cp, rename } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import Database from 'better-sqlite3' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { createCheckpoint } from './checkpoints' +import { restoreCheckpoint, recoverPendingRestore } from './restore' + +let root: string +let dataPath: string +let recoveryPath: string +let id: string +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'xnet-restore-')) + dataPath = join(root, 'data') + recoveryPath = join(root, 'recovery') + await mkdir(dataPath) + for (const name of ['data.db', 'xnet.db']) { + const db = new Database(join(dataPath, name)) + db.exec('CREATE TABLE notes (body TEXT)') + db.close() + } + await writeFile(join(dataPath, 'identity-seed.json'), 'same key bytes') + await writeFile(join(dataPath, 'note.txt'), 'before update') + id = (await createCheckpoint({ dataPath, recoveryPath, profile: 'daily', appVersion: 'test' })).id + await writeFile(join(dataPath, 'note.txt'), 'newer work') +}) +afterEach(() => rm(root, { recursive: true, force: true })) + +describe('restore with retained generations', () => { + it('restores a verified point and preserves all newer work', async () => { + await restoreCheckpoint({ id, dataPath, recoveryPath, profile: 'daily' }) + expect(await readFile(join(dataPath, 'note.txt'), 'utf8')).toBe('before update') + const [generation] = await readdir(join(recoveryPath, 'preserved')) + expect(await readFile(join(recoveryPath, 'preserved', generation, 'note.txt'), 'utf8')).toBe( + 'newer work' + ) + expect(await readFile(join(dataPath, 'identity-seed.json'), 'utf8')).toBe('same key bytes') + expect( + JSON.parse(await readFile(join(recoveryPath, 'review-required.json'), 'utf8')).version + ).toBe(1) + }) + + it('rejects tampering before replacing the live workspace', async () => { + await writeFile(join(recoveryPath, id, 'workspace/note.txt'), 'tampered') + await expect( + restoreCheckpoint({ id, dataPath, recoveryPath, profile: 'daily' }) + ).rejects.toThrow('verification failed') + expect(await readFile(join(dataPath, 'note.txt'), 'utf8')).toBe('newer work') + }) + + it.each(['before-rename', 'between-renames', 'after-promotion'] as const)( + 'resumes a crash %s before normal database startup', + async (phase) => { + const candidate = join(recoveryPath, 'restore', id) + await cp(join(recoveryPath, id), candidate, { recursive: true }) + await writeFile( + join(recoveryPath, 'pending-restore.json'), + JSON.stringify({ version: 1, id }) + ) + if (phase !== 'before-rename') { + await mkdir(join(recoveryPath, 'preserved')) + await rename(dataPath, join(recoveryPath, 'preserved', id)) + } + if (phase === 'after-promotion') await rename(join(candidate, 'workspace'), dataPath) + await recoverPendingRestore(dataPath, recoveryPath) + expect(await readFile(join(dataPath, 'note.txt'), 'utf8')).toBe('before update') + expect(await readFile(join(recoveryPath, 'preserved', id, 'note.txt'), 'utf8')).toBe( + 'newer work' + ) + await expect(readFile(join(recoveryPath, 'pending-restore.json'))).rejects.toMatchObject({ + code: 'ENOENT' + }) + } + ) + + it('does not let a journal choose paths outside recovery storage', async () => { + await writeFile( + join(recoveryPath, 'pending-restore.json'), + JSON.stringify({ version: 1, id: '../../outside' }) + ) + await expect(recoverPendingRestore(dataPath, recoveryPath)).rejects.toThrow( + 'Unreadable restore journal' + ) + expect(await readFile(join(dataPath, 'note.txt'), 'utf8')).toBe('newer work') + }) +}) diff --git a/apps/electron/src/storage/restore.ts b/apps/electron/src/storage/restore.ts new file mode 100644 index 000000000..6bf891660 --- /dev/null +++ b/apps/electron/src/storage/restore.ts @@ -0,0 +1,120 @@ +import { randomUUID } from 'node:crypto' +import { constants } from 'node:fs' +import { copyFile, mkdir, open, readFile, rename, rm, stat } from 'node:fs/promises' +import { dirname, join } from 'node:path' +import { verifyCheckpoint } from './checkpoints' + +const JOURNAL = 'pending-restore.json' +const validId = (id: string) => /^\d+-[a-f0-9-]{36}$/.test(id) + +async function exists(path: string): Promise { + try { + await stat(path) + return true + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return false + throw error + } +} + +async function sync(path: string): Promise { + const file = await open(path, 'r') + try { + await file.sync() + } finally { + await file.close() + } +} + +/** Finish a recorded restore before normal startup can mistake a rename gap for a new workspace. */ +export async function recoverPendingRestore( + dataPath: string, + recoveryPath: string, + options: { allowTestIdentity?: boolean } = {} +): Promise { + const journal = join(recoveryPath, JOURNAL) + if (!(await exists(journal))) return + const value = JSON.parse(await readFile(journal, 'utf8')) as { version?: unknown; id?: unknown } + if (value.version !== 1 || typeof value.id !== 'string' || !validId(value.id)) + throw new Error('Unreadable restore journal. Workspace copies have been preserved.') + const container = join(recoveryPath, 'restore', value.id) + const candidate = join(container, 'workspace') + const preserved = join(recoveryPath, 'preserved', value.id) + if (await exists(candidate)) { + await verifyCheckpoint(container, options) + if (await exists(dataPath)) { + if (await exists(preserved)) + throw new Error('Ambiguous restore state. All workspace generations were preserved.') + await mkdir(dirname(preserved), { recursive: true, mode: 0o700 }) + await rename(dataPath, preserved) + await sync(dirname(dataPath)) + await sync(dirname(preserved)) + } else if (!(await exists(preserved))) { + throw new Error('The original workspace is missing during restore. No files were replaced.') + } + await rename(candidate, dataPath) + await sync(dirname(dataPath)) + await sync(dirname(candidate)) + } else if (!(await exists(dataPath)) || !(await exists(preserved))) { + throw new Error('Incomplete restore state. No workspace was reset.') + } + // A restored identity must not reconnect and replay older state before review. + const review = await open(join(recoveryPath, 'review-required.json'), 'w', 0o600) + try { + await review.writeFile( + JSON.stringify({ version: 1, id: value.id, restoredAt: new Date().toISOString() }) + ) + await review.sync() + } finally { + await review.close() + } + await rm(journal) + await sync(recoveryPath) +} + +/** Requires stopped workspace writers. The pre-restore generation is retained in full. */ +export async function restoreCheckpoint(options: { + id: string + dataPath: string + recoveryPath: string + profile: string + allowTestIdentity?: boolean +}): Promise { + if (!validId(options.id)) throw new Error('Invalid recovery point') + const source = join(options.recoveryPath, options.id) + const manifest = await verifyCheckpoint(source, options) + if (manifest.profile !== options.profile) + throw new Error('This recovery point belongs to another profile') + if (await exists(join(options.recoveryPath, JOURNAL))) + throw new Error('Finish the pending restore before starting another') + // Repeated restores must preserve each generation rather than overwrite the last one. + const id = `${Date.now()}-${randomUUID()}` + const container = join(options.recoveryPath, 'restore', id) + await mkdir(join(container, 'workspace'), { recursive: true, mode: 0o700 }) + for (const file of manifest.files) { + const target = join(container, 'workspace', file.path) + await mkdir(dirname(target), { recursive: true, mode: 0o700 }) + await copyFile(join(source, 'workspace', file.path), target, constants.COPYFILE_FICLONE) + await sync(target) + await sync(dirname(target)) + } + await copyFile(join(source, 'manifest.json'), join(container, 'manifest.json')) + await verifyCheckpoint(container, options) + await sync(join(container, 'workspace')) + await sync(join(container, 'manifest.json')) + await sync(container) + await sync(dirname(container)) + const journal = join(options.recoveryPath, JOURNAL) + const temporary = `${journal}.${id}.tmp` + const file = await open(temporary, 'wx', 0o600) + try { + await file.writeFile(JSON.stringify({ version: 1, id })) + await file.sync() + } finally { + await file.close() + } + await rename(temporary, journal) + await sync(options.recoveryPath) + await recoverPendingRestore(options.dataPath, options.recoveryPath, options) + await rm(container, { recursive: true }) +} diff --git a/apps/electron/tsconfig.node.json b/apps/electron/tsconfig.node.json index 0862abc73..cc969e44d 100644 --- a/apps/electron/tsconfig.node.json +++ b/apps/electron/tsconfig.node.json @@ -9,8 +9,12 @@ "outDir": "./out" }, "include": [ + "src/shared/**/*.ts", "src/main/**/*.ts", "src/preload/**/*.ts", + "src/storage/**/*.ts", + "src/library/**/*.ts", + "src/data-process/**/*.ts", "electron.vite.config.ts" ] } diff --git a/apps/electron/tsconfig.web.json b/apps/electron/tsconfig.web.json index aaa337160..6e7f0a460 100644 --- a/apps/electron/tsconfig.web.json +++ b/apps/electron/tsconfig.web.json @@ -6,5 +6,10 @@ "lib": ["ES2022", "DOM", "DOM.Iterable"], "noEmit": true }, - "include": ["src/renderer/**/*.ts", "src/renderer/**/*.tsx", "src/preload/index.ts"] + "include": [ + "src/shared/**/*.ts", + "src/renderer/**/*.ts", + "src/renderer/**/*.tsx", + "src/preload/index.ts" + ] } diff --git a/docs/explorations/0241_[_]_OPEN_COLLECTIVE_FOUNDATION_OR_COMPANY_LEGAL_AND_FUNDING_STRUCTURE.md b/docs/explorations/0241_[_]_OPEN_COLLECTIVE_FOUNDATION_OR_COMPANY_LEGAL_AND_FUNDING_STRUCTURE.md index 758eccab8..25ee72817 100644 --- a/docs/explorations/0241_[_]_OPEN_COLLECTIVE_FOUNDATION_OR_COMPANY_LEGAL_AND_FUNDING_STRUCTURE.md +++ b/docs/explorations/0241_[_]_OPEN_COLLECTIVE_FOUNDATION_OR_COMPANY_LEGAL_AND_FUNDING_STRUCTURE.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # 0241 — Open Collective, Foundation, Or Company? A Practical Legal & Funding Structure For xNet > **Status:** Exploration diff --git a/docs/explorations/0249_[_]_THE_COLD_OPEN_STALL_NAMING_THE_15S_QUERY_AND_THE_9S_IDENTITY_BUCKET.md b/docs/explorations/0249_[_]_THE_COLD_OPEN_STALL_NAMING_THE_15S_QUERY_AND_THE_9S_IDENTITY_BUCKET.md index 0d8ce5d8b..1195cb0c3 100644 --- a/docs/explorations/0249_[_]_THE_COLD_OPEN_STALL_NAMING_THE_15S_QUERY_AND_THE_9S_IDENTITY_BUCKET.md +++ b/docs/explorations/0249_[_]_THE_COLD_OPEN_STALL_NAMING_THE_15S_QUERY_AND_THE_9S_IDENTITY_BUCKET.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # The Cold-Open Stall, Take Six: Name the 15 s Query and Split the 9 s "Identity" Bucket ## Problem Statement diff --git a/docs/explorations/0250_[_]_THE_EVERYPERSON_SHELL_A_CLAUDE_DESKTOP_UI_FOR_XNET.md b/docs/explorations/0250_[_]_THE_EVERYPERSON_SHELL_A_CLAUDE_DESKTOP_UI_FOR_XNET.md index ce1ca3c5e..092862a64 100644 --- a/docs/explorations/0250_[_]_THE_EVERYPERSON_SHELL_A_CLAUDE_DESKTOP_UI_FOR_XNET.md +++ b/docs/explorations/0250_[_]_THE_EVERYPERSON_SHELL_A_CLAUDE_DESKTOP_UI_FOR_XNET.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # The Everyperson Shell — A Claude-Desktop UI/UX for xNet (Desktop + Mobile) ## Problem Statement diff --git a/docs/explorations/0251_[_]_ELECTRON_TO_DENO_DESKTOP_MIGRATION.md b/docs/explorations/0251_[_]_ELECTRON_TO_DENO_DESKTOP_MIGRATION.md index 8cebedb13..563bf0791 100644 --- a/docs/explorations/0251_[_]_ELECTRON_TO_DENO_DESKTOP_MIGRATION.md +++ b/docs/explorations/0251_[_]_ELECTRON_TO_DENO_DESKTOP_MIGRATION.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Electron → Deno Desktop: Should We Switch? > Status: `[_]` exploration. Prompt: *"should we switch from electron to deno diff --git a/docs/explorations/0252_[_]_WHY_THE_AI_CHAT_BOX_IS_DISABLED_LOCAL_MODEL_CONNECTOR_GAPS.md b/docs/explorations/0252_[_]_WHY_THE_AI_CHAT_BOX_IS_DISABLED_LOCAL_MODEL_CONNECTOR_GAPS.md index 0b4ffc93c..cfaf0e3a7 100644 --- a/docs/explorations/0252_[_]_WHY_THE_AI_CHAT_BOX_IS_DISABLED_LOCAL_MODEL_CONNECTOR_GAPS.md +++ b/docs/explorations/0252_[_]_WHY_THE_AI_CHAT_BOX_IS_DISABLED_LOCAL_MODEL_CONNECTOR_GAPS.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Why The AI Chat Box Is Disabled — Local-Model Connector Gaps ## Problem Statement diff --git a/docs/explorations/0253_[_]_THE_SEVENTH_COLD_OPEN_MIGRATION_THE_STALL_LEFT_EXECMS.md b/docs/explorations/0253_[_]_THE_SEVENTH_COLD_OPEN_MIGRATION_THE_STALL_LEFT_EXECMS.md index d9b9da0e1..f36349ad2 100644 --- a/docs/explorations/0253_[_]_THE_SEVENTH_COLD_OPEN_MIGRATION_THE_STALL_LEFT_EXECMS.md +++ b/docs/explorations/0253_[_]_THE_SEVENTH_COLD_OPEN_MIGRATION_THE_STALL_LEFT_EXECMS.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # The Seventh Cold‑Open Migration: The Stall Left `execMs` — Bracket the Open/Dispatch Window ## Problem Statement diff --git a/docs/explorations/0254_[_]_COMPACT_THE_CHANGE_LOG_SNAPSHOT_THE_STATE_KEEP_THE_TAIL.md b/docs/explorations/0254_[_]_COMPACT_THE_CHANGE_LOG_SNAPSHOT_THE_STATE_KEEP_THE_TAIL.md index 2fb1a9b45..71ef86d8f 100644 --- a/docs/explorations/0254_[_]_COMPACT_THE_CHANGE_LOG_SNAPSHOT_THE_STATE_KEEP_THE_TAIL.md +++ b/docs/explorations/0254_[_]_COMPACT_THE_CHANGE_LOG_SNAPSHOT_THE_STATE_KEEP_THE_TAIL.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Compact The Change Log: The Durable Cold‑Open Fix — Snapshot The State, Keep The Tail ## Problem Statement diff --git a/docs/explorations/0257_[_]_CLOSING_THE_LAST_MILE_ALIGNING_THE_CODE_WITH_THE_ETHOS.md b/docs/explorations/0257_[_]_CLOSING_THE_LAST_MILE_ALIGNING_THE_CODE_WITH_THE_ETHOS.md index 452f51d55..2950c3d0d 100644 --- a/docs/explorations/0257_[_]_CLOSING_THE_LAST_MILE_ALIGNING_THE_CODE_WITH_THE_ETHOS.md +++ b/docs/explorations/0257_[_]_CLOSING_THE_LAST_MILE_ALIGNING_THE_CODE_WITH_THE_ETHOS.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Closing the Last Mile: Aligning the Code With the Ethos > A deep introspection into the gaps between what xNet's essays *say* it is diff --git a/docs/explorations/0258_[_]_MULTI_HOME_SYNC_FEDERATED_HUBS_PEERS_AND_THE_REPLICATION_MANIFEST.md b/docs/explorations/0258_[_]_MULTI_HOME_SYNC_FEDERATED_HUBS_PEERS_AND_THE_REPLICATION_MANIFEST.md index 6d46ea2d9..8249e228f 100644 --- a/docs/explorations/0258_[_]_MULTI_HOME_SYNC_FEDERATED_HUBS_PEERS_AND_THE_REPLICATION_MANIFEST.md +++ b/docs/explorations/0258_[_]_MULTI_HOME_SYNC_FEDERATED_HUBS_PEERS_AND_THE_REPLICATION_MANIFEST.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Multi-Home Sync — Federated Hubs, Community Hubs, Peers, and the Replication Manifest ## Problem Statement diff --git a/docs/explorations/0262_[_]_MAIN_THREAD_SQLITE_AND_THE_MULTIPLE_READER_QUESTION.md b/docs/explorations/0262_[_]_MAIN_THREAD_SQLITE_AND_THE_MULTIPLE_READER_QUESTION.md index e9af57795..0cd3293f3 100644 --- a/docs/explorations/0262_[_]_MAIN_THREAD_SQLITE_AND_THE_MULTIPLE_READER_QUESTION.md +++ b/docs/explorations/0262_[_]_MAIN_THREAD_SQLITE_AND_THE_MULTIPLE_READER_QUESTION.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Main-Thread SQLite And The Multiple-Reader Question ## Problem Statement diff --git a/docs/explorations/0268_[_]_XNET_FOR_ELECTRONIC_MEDICAL_RECORDS_SECURITY_AND_LEGALITY.md b/docs/explorations/0268_[_]_XNET_FOR_ELECTRONIC_MEDICAL_RECORDS_SECURITY_AND_LEGALITY.md index 968fccf4d..81b438fd6 100644 --- a/docs/explorations/0268_[_]_XNET_FOR_ELECTRONIC_MEDICAL_RECORDS_SECURITY_AND_LEGALITY.md +++ b/docs/explorations/0268_[_]_XNET_FOR_ELECTRONIC_MEDICAL_RECORDS_SECURITY_AND_LEGALITY.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # xNet For Electronic Medical Records — Security And Legality ## Problem Statement diff --git a/docs/explorations/0270_[_]_DESKTOP_FILESYSTEM_AS_A_GOVERNED_CAPABILITY.md b/docs/explorations/0270_[_]_DESKTOP_FILESYSTEM_AS_A_GOVERNED_CAPABILITY.md index 55954acf5..cdbeee1f8 100644 --- a/docs/explorations/0270_[_]_DESKTOP_FILESYSTEM_AS_A_GOVERNED_CAPABILITY.md +++ b/docs/explorations/0270_[_]_DESKTOP_FILESYSTEM_AS_A_GOVERNED_CAPABILITY.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # 0270 - Desktop Filesystem As A Governed Capability ## Problem Statement diff --git a/docs/explorations/0274_[_]_TABLEPLUS_GRADE_DATA_TAB.md b/docs/explorations/0274_[_]_TABLEPLUS_GRADE_DATA_TAB.md index ac62fb73e..4b64137aa 100644 --- a/docs/explorations/0274_[_]_TABLEPLUS_GRADE_DATA_TAB.md +++ b/docs/explorations/0274_[_]_TABLEPLUS_GRADE_DATA_TAB.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # TablePlus-Grade Data Tab: Bringing Desktop DB-GUI Ergonomics To Devtools And Databases ## Problem Statement diff --git a/docs/explorations/0279_[_]_BOTLESS_MEETING_TRANSCRIPTION_AND_AI_NOTES.md b/docs/explorations/0279_[_]_BOTLESS_MEETING_TRANSCRIPTION_AND_AI_NOTES.md index cfd6c9b6f..096399381 100644 --- a/docs/explorations/0279_[_]_BOTLESS_MEETING_TRANSCRIPTION_AND_AI_NOTES.md +++ b/docs/explorations/0279_[_]_BOTLESS_MEETING_TRANSCRIPTION_AND_AI_NOTES.md @@ -1,3 +1,7 @@ +--- +review: 2026-11-06 +--- + # Botless Meeting Transcription And AI Notes (Granola / Notion Style) ## Problem Statement diff --git a/docs/explorations/0424_[-]_ASTERISK_13_AND_THE_EPISTEMICS_OF_A_WORKSPACE.md b/docs/explorations/0424_[-]_ASTERISK_13_AND_THE_EPISTEMICS_OF_A_WORKSPACE.md index d792fb369..71ec9795b 100644 --- a/docs/explorations/0424_[-]_ASTERISK_13_AND_THE_EPISTEMICS_OF_A_WORKSPACE.md +++ b/docs/explorations/0424_[-]_ASTERISK_13_AND_THE_EPISTEMICS_OF_A_WORKSPACE.md @@ -2,7 +2,7 @@ title: Asterisk 13 And The Epistemics Of A Workspace status: draft last_updated: 2026-08-01 -review: 2026-10-01 # short on purpose: the one code finding (F1) is small and rots into a shipped bug if it waits +review: 2026-11-06 decider: Chris Smothers door: two-way tags: [research, ai, epistemics, retrieval, process, strategy] diff --git a/docs/explorations/0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md b/docs/explorations/0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md new file mode 100644 index 000000000..4f350d5ef --- /dev/null +++ b/docs/explorations/0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md @@ -0,0 +1,1163 @@ +--- +title: A personal library for learning and sharing, safe enough to use every day +status: draft +last_updated: 2026-10-02 +review: 2026-11-10 +decider: Chris Smothers +door: two-way +tags: + [daily-driver, personal-library, social-import, knowledge-graph, durability, desktop, publishing] +--- + +# A personal library for learning and sharing, safe enough to use every day + +> [!TIP] +> Build Chris's library from the garden, website, Twitter/X and Instagram exports, YouTube playlists, and GitHub stars. Enrich every imported link with useful metadata and a local thumbnail, and obtain video transcripts wherever possible. Preserve how those things connect, then make them easy to find, annotate, and turn into useful guides. Start with a packaged Mac app whose data survives development and updates. + +## Implementation status + +A development preview now runs in Electron. It includes native recovery copies, encrypted native export/restore, resumable retained-source imports, a persistent enrichment queue, local thumbnails and retrieved captions, offline Library search, and editable URL-plus-note capture. [Try the preview](../reference/personal-library.md) and review the [storage coverage](../reference/desktop-storage.md) before relying on it. + +This is partially implemented. A Developer ID signing certificate is unavailable in this environment, the installed signed-upgrade test is open, and complete settings/key coverage and off-device retention are not verified. Selected real exports are now imported into the development profile. This remains separate from the signed daily-use app and complete enrichment coverage. Successful live managed-helper installation, local transcription, full collection/creator navigation, complete website and overlapping-export reconciliation, guides, publishing, and the human trial remain work. The checklists below distinguish these gaps from the narrower proofs already obtained. + +## The job to earn + +Chris already collects ideas, builds things, and shares resources. Much of that work has accumulated in social bookmarks, likes, saved videos, playlists, and starred repositories. The first library should bring that existing collection home, alongside the garden and website. A paper saved years ago should help answer a question next month. Notes from several sources should become a guide for a friend, a public page, or an optional resource for a coaching client. + +The starting loop is **import → enrich → connect → rediscover**. The daily loop is **save → find → compose → share**. Each step must be useful on its own. Import should preserve the evidence and organization already present. Enrichment should recover the titles, descriptions, images, and spoken content that make a saved link useful. Finding should work with a half-remembered phrase, a creator, or a playlist. Writing should start with notes already at hand. Sharing should give someone a readable link they can open without installing xNet. + +There is a prerequisite: Chris must be able to trust the app while changing its code. If using xNet means rebuilding a checkout, watching migrations, or wondering whether an update will erase notes, the library will stay empty. The first milestone is a safe daily desktop installation with a simple update and recovery path. + +This exploration records the direction chosen in conversation, the durability concern, and the decision to start with enriched social exports and GitHub stars. Metadata and thumbnails for every link, plus transcripts where obtainable, are core import work. It proposes work; it does not certify the current app as safe for irreplaceable data. + +| Choice | Direction | +| ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | +| First personal value | Learning and sharing | +| Starting material | Garden and website content; Twitter/X likes and bookmarks; YouTube playlists; Instagram saves, collections, and likes; GitHub stars | +| Existing collection | Import the available corpus in resumable batches; use a small sample to prove fidelity, not to cap the library | +| Enrichment | Automatic metadata and local thumbnails for every link; caption retrieval and accessible-media transcription for videos, with measured coverage | +| Primary authoring surface | Packaged Mac desktop app | +| First reader experience | A public page, with no account or installation | +| AI's role | Optional help over deliberately selected sources | +| First trust requirement | Keep real data safe while the app and its data model evolve | +| Publication destination | Proposed default: `crs.garden/guides/`; configurable, not a confirmed hosting decision | + +The review date leaves roughly six weeks for initial work and a four-week usage trial. It is a date to reconsider this direction, not a promised delivery date. Chris decides whether to continue, narrow, or stop. Choosing this workflow is reversible. A new persistent format or public compatibility promise needs a separate ADR under the repository's decision policy before implementation. + +```mermaid +flowchart LR + Archives[Social archives and GitHub stars] --> Import[Import with provenance] + Garden[Garden and website] --> Import + Import --> Library + Library -. background enrichment .-> Enrich[Metadata, thumbnails, and available transcripts] + Enrich --> Library + Source[Link, paper, or video] --> Capture[Save URL and a thought] + Capture --> Library[Private personal library] + Library --> Find[Find and revisit] + Find --> Draft[Compose a guide] + Draft --> Preview[Review a public snapshot] + Preview --> Reader[Friend or client opens a link] + Find --> Library + Draft --> Library + Library -. optional selected sources .-> Helper[Research helper] + Helper -. cited suggestions .-> Draft + Library --> Recovery[Verified recovery copies] +``` + +## Why this fits Chris's work + +[crs.land](https://crs.land) connects coaching, software, simulations, body-related resources, food forests, housing, and other experiments. The common activity is making something interesting easier for another person to approach. xNet can support the collection and writing behind those projects without absorbing each project's interface. + +[crs.garden](https://crs.garden) already pairs links with personal commentary. Its software and body-related reading shows the shape of a useful source note: a source, a reason to care, and enough context to return later. The garden's existing Bluesky collection flow should keep working. A guide can draw from that material without making xNet the source of every garden post. + +[crs.coach](https://crs.coach) describes a practice built around presence and working with the person in front of Chris. That favors a small resource offered at the right moment. It gives little reason to begin with a client dashboard, a habit score, or a prescribed sequence of worksheets. The [nervous-system resource site](https://crs48.github.io/nervous-system-healing/) is another useful precedent: resources have context, and personal experience is distinguished from evidence. A guide should preserve that distinction. This proposal makes no clinical claims about the material. + +The adjacent repositories also suggest a clear division of work: + +| Existing project | Keep its job | What xNet contributes | +| ----------------------------------------------------------------------- | -------------------------------------------------- | ------------------------------------------------------- | +| `digitalgarden` / [crs.garden](https://crs.garden) | Public browsing and the Bluesky-to-garden pipeline | Draft and export longer guides | +| `whole-body-cookbook` | A purpose-built interactive body atlas | Notes and source collections that can link to the atlas | +| `feedme` | Creator support and its own funding model | Links and context where useful; no new payment system | +| Simulations and small websites linked from [crs.land](https://crs.land) | Their own visual and interactive experiences | Research notes and companion reading | +| Coaching | A human relationship and live practice | Optional, carefully chosen public resources | + +These are observations from public pages and selected local project documents, not evidence that clients want a new app. Friends and clients are initially readers. Test another person's authoring needs separately before treating this as a product for coaches. + +## The seed corpus is already here + +Chris identified `.exports/` as the local archive directory. A read-only inspection on 2026-09-29 found the three requested social archives below. The directory is Git-ignored. Only archive metadata, selected saved/liked records, playlist CSVs, and structural field shapes were inspected. Nothing was imported, extracted onto disk, sent for enrichment, or changed. Private messages and account-security records were not opened. + +| Source | Observed local material | What the first import must preserve | +| ------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | +| Garden and website | Existing public pages and the local `digitalgarden` project | Chris's commentary, original URLs, tags, dates where available, and links to source resources | +| `.exports/twitter.zip` | 106.7 MiB; 10,028 records in `like.js`; no bookmark-named archive member detected | Tweet IDs, exported text and links, and the fact these are likes. Bookmark coverage remains unverified | +| `.exports/youtube.zip` | 5.0 MiB; 33 playlist catalog rows; 32 playlist-video CSVs containing 11,259 membership rows and 9,273 distinct video IDs | Playlist identities and names, every membership, row order as exported, and timestamps with their original meaning | +| `.exports/instagram.zip` | 423.1 MiB; 7,230 saved-post records, 8,827 liked-post records, 15 saved-collection records; also 74 saved-music and 25 liked-comment records | Separate saves and likes, named collections, nested item relationships, source URLs, captions, and available timestamps | +| GitHub stars | Later captured in `.exports/github-stars.json`: 1,514 repositories with native star times | Repository identity and URL, owner, description, available topics/language, and the star relationship with its timestamp when supplied | + +These counts describe the source files. They do not establish how many unique resources will import successfully. For example, one YouTube video can belong to several playlists. The difference between 33 catalog entries and 32 membership files needs a reconciliation report; it must not be guessed away as either data loss or empty playlists. + +The YouTube membership CSVs contain video IDs and playlist-video timestamps, without video titles or descriptions. The Twitter likes contain `tweetId`, `fullText`, and `expandedUrl`, without a like timestamp. Preserve those limits. Import time is not save time, and a video's presence in a playlist is not evidence that Chris watched it. + +Other archives are present for TikTok, Reddit, and AI chat services. Keep them intact and available for later selected imports through existing adapters. The first acceptance run covers the garden, website, Twitter/X, YouTube, Instagram, and GitHub. It does not silently ingest every category in every archive. + +> [!IMPORTANT] +> The full saved-resource corpus is now core scope. The earlier suggestion of about 25 hand-picked resources becomes a small validation sample from the real archives. It is not a substitute for importing the rest. + +## What the repository says, and what the code says + +The research inventoried 525 numbered exploration documents, then followed the roadmap, graph queries, relevant explorations, and their implementation paths. It did not read every document line by line. Code observations below refer to commit `fbedf30e7`, inspected on 2026-09-29. Filename checkboxes were treated as leads, not proof of a finished experience. This pass changed documentation only. The desktop recovery and update paths were inspected, not exercised end to end. + +The [roadmap](../ROADMAP.md) already makes founder daily use the near-term test. Its emphasis is the agent-assisted workspace. This proposal gives that platform a concrete daily job and puts local recovery before reliance on it. It fits the ownership commitments in the [charter](../CHARTER.md) and the quiet product experience in [VIBE](../VIBE.md). The website's [roadmap data](../../site/src/data/roadmap.ts) and the repository roadmap should be reconciled after this direction earns acceptance through use. + +| Area | Status from inspection | Consequence for this plan | +| -------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | +| Pages, folders, tags | ✅ Existing [Page schema](../../packages/data/src/schema/schemas/page.ts) and organization work | Start with ordinary Pages; avoid a new knowledge model | +| Capture | 🚧 The [quick-capture tray](../../packages/workbench/src/views/tray.tsx) is oriented around tasks and page navigation | Add a small, durable resource capture flow | +| URL handling | ✅ [External-reference helpers](../../packages/data/src/external-references.ts) exist | Reuse URL handling; keep personal commentary in a Page | +| Search | 🚧 [Page search](../../packages/workbench/src/hooks/usePageSearchSurface.ts) reads document bodies; [store FTS extraction](../../packages/data/src/store/indexing/full-text.ts) uses properties | Verify the actual capture-to-search path; do not assume all retrieval sees note bodies | +| AI | 🚧 [Graph retriever](../../packages/workbench/src/views/ai-graph-retriever.ts) and agent tools exist | Start with explicit selected Page bodies and visible citations | +| Static publication | 🚧 [Renderer and site builder](../../packages/publish/src/site.ts) exist | Finish the author-to-export flow and its privacy boundary | +| Portable data | 🚧 [Bundle format and ports](../../packages/data/src/portability/types.ts) cover signed changes, optional blobs, and optional Yjs documents | Existing export code needs complete desktop wiring and a restore proof | +| SQLite snapshots | ✅ [`xnet data snapshot`](../../packages/cli/src/commands/data.ts) uses `VACUUM INTO` | Reuse the snapshot capability inside a complete backup flow | +| Desktop updates | 🚧 [Updater](../../apps/electron/src/main/updater.ts) downloads and installs releases | Add a verified data checkpoint and coordinated shutdown before installation | +| Development profiles | 🚧 [Worktree scoping](../../apps/electron/scripts/dev-scope.mjs) exists; the main checkout keeps `default` | Protect daily data from every development launch, including the main checkout | +| Startup recovery | 🛑 [Data-service initialization](../../apps/electron/src/data-process/data-service.ts) deletes an unversioned database, or deletes after an inspection exception | Remove this behavior before moving irreplaceable notes into the app | + +> [!WARNING] +> The startup deletion path is a concrete blocker. An old format, an unreadable database, and a new empty workspace must produce different outcomes. None is permission to delete the user's files. + +The social imports have a substantial foundation, but the actual archive shapes reveal gaps: + +| Existing code | Evidence and remaining work | +| ---------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| [Importer registry](../../packages/social/src/importers/registry.ts) | Twitter/X, YouTube, and Instagram adapters are registered. GitHub stars are absent | +| [Twitter adapter](../../packages/social/src/importers/x.ts) | Maps likes and several other archive categories; no bookmark bucket is defined. Add bookmark support against a supplied format rather than relabeling likes | +| [YouTube adapter](../../packages/social/src/importers/youtube.ts) | Maps playlist catalogs and memberships. Reconcile the real files, preserve repeated memberships, and make sparse video records useful | +| [Instagram adapter](../../packages/social/src/importers/instagram.ts) | Creates a collection per saved file and reads shallow labels. The real collection records contain nested item-shaped `dict` arrays; these need proper named-collection and membership mapping. The real liked-comments file is wrapped in `likes_comment_likes`, while the current mapper expects an array | +| [Social schemas](../../packages/social/src/schemas/index.ts) | Already model content, actors, interactions, collections, membership, import runs, and source records. Reuse them | +| [IDs](../../packages/social/src/import/ids.ts) and [commit policy](../../packages/social/src/import/policy.ts) | Deterministic IDs and batched commits exist. Some IDs depend on paths, positions, or interaction kinds; cross-export reconciliation still needs proof. Default source-record mode is `sidecar`; verify actual retention, not only its count | +| [Import jobs](../../packages/social/src/import/jobs.ts) and [desktop import IPC](../../apps/electron/src/main/social-import-ipc.ts) | Progress, cancellation, and checkpoint records exist. Prove restart/resume and complete backup coverage with these archives | +| [Graph lenses](../../packages/social/src/lenses/graph-lenses.ts) and [canvas projection](../../packages/social/src/projection/canvas.ts) | Saved-content-by-creator and bounded graph projection primitives exist. Wire useful Library views rather than building a separate graph store | + +The GitHub addition can reuse [ExternalItem](../../packages/data/src/schema/schemas/external-item.ts) for a repository's stable external identity and payload. The social vocabulary currently lacks a GitHub platform entry. Choose one canonical repository representation and connect the star activity to it; avoid creating disconnected repository copies in two schema families. + +This extends [0152: social importer](./0152_%5Bx%5D_ACTUAL_SOCIAL_GRAPH_IMPORTER.md), [0153: social workspace](./0153_%5Bx%5D_SOCIAL_DATA_WORKSPACE_UI.md), and [0419: social graph atlas](./0419_%5B-%5D_SOCIAL_GRAPH_ATLAS.md). Their primitives are useful; their filename status does not establish fidelity for these specific archives. + +
+Durability findings and the files behind them + +The desktop opens two database paths. [Main-process setup](../../apps/electron/src/main/index.ts) sends `xnet-data/data.db` to the data utility process. [Storage IPC](../../apps/electron/src/main/ipc.ts) also opens `xnet-data/xnet.db` through a [blob storage adapter](../../apps/electron/src/main/storage.ts). A backup of one SQLite file is not yet proof of a complete desktop backup. Trace all live writers before deciding which paths are authoritative. + +In `data-service.ts`, an existing database is opened to inspect its schema version. Version zero causes deletion of the database, WAL, and SHM files. The surrounding catch also attempts deletion. In the [Electron SQLite adapter](../../packages/sqlite/src/adapters/electron.ts), `getSchemaVersion()` maps any query error to zero. This compounds the problem: an inspection failure can look like an obsolete format. + +The adapter sets `synchronous = NORMAL`. Its `applySchema()` returns without applying DDL when the existing version is greater than or equal to the requested version. That is not a refusal to write a database from a newer app. The factory applies the current DDL; the presence of [migration SQL helpers](../../packages/sqlite/src/schema.ts) alone does not establish a tested desktop upgrade chain. + +The [updater](../../apps/electron/src/main/updater.ts) sets `autoDownload = false` and `autoInstallOnAppQuit = true`. Its install paths call `quitAndInstall()`. [Main-process shutdown](../../apps/electron/src/main/index.ts) has asynchronous cleanup in a `before-quit` listener, but that listener does not prevent quit and explicitly resume it after an acknowledged flush. [Data-process shutdown](../../apps/electron/src/main/data-process-manager.ts) requests shutdown and can then kill the process. These paths need a tested barrier that waits for the renderer's final edits, persistence, and backup completion. + +The [identity seed](../../apps/electron/src/main/identity-seed.ts) is now random and stored per profile, using platform encryption when available. Invalid stored seeds fail loudly. Older exploration 0335's deterministic-key finding is therefore stale. A machine-bound encrypted seed file still needs a separate recovery design for restoring onto a new Mac. + +The [portable bundle API](../../packages/data/src/portability/types.ts) makes blob and Yjs ports optional. A valid signed bundle can contain no document bodies when its caller omitted that port. The [store Yjs port](../../packages/data/src/portability/store-yjs-port.ts) also needs comparison with the desktop's actual document/update persistence path. Manifest verification and complete workspace recovery are different checks. + +The [release workflow](../../.github/workflows/electron-release.yml) already builds desktop artifacts, checks native packaging, and has a nightly packaging run. Nightly runs do not publish updates. The workflow supports Developer ID signing and notarization when configured, and a self-signed fallback. This research did not inspect release secrets or prove that two installed releases update smoothly on Chris's Mac. + +
+ +
+Explorations to reuse, and scope to defer + +| Existing exploration | Use here | +| -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- | +| [0105: what to work on next](./0105_%5B_%5D_WHAT_TO_WORK_ON_NEXT_AFTER_OPEN_SOURCE_LAUNCH.md) | The daily-use question has recurred; settle it with a small real trial | +| [0112: universal clipper](./0112_%5B_%5D_UNIVERSAL_CLIPPER_AND_AI_KNOWLEDGE_GRAPH_INGESTION.md) | Reuse capture intent; include metadata, thumbnails, and obtainable transcripts; defer automatic entity inference | +| [0169: folders and tags](./0169_%5Bx%5D_CONTENT_ORGANIZATION_FOLDERS_TAGS_AND_CHANNELS.md) | Reuse organization already present | +| [0179: spaces and sharing](./0179_%5B_%5D_SPACES_GROUPS_AND_UNIFIED_SHARING.md) | Revisit for private collaboration after public reader value is proven | +| [0180: experiment journal](./0180_%5B_%5D_EXPERIMENT_JOURNAL_AND_HABIT_TRACKER.md) | Keep available; it is not the selected daily job | +| [0344: portability](./0344_%5Bx%5D_FIRST_CLASS_DATA_EXPORT_IMPORT_AND_PORTABLE_BUNDLES.md) | Use the export/import primitives, then prove desktop completeness | +| [0362: publishing](./0362_%5B_%5D_PUBLISHING_ON_XNET_GHOST_SUBSTACK_AND_THE_OWNED_AUDIENCE.md) | Finish one small route to a public guide | +| [0379: knowledge base](./0379_%5B_%5D_A_KNOWLEDGE_BASE_ON_XNET_PRIMITIVES_DISTILLATION_BURSTS_AND_THE_GOVERNED_CORPUS.md) and [0391: daily AI interface](./0391_%5Bx%5D_XNET_AS_THE_DAILY_DRIVER_AI_INTERFACE.md) | Reuse retrieval work; check later implementation before repeating old gaps | +| [0406: shared shell](./0406_%5Bx%5D_ONE_SHELL_TWO_SURFACES_ENDING_THE_DESKTOP_WEB_UI_FORK.md) | Put the Library experience in the shared workbench | +| [0413: worktree desktop development](./0413_%5B-%5D_PLURAL_ELECTRON_WORKTREE_SCOPED_DESKTOP_DEV.md) | Extend profile isolation to a protected daily installation | +| [0430: risk-adjusted engineering](./0430_%5B-%5D_RISK_ADJUSTED_ENGINEERING_READING_ASTERISK_14.md) | Give upgrade checks a consumer, a pass condition, and proof they can fail | +| [0455: plugin composition](./0455_%5B-%5D_CORDIS_LESSONS_FOR_XNET_PLUGIN_COMPOSITION.md), [0456: agent door](./0456_%5B-%5D_ENTRY_VECTOR_THE_AGENT_DOOR_FIRST.md), and [0457: site](./0457_%5B-%5D_AGENT_FIRST_SITE_REARCHITECTURE.md) | Preserve completed agent work; use this library as a concrete human workflow to validate | + +
+ +## Choose a desktop home for the library + +| Option | Useful property | Cost for this use case | Decision | +| ----------------------------------------------- | ----------------------------------------------------- | --------------------------------------------------------------------------- | -------------------------------------- | +| Packaged Electron app with protected daily data | Local files, OS integration, existing release updater | Must finish recovery and prove upgrades | **Primary path** | +| Run the library from a development checkout | Fast iteration with hot reload | Branch changes and experiments become data risks; too much daily ceremony | Use for development with separate data | +| Browser app / PWA | Easy access and web delivery | Browser storage policy and recovery add uncertainty for the only local copy | Keep as a later authoring option | +| Markdown files as the primary store | Easy to inspect and copy | Would require a new editing/sync contract and abandon useful existing work | Offer readable export as an exit path | + +[MDN's persistent-storage API](https://developer.mozilla.org/en-US/docs/Web/API/StorageManager/persist) lets a browser request protection from automatic eviction; the browser may refuse. That is useful for a future web authoring surface, but it does not replace a backup. A public guide can be an ordinary website regardless of where it was authored. + +For Chris, the normal experience should be: open xNet from Applications, save something, close it when done. A release can arrive in the background. Installing it should require at most a short restart, with notes and identity intact. No terminal, package install, checkout selection, or manual data conversion should be part of using the library. + +## A durability promise with clear limits + +“Local-first” describes where work happens. Trust also requires a clear answer to what survives a crash, a bad release, a mistaken deletion, and the loss of the Mac. + +The following is the proposed contract to implement and test. It is not a guarantee the current build already makes. + +| Failure | Intended protection | Honest boundary | +| ------------------------------------- | ------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------- | +| App crash or forced quit | Every edit acknowledged as **Saved on this Mac** has reached durable storage, including its document content | Text still marked **Saving…** may be lost | +| Power loss | Durable transactions and flushed data before the saved acknowledgement | Subject to filesystem and hardware guarantees; prove the configured path and test interruption | +| Bad app update or migration | A complete verified checkpoint of the old workspace and a retained compatible app version | Recovery must preserve any edits made after the update too | +| Accidental deletion or bad agent edit | Retained recovery points, plus existing history where applicable | History and sync can carry mistakes; neither replaces independent copies | +| Failed disk or lost Mac | A completed encrypted backup on another device or storage service, with usable recovery material | A second folder on the same disk does not cover this failure | +| xNet development stops | Portable bundle plus a readable Markdown/assets export | A readable export can omit protocol history; say exactly what each export contains | + +The saved acknowledgement needs a path from the editor through Yjs persistence and the data process to the completed transaction. Updating React state or sending an IPC message is too early. If writing fails, retain the unsaved text, show the failure, and allow a copy/export. + +[SQLite's WAL documentation](https://sqlite.org/wal.html) distinguishes integrity from power-loss durability: `NORMAL` omits a sync on each commit, while `FULL` syncs the WAL at commit. Use `FULL` as the starting point for the daily desktop store and measure its cost. Batch edits where appropriate, keeping **Saving…** visible until the batch commits. Do not promise that every keystroke has reached disk before it has. + +### Three useful forms of recovery + +| Form | Main job | Required proof | +| ------------------------------------------------- | ------------------------------------------------------- | -------------------------------------------------------------------------------- | +| Native workspace checkpoint | Recover quickly from a bad desktop update | Reopen with the compatible app, with notes, blobs, and identity intact | +| Full `.xnetpack` plus encrypted recovery material | Move or recover data independently of one SQLite layout | Restore into an empty isolated workspace with all expected content and ownership | +| Markdown, source URLs, and assets | Read and reuse the collection without xNet | Open outside xNet; list anything omitted | + +Reuse the existing bundle and snapshot machinery. The missing product is automatic creation, a complete content inventory, verification, retention, and a usable Restore action. + +A native checkpoint must cover the workspace consistently: both live database stores where applicable, document states and pending updates, referenced blobs, custom schema definitions, and the identity/encryption material needed to open them. Include app and storage versions in its manifest. Distinguish required data from rebuildable search indexes, caches, and window preferences. Account for data outside `xnet-data` before claiming the inventory is complete. + +At checkpoint time, pause new writes, drain every writer, and capture a consistent set. SQLite's [Online Backup API](https://www.sqlite.org/backup.html) can make a consistent database copy; the existing `VACUUM INTO` path is another available primitive. Copying only a live main database file can omit committed WAL content. Keep the live database on a local filesystem. Send completed immutable archives to a backup destination after they are closed and verified. + +For a full portable backup, require the desktop blob and Yjs ports. Compare its manifest against the frozen workspace inventory. Missing bodies or attachments must fail completeness validation even if the bundle's signatures are valid. Verify checksums and restore into a temporary workspace with network access disabled. Delete no previous good backup until the replacement is complete. + +Platform-encrypted keys are useful on the current Mac. Recovery on a new Mac must also work without the old Keychain. Design an encrypted recovery kit that covers the actual signing and content keys, with a recovery secret Chris can store separately. Never put plaintext keys in an ordinary cloud-synced export. A wrong secret, missing key, or unreadable seed must stop recovery without silently making a new identity. + +### Small, visible backup policy + +Proposed initial defaults: make a local checkpoint every 15 minutes while data has changed, on the next launch when overdue, and before each update or storage migration. These are targets for a healthy running app, not a claimed loss bound after failed backups. Keep recent points for a day, daily points for a week, and weekly points for a month. Pin the last good pre-migration checkpoint and its app-version reference until the replacement has been verified and retained long enough to recover. + +Bound disk use, deduplicate immutable blobs where practical, and report lack of space. Pruning must never erase the last verified recovery point to make room for an unverified one. Start with full recoverable points; incremental chains add another recovery dependency and can wait. + +Let Chris choose an off-device destination once. Show local recovery and off-device backup as separate facts: **Recovery copy: 3 minutes ago** and **External backup: yesterday**. Without a completed external copy, say **Backed up on this Mac only**. Do not require a hosted xNet account for local use. An existing backup service can carry closed archives; the app must distinguish writing an archive locally from confirmed off-device protection. + +## Updates that preserve the work + +The app binary and the workspace have separate lifetimes. An update can replace the app without moving or replacing the user's home for data. Data conversion happens only when the storage format actually needs it. + +```mermaid +sequenceDiagram + actor Chris + participant App as Daily app + participant Data as Workspace writers + participant Backup as Recovery manager + participant Next as Updated app + Chris->>App: Restart to update + App->>Data: Pause writes and flush all edits + Data-->>App: Durable flush acknowledged + App->>Backup: Create and verify complete checkpoint + Backup-->>App: Recovery point ready + App->>App: Hand control to installer + Next->>Backup: Inspect versions before opening writable stores + alt Format unchanged + Next->>Data: Open existing workspace + else Supported migration + Next->>Backup: Migrate a candidate copy and validate + Backup-->>Next: Candidate passed + Next->>Data: Atomically select candidate generation + else Unknown format or verification failure + Next-->>Chris: Recovery view; original files preserved + end +``` + +The updater should download while Chris works and offer a calm restart action. Every install path, including install-on-quit, must use the same safety barrier. If a fresh checkpoint cannot complete, postpone installation and leave the current app usable. Do not reinterpret a failed update check as “up to date.” Use explicit outcomes for offline, unavailable, failed, available, and current. + +The shutdown barrier belongs before the installer closes the renderer. Awaiting an async event listener alone is insufficient. The app must hold the quit event, drain writes, verify the checkpoint, and then resume the one approved installation or quit. Test an edit made immediately before that action. + +On startup, first probe compatibility without migrating or opening the store for writes. A supported old workspace goes through an explicit migration chain on a candidate copy. Check integrity, content counts, attachment hashes, representative document rendering, and identity continuity. Switch a small active-generation pointer only after the candidate passes. Make that switch crash-safe and resumable. Retain the untouched source generation. An unversioned workspace enters an import/recovery path; an unreadable one is preserved for diagnosis. + +Do not run the new app's normal writable database factory just to inspect compatibility. The version check must precede DDL, write pragmas, and any cleanup that changes the source. + +### Format changes should be uncommon and explicit + +| Versioned thing | What it means | Policy | +| -------------------------- | ------------------------------------- | ------------------------------------------------------------------------------------- | +| App release | UI and behavior | Most releases should open the same data unchanged | +| SQLite storage layout | Tables, columns, indexes | Ordered, transactional migrations on a candidate copy | +| Node schema | Meaning of stored fields | Additive changes first; explicit conversion or read adapters for old data | +| Yjs/editor document format | Rich-text content and embedded blocks | Preserve unknown content; never discard an old block while loading | +| Bundle and sync protocol | Data exchange and portable recovery | Follow [compatibility policy](../COMPATIBILITY.md); reject unsupported writes clearly | + +Preserve original signed records. Schema evolution can add a conversion or a new view of old content without rewriting history as if it always had the new meaning. Unknown schema data should remain recoverable, even if the current UI can only show a read-only fallback. Avoid a new source-note schema in the first place: ordinary Pages lower the migration burden for this workflow. + +### Rollback must not erase newer notes + +Rolling back the binary is safe only when the older app can read the current workspace. Otherwise the recovery unit is the old app plus its compatible checkpoint. The failed candidate and any later edits remain preserved. + +If Chris has added notes since updating, never silently point the old app at yesterday's data. Show the recovery point's time and the later work at risk. Offer a supported replay/export of that work, or keep it in a separate recoverable generation until a fix is available. Restoring a checkpoint is not an inverse migration. Keep network sync disabled during validation and recovery; reconnect deliberately after resolving the current state so stale restored data does not surprise other peers. + +The recovery view must open before normal workspace startup, so a broken migration cannot hide the Restore action. If the new app binary itself cannot launch, provide a way to reinstall the last compatible signed build without a terminal. Reinstalling that build must still preserve all workspace generations and leave any data rollback explicit. + +## Using and developing xNet at the same time + +```mermaid +flowchart TB + subgraph Everyday[Everyday use] + Installed[Packaged xNet app] --> Daily[Protected daily workspace] + Daily --> Checkpoint[Verified recovery copies] + end + subgraph Development[Development] + Checkout[Checkout or worktree with hot reload] --> Dev[Separate disposable profile] + Checkpoint -. explicit isolated restore .-> Sandbox[Recovery test sandbox] + Candidate[Packaged candidate] --> Sandbox + end + Candidate --> Gate[Upgrade and recovery checks] + Gate --> Release[Promote tested artifact] + Release --> Installed +``` + +Use one packaged daily app with a stable data path independent of repository location, branch, and build version. Every development launch gets separate storage, including launches from the main checkout. Worktree scoping already handles much of this; close the default-profile gap and refuse accidental development access to the protected workspace. Verify actual resolved paths rather than assuming a profile name proves isolation. + +Adopting an existing profile needs a one-time, backed-up transition that preserves its identity and data. Do not change a path and quietly show an empty library. Make the selected workspace clear, and never merge profiles merely because their names look similar. + +Develop UI changes with the existing Electron hot-reload and cached-build workflow. Build a packaged candidate only when validating a release. Use the existing release pipeline to produce daily-app updates, then promote a tested artifact without asking Chris to rebuild it locally. If faster access is useful, offer an explicit early-release channel using the same recovery checks. There is no need for a second installer system. + +A test copy of real data must start with network sync, background agents, and public serving disabled before any document opens. Keep identity keys inside the isolated recovery test when continuity must be verified; do not let a duplicated identity act as a second live client. Ordinary development should use synthetic data and a separate identity. + +Stable signing is part of convenience. Electron documents that [code signing](https://github.com/electron/electron/blob/main/docs/tutorial/code-signing.md) matters for macOS updating and Keychain behavior. Test the installed release-to-release path, including access to existing encrypted identity material. The repository currently declares electron-updater 6.x and electron-builder 25.x; current [builder documentation](https://www.electron.build/docs/features/auto-update/) describes newer APIs too. Match implementation to the locked versions rather than copying a current example blindly. + +## The library: useful in the first minute + +After the recovery milestone, import the saved-resource corpus from `.exports/`, the garden and website, and a GitHub star snapshot. First validate a small sample that includes each platform and its awkward cases, then process the full selected categories in bounded batches. Create two guide drafts from rediscovered sources. Success includes making past collecting useful immediately. + +The Library entry point offers **Inbox**, **Resources**, **Collections**, and **Guides**. Use existing folders and tags for Pages, and project imported social content and collections into the same Library. Do not turn every like into a blank Page. Inbox means “saved for later”; it is not a queue that demands completion. An imported resource can remain a source record until Chris wants to add a note. + +### Import the history without flattening it + +Keep the original archives unchanged. Record an archive hash, account identity, export date when supplied, parser version, selected categories, and the source file and record for each imported fact. Preserve raw selected records and nested fields so better parsers can recover more later. Unknown shapes must appear in the report as unsupported or quarantined, with their source bytes retained. + +The import preview should answer: what is here, what will become searchable, what could not be read, and how much space the import and its backup need. Select saved resources, likes, stars, and collections for this workflow. Direct messages, account-security data, ad records, and watch/search history remain separate choices. Public source content does not make Chris's saving activity public. + +Reuse the existing import stream and write records in batches. Commit bounded chunks and persist the last committed checkpoint, tied to archive hash, parser version, and selection. A restart must resume safely or offer an explicit replay. Cancellation reports what was committed; it is not completion. The report must account for every selected input record: created, updated, duplicate, skipped with a reason, or failed. Unsupported and missing categories must never look like an empty successful import. + +Treat archive content as data. Parse Twitter's JavaScript assignment wrapper without executing it, and apply size and path limits when reading ZIP entries. Keep personal archive contents out of Git, fixture files, and routine logs. Commit only sanitized structural fixtures; detailed local reconciliation reports remain private. + +Reimporting the same archive must add no duplicate resources, activity, or collection memberships. A later export should add new observations without overwriting personal notes or discarding old evidence. Reordered files and repeated playlist entries are real cases. Do not assume a deterministic ID based on row position solves them. A resource absent from a later partial export must not be treated as deleted, unliked, or unstarred. + +The garden and website need their own source mapping. Preserve Chris's authored commentary as Pages, and link its source URLs to imported resources. Preserve stable post IDs where available, otherwise retain URL aliases and snapshot provenance. Keep a third-party resource's content separate from Chris's writing about it. The existing garden pipeline keeps its own source of truth. + +### Add GitHub stars through the same import boundary + +GitHub is a first-class seed source. Since there is no local star export yet, provide a one-time read-only fetch for the selected account or accept a saved JSON snapshot. GitHub's [starring API](https://docs.github.com/en/rest/activity/starring) lists starred repositories and offers a custom media type containing the star timestamp. Fetch every response page. If permissions or rate limits stop the fetch, mark the snapshot incomplete. + +Save the response as a local import source with fetch time and coverage information, excluding credentials. The rest of ingestion uses the same preview, provenance, deduplication, and recovery path as ZIP exports. No continuous sync or repository cloning is required for the seed import. + +Use stable repository IDs when supplied, keeping owner/name and old URLs as aliases across renames. Preserve `star` as the native action, distinct from a social like or follow. Reuse the existing content/external-item and interaction schemas with an explicit GitHub mapping; review any vocabulary extension for compatibility. If star lists or categories are supplied, preserve them as collections; the basic REST star list does not by itself prove that grouping was captured. + +The description, topics, language, owner, and URL make a star useful immediately. Fetch README text and an available social preview as part of enrichment; repository cloning is unnecessary. Match accounts before a fetch; an inaccessible or private profile must not be reported as having no stars. + +### The graph should explain why things belong together + +Use the existing social graph as the durable source of imported facts, with ordinary Pages for notes and guides. A resource is the thing Chris saved; a like, save, star, or playlist membership is a separate observation about it. Keep every observation when matching the same resource across collections or platforms. + +```mermaid +flowchart LR + Archive[Archive or API snapshot] --> Evidence[Source record and import run] + Evidence --> Activity[Like, save, bookmark, or star] + Chris[Chris's source account] --> Activity + Activity --> Item[Post, video, paper, or repository] + Creator[Creator or owner] -->|authored or owns| Item + Collection[Playlist or saved collection] --> Membership[Membership with source order] + Membership --> Item + Post[Garden post or website page] -->|links to| Item + Note[Personal note] -->|comments on| Item + Item -->|cited by| Guide[Guide] + Suggested[Suggested topic] -. inferred relationship .-> Item +``` + +Start with relationships the source actually supplies: saved by, liked by, starred by, belongs to collection, authored by, and links to. Keep a post that links to a paper separate from the paper itself. Match platform items by stable platform IDs; use conservative URL aliases to connect references. Shared titles or similar creator names are not enough to merge identities. + +The same Instagram post can be both liked and saved. The current mapper includes the interaction kind in the content ID, so this needs an explicit resolution layer. Preserve old IDs and signed history while connecting equivalent resources; do not rewrite past records to manufacture a clean graph. Human notes remain separate from imported fields so later imports cannot overwrite them. + +Concepts and related-topic edges can grow over time through Chris's tags and optional enrichment. Each inferred relationship carries its evidence, method/model, and review state. Show it as a suggestion that can be rejected. A liked post, watched video, or starred repository does not establish a belief, an endorsement, or use of that software. + +Offer useful entry points before a whole-library graph: collections, saved items by creator, repositories by topic, and a small neighborhood around the open resource. Reuse the [saved views](../../packages/social/src/views/defaults.ts), graph lenses, and bounded canvas projection. Show when a result is paginated or truncated. A large map is optional; browsing relationships and answering a real question are required. + +Initial questions to make possible include “What did I save about this topic across platforms?”, “Which playlists contain this video?”, and “Which starred repos relate to this garden post?” An answer must open the underlying resource and show where the connection came from. If there is not enough text or evidence, say so. + +### Enrichment is part of importing + +Every unique imported link enters an enrichment job. This applies to the full corpus and to new captures, including ordinary web links. Chris should not have to open each card or select thousands of items to get useful titles and images. Saving the source record remains immediate and works offline; queued enrichment continues when the network returns. A durable import and a fully enriched library are separate milestones shown in the UI. + +| Resource | Metadata to seek | Visual preview | Searchable content | +| -------------------------------- | ------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- | +| YouTube video | Full title and description, channel, publication date, duration, language, canonical ID/URL | Best usable source thumbnail, cached locally | Description plus available captions or generated transcript | +| Instagram post or reel | Written caption, creator, permalink, post date, media type and carousel structure | Source image or video poster; a local frame if needed and media is available | Written caption plus speech transcript for accessible video/audio | +| Twitter/X post | Full available post text, author, date, and outbound links | Attached image or video poster when available | Post text and metadata for linked resources | +| GitHub repository | Name, owner, description, topics, language, canonical ID/URL | Repository social preview when obtainable; otherwise a labeled repository card | Description and fetched README text | +| Garden, website, and other links | Page title, description, author/date where supplied, canonical URL and site | Source preview image, then a site icon or labeled card | Existing authored text and available page metadata; extracted article text when supported | + +“All links” is the coverage target. It cannot mean that every provider will return every field. Each applicable field needs a value, an explicit pending/retry state, or a recorded reason it could not be obtained. An Instagram written caption is distinct from its video's spoken transcript. A synthetic card title derived from a caption must be labeled as derived, not presented as an original video title. + +#### Provider strategy + +For YouTube metadata, use the [Data API's `videos.list`](https://developers.google.com/youtube/v3/docs/videos/list) with video IDs and the relevant parts. It returns titles and descriptions and supports multiple IDs per request. Reconcile every requested ID against the response. An omitted video needs an unresolved/unavailable outcome, not an empty success. Use a managed provider credential when required; the desktop setup should not require a terminal. oEmbed or public page metadata can provide a limited fallback, but a title and thumbnail alone do not satisfy description coverage. + +For transcripts, prefer an existing caption track, including automatic captions when that is all the source offers. Preserve track language, timing, and whether the text was human-authored or machine-generated. YouTube's official [caption download API](https://developers.google.com/youtube/v3/docs/captions/download) requires permission to edit the video, so it cannot be the general solution for Chris's saved third-party videos. + +Evaluate a maintained local extractor such as [yt-dlp](https://github.com/yt-dlp/yt-dlp/blob/master/README.md) behind a replaceable provider adapter. It supports subtitle discovery and fetching without downloading the video. Prove its behavior on a representative sample before a large run; pin the tested helper version and report provider failures. The existing guessed-language timed-text fetcher is a starting seam, not a proven archive-wide caption service. Discover available tracks rather than treating a failed English request as proof that no captions exist. + +For Instagram, preserve the exported captions and nested metadata first, then resolve missing public metadata and posters through a tested provider. The upstream [Instagram extractor](https://github.com/yt-dlp/yt-dlp/blob/master/yt_dlp/extractor/instagram.py) is one candidate for a local adapter, not a guarantee that every saved reel is accessible. Meta's current documentation endpoints returned HTTP 429 during this research; exact official API coverage still needs verification against the chosen account and content types. Do not assume an embed response contains full text, speech captions, or downloadable media. + +If no usable caption track exists, offer automatic speech recognition over media already in the archive or otherwise accessible through the configured provider. Prefer the existing local engines, reusing [recording transcription](../../packages/recordings/src/transcribe/transcribe.ts) where its audio-processing contract fits. This fallback belongs in the first enrichment implementation for both YouTube and Instagram. A remote transcription provider is optional and needs an explicit budget/data-sharing choice. If the media cannot be accessed, retain an honest unavailable transcript state. Do not bypass private-content restrictions or silently extract browser credentials. + +Metadata lookup for the selected import is enabled by default under this requested workflow. Show the provider choices once, allow pause and retry, and keep paid services or account access explicit. No prompt is needed for every public link. Full article parsing can expand by provider; embeddings and model-generated topic clusters can wait. Captions, descriptions, and local thumbnails cannot. + +#### Thumbnails that remain useful + +Store thumbnail bytes in the managed blob store. An expiring CDN URL is provenance, not an offline image. Prefer a clear source thumbnail or poster with enough resolution for a large card, within a fixed byte and dimension budget. Keep its aspect ratio and offer sensible crops in the UI. For a carousel, retain the lead image and references to the other supplied media. If accessible local video has no poster, derive one and label its origin. + +When no image is available, show a deliberate resource card using a title, source identity, and icon. Record this as a fallback so it does not inflate fetched-thumbnail coverage. Do not generate an invented image that might be mistaken for the source. Distinguish absent images from a failed or interrupted image download. + +Validate content type, size, and decoded dimensions before accepting a fetched image. Reuse URL validation for redirects and avoid treating archived URLs as permission to fetch arbitrary local-network resources. Deduplicate by content hash, keep the last good preview when refresh fails, and retry expired source URLs through their provider. Cache small display variants if needed for fast grids. Backups must include the image blobs, not just their URLs. Publishing only includes the images selected in the guide preview. + +#### A persistent queue for the whole corpus + +The current [enrichment queue](../../packages/social/src/enrichment/queue.ts) is scoped to a session. The [shared feed hook](../../packages/views/src/social-enrichment/useSocialFeedEnrichment.ts) requests previews visible on screen and loads a limited set of enrichment rows. The [fetch path](../../packages/social/src/enrichment/fetch.ts) depends on a hub for platforms without its direct oEmbed path. These pieces do not yet meet full-corpus, restart-safe desktop enrichment. + +Add durable work items keyed by resource identity, capability, provider version, and language where relevant. Capabilities include metadata, thumbnail bytes, caption discovery, transcript fetch, local transcription, and indexing. Deduplicate work across likes, saves, and playlists; the observed 11,259 YouTube memberships should not fetch the same 9,273 video IDs repeatedly. Cover the entire corpus through paginated queries, not the first screen or the first 2,000 rows. + +Use bounded concurrency per provider, backoff, `Retry-After` where supplied, and persisted retry times. Pause across sleep, offline periods, app quit, or a run of refusals. Resume without redoing completed work. An unavailable credential or blocked provider should leave actionable queued work; it must not classify thousands of videos as having no captions. Prioritize recently opened resources while the rest of the corpus continues in the background. + +Run network work in a desktop background process so a purely local workspace can enrich without hosting a hub. Keep provider helpers separate from storage migration. Store attempt history and last successful results, and make changed extractor versions eligible for controlled retry. Installing a new app version should preserve queue progress and previously indexed material. + +Budget the expensive fallback separately. Discover metadata and existing captions before downloading audio. Estimate remaining audio duration and storage from a sample, limit local transcription concurrency, and keep the app responsive while work runs. Temporary media can be discarded after verified transcription under the chosen retention policy; the full transcript and its provenance stay in the library. Do not promise that thousands of videos will finish in one import session. + +Track metadata, thumbnail, and transcript outcomes independently. Use states such as queued, running, complete, partial, retryable failure, needs credentials, unavailable, and not applicable. A successfully fetched title must not mark its missing description or transcript complete. Report totals by platform and capability, with denominators based on distinct resources. Use **Finished with gaps** when work is blocked, unavailable, or deferred. Reserve complete field coverage for actual retrieved content; a stopped worker is not evidence of completeness. + +#### Index the recovered content, not just the card + +Fetched metadata already has a separate [SocialEnrichment schema](../../packages/social/src/schemas/enrichment.ts). Keep that separation from archive facts and personal notes, and record source URL, fetch time, language, provider version, content hash, and field-level completeness. Preserve full text outside bounded preview fields when necessary. Refresh failures must not erase an earlier successful result. + +The existing [transcript node builder](../../packages/social/src/transcripts/nodes.ts) creates searchable segments linked to their source video. Reuse that model, with explicit transcript version and track identity. Retain the full raw caption/transcript and cue timing so indexing can be rebuilt. Distinguish source captions, platform-generated captions, and local speech recognition. Translations and summaries remain separate derived records. + +Wire metadata and every transcript segment into the same search and retrieval path used by the Library and helper. Group results by resource, show the matching passage, and open the video at its timestamp where supported. A phrase late in a long transcript must be findable after restart with no network. Do not silently truncate at `textPreview` or `searchText` limits. An oversized cue, interrupted transcription, or incomplete segment write needs an explicit partial/error result. + +The current [YouTube fetcher](../../packages/social/src/transcripts/youtube.ts) and [scheduler](../../packages/social/src/transcripts/schedule.ts) have parsers and pacing, but this review found no application call site for running them. The fetcher also turns some read/parse failures into empty caption results. Close those gaps before claiming transcript coverage: failed reads, malformed payloads, login pages, and an exhausted language guess must not become “no captions.” + +Enrichment is complete as a product capability when every imported link is scheduled, progress survives restarts, recovered text is searchable, cached images render offline, and gaps are visible. Actual availability will vary. Measure metadata, image, and transcript coverage on the real corpus before deciding whether another provider is needed. + +### The seed data belongs in the durability contract + +Back up the imported graph, personal notes, collection memberships, job checkpoints, and retained source evidence together. Include fetched descriptions, full transcripts and timing, thumbnail blobs, and enrichment provenance. Derived search indexes can be rebuilt from that content. The current [large-archive storage policy](../../packages/social/src/import/storage.ts) can split archive storage; default source-record handling can also use sidecars. A backup of canonical node rows alone may therefore be incomplete. + +Manage a durable copy of each archive or all required source entries with hashes and an inventory, and include it in recovery. A path back to `.exports/` is useful provenance, but it is not a backup. Avoid duplicating hundreds of MiB at every 15-minute checkpoint: store immutable source blobs once and reference them from complete recovery manifests. A missing sidecar or source blob must fail the restore's completeness check. + +After importing, restore into an isolated workspace with the original `.exports/` directory unavailable. Verify that the imported collection, its evidence, and Chris's new notes still work. This also tests whether parser improvements can reprocess retained sources without asking Chris to download the original platform exports again. + +### Capture a URL and a thought + +Extend the shared workbench's capture flow. Pasting a URL opens a small form with title, optional selected excerpt, and “Why I saved this.” Save locally before fetching metadata. Offline capture succeeds with the URL as its initial label. Failed writes preserve the text. Repeated URLs offer the existing note or an explicit second note; URL normalization must not remove meaningful query parameters. + +For desktop convenience, add one explicit global shortcut that opens this form and returns focus to the prior app after saving. Read the clipboard only when invoked or pasted. A browser share target or extension can follow if the shortcut proves awkward. New URLs join the same metadata, thumbnail, and transcript pipeline as imported links. Bulk full-text PDF extraction can follow; the structural graph and video enrichment are part of the first pass. + +Use a normal Page whose document contains the source URL, excerpt, and commentary. Reuse the [URL utilities](../../packages/data/src/external-references.ts). Do not overload the Page's `canonicalUrl`, which belongs to publication identity. If the URL already belongs to an imported resource, offer to attach this note to it. The [ExternalReference schema](../../packages/data/src/schema/schemas/external-reference.ts) supports resource links; a standalone thought should still save without a new schema or a required import record. + +The notes are the value Chris owns. The linked site may disappear. Clearly distinguish a saved link from a saved copy of its content. Back up notes and every fetched description, transcript, and image. Keeping a full original video or a complete offline copy of every linked page is a separate storage choice. + +### Find something you only half remember + +Search imported and enriched titles, captions, descriptions, transcripts, source URLs, collection names, and personal note bodies. An exact URL should find its resource and linked notes. A phrase present only in a note body should work after the app restarts, with no network. Filter by source, creator, collection, and known dates. Show missing dates honestly and distinguish save time from import time. Each result should open the resource or Page with its source context. + +The current global Page search already loads document content. Reuse it for personal notes and combine results with indexed social fields. Measure search on the full imported corpus before adding another index. The AI retriever's property-based text path needs separate attention. “Search exists” does not prove a helper can see the same content the human finds. + +### Make the next useful resource + +A guide is another Page. Put source notes beside it, link back to sources, and write the missing context: who might find this useful, why these few links belong together, and where Chris's own experience ends. Possible first drafts include a reading path through local-first software or an introduction to resources Chris already shares in conversation. Chris chooses the topics; the app does not infer a client's needs. + +The optional helper operates on selected imported resources, linked notes, and the current draft. It can compare sources, suggest an outline, or identify an unsupported claim. It must cite the Page or source record behind a suggestion and admit when the selection lacks evidence. A video title is not a transcript. The import-level enrichment choice covers external metadata and caption fetching. Sending material to a remote model remains an explicit action with clear scope. No background rewriting, automatic publication, or unstated access to private notes. + +## Share a guide without sharing the workspace + +The first publishing path should produce a static page. The recipient needs neither an xNet account nor a running desktop app. Chris sees an exact preview, selects what to include, and exports a named snapshot. + +The existing [publication pipeline](../../packages/publish/src/pipeline.ts) provides useful parts. Two seams need special care. [Published-document resolution](../../packages/publish/src/published-doc.ts) can fall back to the live document when a pinned snapshot is unavailable; public export must refuse that fallback. The [CLI publisher](../../packages/cli/src/commands/publish.ts) takes a pre-rendered `PublicationFile` JSON input. It is not already a one-command export of a live workspace. + +Build the live-workspace adapter with an explicit allowlist of guide Pages and selected assets. No transitive export of private backlinks, notes, tags, client names, or imported saving activity. Linking to a public repo or video does not approve publication of Chris's star date, collection membership, or other saved items. Even a private link's visible label can leak context, so preview the final rendered page and handle unresolved links deliberately. Republishing updates the public snapshot only after another explicit action. + +The proposed destination is `digitalgarden/public/guides/`. Its existing [build script](https://github.com/crs48/digitalgarden/blob/main/scripts/build.mjs) copies `public/` into the generated site, and its [Pages workflow](https://github.com/crs48/digitalgarden/blob/main/.github/workflows/pages.yml) already deploys it. That keeps the garden's existing content flow intact. Make the output directory configurable. + +Export through a staging directory, then replace only the managed guide directory after validation. Remove obsolete generated pages inside that boundary; never clean the rest of the garden. In the first iteration, Chris commits and pushes the generated output through the existing site workflow. The desktop says **Exported**, then **Published** only after deployment is confirmed. Automating this last step can follow once the export earns use. + +Deleting a local draft does not retract a public page. Offer an explicit unpublish/export operation and explain that copies and caches may remain. Sensitive session notes and client records are outside this public-guide workflow. Private client spaces need a separate review of authorization, revocation, and recovery before relying on them. + +## Implementation checklist: five bounded passes + +Unchecked items are proposed work. Checked items carry implementation evidence below. Each pass should end with a useful, inspectable result; do not reopen the entire platform backlog. + +### A. Make the daily desktop safe to trust + +- [x] Remove database deletion on old schema, inspection error, or failed startup. Preserve originals and expose a recovery state. +- [x] Add a read-only compatibility probe with distinct missing, supported, old, future, and unreadable outcomes before any writable open. +- [ ] Inventory all desktop data and key locations; define a complete checkpoint manifest and fail if required content is missing. +- [ ] Protect the daily profile from every development launch; migrate existing profile selection without losing data or identity. +- [x] Wire acknowledged text saves and a coordinated renderer/data-process flush; retain failed document writes for retry. +- [ ] Extend acknowledgement to all mutation paths and measure `FULL` transaction durability on the daily workload. +- [x] Create verified local checkpoints on quit and before updates, keep a bounded history, surface failures, and provide Restore in Settings and startup recovery. +- [x] Add changed-data periodic checkpoints, recent/daily/weekly retention, pinned pre-upgrade copies, and visible failure status. +- [ ] Complete recovery coverage for non-reconstructible settings and keys beyond native workspace files. +- [ ] Wire complete portable export, encrypted cross-Mac key recovery, and one off-device backup destination; show its actual protection status. +- [x] Implement ordered migration on a candidate copy, validated promotion, and a recovery path that preserves post-update edits. +- [x] Route every updater install path through the flush/checkpoint barrier; keep normal no-format-change updates simple. +- [ ] Extend the existing release checks with a real installed Mac upgrade and restore exercise; prove signing and Keychain continuity across releases. + +**Implementation evidence (2026-09-29):** startup now probes a disposable database/WAL copy before either desktop store opens for writes, preserves unknown and damaged originals, and exposes a native recovery dialog before the renderer starts. Targeted SQLite, compatibility, and profile tests passed (74 tests); `pnpm turbo run typecheck` passed (101 tasks). The real Electron smoke checks passed for clean boot and restart persistence (2 tests). A separate isolated Electron recovery run observed the unversioned warning, zero normal windows, unchanged source bytes, and no newly created blob database. Desktop Node typechecking also exposed two missing import-preview coverage fields; both now reach the caller. Source-launch profile isolation is implemented, but profile transition and full recovery remain unchecked until the rest of pass A is proven. + +**Atomic record-save evidence (2026-09-30):** the desktop IPC adapter now uses +`applyNodeBatch` for ordinary structured edits and explicit transactions. Native +SQLite commits records, indexes, signed history, and the clock together; failed +batches emit no success notification. Desktop history reads now decode the +shared envelope used by imports, preserving signed change IDs and batch positions. +Five backend integration tests cover restart, rollback/retry, deleted-record +restore, imported-record edits, and closed-storage failures. The real Electron +app also passed injected-write failure and retry, direct-exit restart, and an +acknowledged last-moment edit before normal quit with a verified recovery copy. +A synthetic 100-create/100-update sample measured median 4.3/5.3 ms and p95 +8.6/10.0 ms with `FULL` durability. The broader mutation-acknowledgement item stays +open: pending work before IPC, separate mutation paths, overlapping writers, and +the full daily workload are not covered by this proof. + +**Recovery implementation evidence (2026-09-29):** document writes now serialize by store and document, retain failed snapshots, and retry before reopening. Desktop SQLite uses `synchronous=FULL`. A native checkpoint covers both databases and files under `xnet-data`, verifies each file and database, and retains twenty copies. Restore uses a durable rename journal and keeps the replaced workspace separately. Restored workspaces pause automatic sync, the local API, and the agent bridge until an explicit reconnect. This scope excludes Chromium settings and sign-in sessions; it is not portable encrypted or off-device protection. + +Focused verification passed: 29 compatibility/checkpoint/restore tests, six document-barrier tests, two quit-barrier tests, 32 SQLite adapter tests, and 22 existing data-process/sync checks. Workspace typechecking passed (101 tasks), desktop Node typechecking passed, and two existing real Electron smoke tests passed. In isolated desktop runs, text typed immediately before quit survived relaunch; Settings created a verified copy; restoring an older node recovered its original value while keeping newer work; and the restored app reported recovery mode with sync paused. The direct desktop renderer typecheck also exposed existing unrelated errors in its composite file list, form props, effect cleanup signatures, and sync interfaces; it is not reported as passing. Signed installed-app upgrades, Keychain continuity, periodic scheduling, and full protection remain unproven. + +**Scheduled recovery and migration evidence (2026-09-29):** the scheduler now checks once a minute and copies changed native data when fifteen minutes overdue. Retention keeps recent, daily, and weekly points, the latest two, and pinned pre-upgrade points. An isolated Electron run observed its first scheduled copy, a visible failure when the destination was unavailable, and successful retry after restoring that destination. Identity initialization now happens before store writes; a separate native run with a missing identity observed zero normal windows, unchanged database bytes, and no newly created blob database. A packaged profile cannot use the reserved development prefix. + +Candidate upgrades currently support storage version 8 to 9, using a historical version-8 schema fixture from `6650c1f39^`. The original is pinned before ordered SQL runs on a separate copy. Column contracts, row counts, foreign keys, and database integrity must pass before journaled promotion. Six migration tests cover successful preservation, failed SQL, malformed schemas, unsupported versions, and locked identity. An isolated Electron run recovered exact note text after the upgrade, retained the original database bytes, and opened with networking paused for review. Earlier schemas still stop in recovery. The [storage inventory](../reference/desktop-storage.md) records remaining gaps; browser state, portable encrypted recovery, off-device protection, and signed release continuity are not complete. The Mac has an Apple Development identity, but no Developer ID release certificate was found locally and the repository's release-signing secrets are absent. + +**Damaged recovery catalog evidence:** three additional failure tests distinguish missing recovery storage from unreadable points, preserve damaged manifests and symlinks without pruning them, and allow new verified copies. The 22 checkpoint, retention, and migration tests passed. A real isolated Electron run displayed the damaged-point warning, made fresh copies, quit, and reopened while preserving the damaged bytes. A source that changes during a copy still fails explicitly and can be retried. + +**Exit:** Chris can save notes, quit, reopen, install an update, and recover an earlier copy without a terminal. Development uses a different workspace. No valuable collection moves in before this exit is demonstrated. + +The encrypted native export now has desktop controls and seven failure/recovery +tests. A real Electron run exported through Settings, removed the entire test +profile, restored from the encrypted folder alone, and recovered the item and +retained source file while preserving the replacement generation. Different +mock key stores verify seed rewrapping independently of the source key store. +This does not complete the portable-backup checkbox: non-native settings and +keys, a retained off-device destination, and a second physical Mac remain +unproven. The format and explicit limits are recorded in ADR-40. + +**Logical settings recovery evidence (2026-09-30):** new recovery points include +an encrypted snapshot of the known desktop settings, AI provider key, workspace +layout, consent choices, and unfinished capture draft. Restore applies it before +settings consumers load, and a per-restore receipt preserves later changes on +ordinary restarts. Portable format 2 rewraps the snapshot with the destination +key store; the reader still accepts format 1. The 70 storage tests passed, +including corrupted settings, locked keys, interrupted application, distinct +source/destination key stores, and a legacy-format fixture. + +A real Electron run exported through Settings, deleted the entire test profile, +restored the encrypted folder, and displayed the exact unfinished capture in the +form with the recovered theme. It also recovered the test provider key, omitted +temporary authorization tokens, kept the original identity and note, and +preserved a later settings change across another restart. A separate damaged-settings launch opened no normal windows, preserved the +original database and damaged settings bytes, and showed native recovery before +creating the blob database. Parser errors omit credential content. The production +build and desktop main-process typecheck passed. The separate renderer typecheck still +has the six previously recorded errors. ADR-45 and the storage reference define +the new coverage. Device-bound sessions and unlisted browser state remain +excluded; the real daily-profile inventory and off-device/physical-Mac proof +remain open, so the broader recovery checkboxes stay unchecked. + +**Existing-profile observation (2026-09-30):** the local `xnet-desktop` directory +contains version-9 storage, 207 nodes, and 20 saved Yjs states, but no native +identity-seed file. Chromium storage is also present. No migration or identity +replacement was attempted. This profile would stop at the missing-identity +guard; it is not a passed upgrade fixture. Preserve it while tracing legacy +ownership rather than enabling the test identity or writing a replacement seed. +The [storage inventory](../reference/desktop-storage.md) records this remaining +pass-A prerequisite. + +### B. Seed the library from the existing corpus + +- [x] Add a read-only inventory and preview for `.exports/`; reconcile selected categories and unknown formats against the observed source counts. +- [x] Fix the Twitter/X, YouTube, and Instagram adapter gaps using sanitized fixtures from the actual archive shapes, including nested Instagram collections. Report unavailable bookmark data explicitly. +- [x] Add GitHub stars as a seed source via saved JSON or a complete paginated read-only snapshot, preserving repository IDs, native star activity, and timestamp/coverage limits. +- [ ] Import garden and website material with authored commentary, source URLs, and stable provenance; connect it to matching imported resources. +- [ ] Prove faithful import on a small representative sample, then import the full selected corpus in resumable batches with an honest reconciliation report. +- [ ] Preserve raw source evidence and sidecars in managed storage and backup; restore the corpus without access to the original `.exports/` directory. +- [ ] Verify same-archive and overlapping-export reimports, preserving distinct save/like/star actions, repeated memberships, and Chris's notes without duplicating resources. +- [ ] Wire Library collections, source filters, creator views, and bounded graph neighborhoods through existing social schemas, saved views, and graph lenses. +- [ ] Schedule every unique imported link and new capture in a persistent desktop enrichment queue, with provider-specific pacing, retry, pause/resume, and separate capability coverage. +- [ ] Fetch complete available titles/descriptions and source metadata for YouTube, Instagram, Twitter/X, GitHub, and ordinary web links; keep unavailable fields explicit and support desktop use without a hub. +- [ ] Fetch and cache source thumbnails/posters in the managed blob store, generate labeled local posters only from accessible media, and provide honest fallback cards. +- [ ] Discover and import available video caption tracks; implement local transcription of accessible media for YouTube and Instagram when captions are unavailable, with language, timing, and partial-result handling. +- [x] Index full enriched descriptions and all transcript segments, join results to source resources and notes, and verify late-transcript search after restart without a network. +- [ ] Validate enrichment on a representative real sample, then run it across the whole corpus; report field, image, and transcript coverage with unresolved reasons, storage use, and remaining work. + +**Exit:** Chris can rediscover an old save by enriched metadata or a phrase in an available transcript, see a useful local preview, inspect its collection, and follow a source-backed connection. Every selected record and enrichment job has an explained outcome. The corpus and its fetched content can be recovered and repeated safely. Any unavailable transcripts or metadata remain visible in the coverage report. + +**Seed preview evidence (2026-09-29):** `pnpm exec tsx scripts/inventory-personal-library.ts --output /tmp/xnet-seed-inventory.json` reads the selected archive categories without opening a workspace database. It records archive hashes, adapter versions, excluded buckets, unclassified entry counts, emitted and unique record counts, and unresolved warnings. The parsed YouTube count corrects the earlier estimate to 33 catalog rows and 11,259 memberships across 32 files. All 11,259 memberships now retain distinct IDs, covering 9,273 videos. One ambiguous catalog-title match is kept as a separate collection with a warning, producing 34 collection nodes rather than guessing a join. + +Instagram now maps all 25 liked comments and 15 named collections, alongside the saved-post and music collections. The preview accounts for 14,763 memberships and 9,599 unique content nodes. One repeated interaction resolves to an existing deterministic interaction ID; every source record remains accounted for. Twitter contributes 10,028 likes and explicitly reports absent bookmark data and missing native like timestamps. The social package's 231 tests and its typecheck passed for these adapters. These are read-only previews; no personal archive has been committed to a workspace. Existing imports from adapter 0.1 still need an ID-consolidation migration before claiming cross-version reimport fidelity. + +**GitHub and source retention evidence (2026-09-29):** the authenticated account's snapshot contains 1,514 unique repositories from 16 completed pages, each with a native star timestamp. It lives in the ignored `.exports/github-stars.json` with private file permissions. The repeatable capture command is `node scripts/snapshot-github-stars.mjs `. It refuses to replace an existing file and promotes output only after every page succeeds. This is a current-stars snapshot, not deleted-star history or an atomic view of a changing account. + +The desktop picker now accepts saved JSON as well as ZIP exports. GitHub repository IDs survive renames; stars remain separate dated interactions; imported visibility stays private. Before writing nodes, the desktop retains the exact reviewed source bytes under `xnet-data/import-sources//`. A mismatch fails before node writes. The screen explains that this private recovery copy includes unselected archive categories. Five custody tests cover changed input, damaged copies, concurrent retention, missing originals, and invalid paths. All 237 social tests and workspace typechecking passed. An isolated real Electron run used the command palette and import controls, imported a synthetic snapshot, removed its original file, restored a recovery point, and found both the repository and its retained source after restart. Persistent import-job resume, the full personal import, and enrichment remain unchecked. + +Desktop import progress now survives restart under `xnet-data/import-jobs`. +Five journal tests cover interrupted state, atomic publication, invalid progress, +relocated sources, and private-key exclusion. A real Electron exercise imported +1,500 synthetic GitHub repositories (6,004 draft records), paused after 5,000 +acknowledged records, quit, removed the original export, and resumed to 6,004. +The resulting store contained exactly 1,500 resource nodes. Its recovery point +included the job journal. This is resume evidence, not the still-pending faithful +full-corpus acceptance run. ADR-41 records replay semantics and version limits. + +**Desktop Library evidence (2026-09-29):** the app now has a Library entry +point, source filters, bounded card pages, source details, and separate capability +coverage. A private, versioned SQLite queue in the existing data process retains +pause/retry state and full source text. Imported resources are scheduled after +commit; the Library can also scan existing imports. Recovery pauses this writer +and includes its database. Source thumbnails are saved in the managed blob store. + +A live read-only sample fetched two exported YouTube links with descriptions of +2,901 and 3,063 characters and 2,557 and 2,649 caption cues. A public GitHub +fixture returned a full README. In an isolated Electron profile, the first video +completed all four capabilities. After quitting and restarting with the renderer +offline, its 157,779-byte thumbnail decoded from local storage and a caption at +2,402,320 ms was searchable. The recovery manifest included `library.db`. + +This is a narrow working sample. Instagram/Twitter coverage, local ASR, a managed +helper setup, garden ingestion, full-corpus reconciliation, and the installed +release exercise remain unproven. No full personal corpus has been imported into +a daily workspace. ADR-42 records the local storage boundary. + +**Garden evidence (2026-09-29):** the standalone JSON picker now accepts the +garden's version-1 format. Its read-only inventory found eight resources, eight +authored commentary records, eleven category/tag collections, and twenty-seven +memberships. Known YouTube, Instagram, and Twitter URL aliases use the same +resource IDs as their archive importers. Distinct source posts keep distinct +notes and memberships. The retained JSON carries the original profile, media, +mentions, and source-post evidence. + +All 242 social tests and eight Library storage tests passed. In an isolated +Electron exercise, a synthetic garden entry appeared in Library search through +its authored note, with a link to the original post. After removing the input +file and restoring a recovery point, both the note and retained JSON survived. +This does not check off the broader garden-and-website item: website material +and URL reconciliation for renamed GitHub repositories remain outstanding. + +The final offline Library check retrieved a stored 2,901-character description, 2,557 caption cues, and a 157,779-byte cached thumbnail. After restart, the image decoded without a network and the cue at 2,402,320 ms was searchable. Selecting that result displayed its matching passage and a YouTube link with the corresponding timestamp. This proves the local index and result path for the sample; whole-corpus coverage and local ASR remain unchecked. + +**Managed helper evidence:** Coverage & gaps now offers installation, cancellation, +and repair of the pinned macOS yt-dlp release. The native process checks its +expected byte count, SHA-256, and executable version before atomic promotion. +It does not start enrichment or read browser cookies. Ten installer tests cover +multi-chunk downloads, damaged bytes, failed repair, cancellation, and symlink +refusal; seven HTTP tests cover public-address validation, redirects, fallback, +byte limits, deadlines, and cancellation. Six provider tests also pass. + +A real isolated Electron run cancelled a download, observed no partial files, +kept enrichment paused, and restarted cleanly. A second attempt displayed a +retryable timeout. This Mac could reach the same official GitHub asset from +Node and curl, but not from Electron’s Node or Chromium networking. Little Snitch +is running; a blocking rule has not been confirmed. No firewall setting was +changed. Successful live installation remains unchecked, and a source-app test +would still not prove signed-release helper behavior. + +**YouTube enrichment repair (2026-10-02):** the founder's development workspace +now contains 9,274 imported resources. Clicking Start enrichment previously put +every local index job ahead of every network job, at one job per second. That +could delay the first title for more than two hours. The YouTube provider also +required a helper that was absent on this Mac. A failure on one video could put +all YouTube work into backoff. + +The queue now drains local indexing alongside bounded network workers. Metadata, +images, and captions have separate pacing; only a provider-wide rate limit pauses +the provider. Pausing cancels active requests and keeps their retry budget. +Restart preserves completed jobs and retry deadlines. Parallel results merge +with the latest saved resource, and native graph writes stay serialized. + +YouTube metadata now comes from the public watch page, with a public oEmbed +fallback that records partial coverage. This needs no helper or browser cookies. +Full descriptions and provider evidence stay in the local Library. The progress +strip shows completed and pending metadata, saved thumbnails, active requests, +and the next retry time. Empty caption responses become explicit access gaps. +They do not count as absent tracks or retrieved transcripts. + +The first live pass saved 120 full metadata records, one partial preview, and +121 thumbnails in the actual imported workspace. Two videos had explicit +metadata access/unavailability gaps. Cached images decoded from local blobs, +including resources beyond the first forty cards. A verified recovery copy was +made before the repair and another after this pass. The batch remains pending; +these counts do not establish whole-corpus coverage or caption availability. +After restarting with enrichment paused, all 121 images and completed metadata +jobs remained saved. Forty cards decoded their local images with the renderer +offline. Starting enrichment resumed the remaining queue. +The next pass reached 218 full metadata records, one partial preview, and 218 +saved thumbnails. Six metadata gaps remained explicit. Empty caption responses +were recorded as blocked jobs, with no repeated JSON-error retries. The batch +will continue from its saved position while the app is running. + +The focused Library suite passed 54 tests. New cases cover a large index backlog, +a hung request, a restricted video, cancellation, provider-version upgrades, +out-of-order image/caption completion, empty captions, and all 250 videos in a +fixture batch across rate limiting and restart. Native-process typechecking, +changed-file lint, and the desktop build also passed. Full-corpus enrichment and +the broader acceptance checklist remain open. + +**Instagram import and enrichment repair (2026-10-02):** the actual archive +completed in the founder's development workspace: 58,293 source records across +24 batches, with 44,262 created and 14,031 updated records. Native graph reads +confirmed 17 collections and all 14,763 memberships. There are 9,661 Instagram +content nodes, of which 9,525 have links that can enter the Library queue. The +136 records without web links remain in the graph. Recovery copies were verified +before and after the import. + +The first native enrichment run found a mismatch the synthetic samples had +missed. Some exported content IDs are numeric Facebook record IDs; the post URL +contains a different Instagram shortcode. The provider now reads that shortcode +from the saved URL while preserving the imported identity and its relationships. +Public post embeds supply written captions, authors, and posters without the +optional video helper. A public-page fallback keeps its preview coverage partial. +Neither path executes source scripts or reads browser credentials. + +The parser keeps the full written caption, including line breaks, entities, and +hashtags, and excludes the author label and comment links. A card's short label +is derived from that caption. Spoken transcripts remain a separate, explicit gap: +public embeds do not expose those tracks, and local transcription has not run. +Private, removed, or login-limited posts record a gap without stalling other posts. +The provider upgrade preserves existing YouTube results and retry deadlines. + +In the native app, four saved posts returned searchable captions of 83, 800, +409, and 372 characters, with cached image blobs of 603,415, 197,024, 89,446, +and 194,673 bytes. Both the embed and public-page fallback produced real cards. +After a verified recovery copy and restart, the queue stayed paused and all +18,799 Library resources remained. With the development modules loaded and the +renderer set offline, 37 Instagram thumbnails decoded from local blobs. The four +sample captions were still searchable after restart. This checks saved data in +the native development app; packaged offline startup is a separate acceptance test. +The focused Library suite passed 64 tests, including numeric export IDs, full +captions, partial previews, rejected login pages, rate limits, and preservation +of YouTube work across restart. The live pass reached 44 enriched posts and 44 saved thumbnails, with four +metadata access gaps and 9,477 posts still queued. Native-process typechecking, +changed-file lint, and the desktop build also passed. These samples do not +establish whole-corpus coverage or spoken transcript support. + +**More importers and source captions (2026-10-02):** The remaining export +adapters were exercised through the native import screen. The selected +scope includes social activity and AI conversations; direct messages, billing, +and account/security categories are excluded. Complete exports remain in the +local recovery copy, including excluded categories. The earlier YouTube playlist +and Instagram saves/likes selection is unchanged. + +GitHub, TikTok, Reddit, X, Claude, ChatGPT, Grok, and garden imports completed. Every expected stable +record ID was found in the native database, and every nonempty text field matched +an observed export value. The comparison covers full text, not card previews. +Six AI messages exceed 20,000 characters. A 157,679-character Claude conversation +was found by words at its end. Its card stays bounded to 600 characters, and its +private conversation text needs no network fetch. + +The Reddit pass found a real loss of text: later saved/voted references could +clear or shorten a post body imported earlier in the same archive. Adapter 0.1.2 +merges those observations and emits each content item once before writing. It preserves the higher-confidence +body, or the longer body when confidence is equal. A regression fixture covers +this overlap. The native reimport also resolved all 11 field mismatches caused by repeated +content in the same write batch. The full reconciliation follows below. + +Enrichment now routes cited links by their URL while retaining their archive +provenance. TikTok public data supplies posters and available subtitle tracks. +GitHub pages supply repository details and their rendered README. Reddit embeds +supply partial previews. Ordinary pages can supply embedded subtitle tracks. +Caption retrieval tries alternate tracks and formats; YouTube can refresh signed +tracks through the tested local helper. Empty caption responses stay gaps. + +In the native queue, one YouTube video returned 125 timed English cues through +the helper fallback. A TikTok video returned four English subtitle cues from its +public page. Both transcripts were searchable, and both thumbnails were read +from saved blobs. Instagram written captions remain separate from spoken audio; +local transcription has not run. + +Large imports also exposed a slow final Library scan. The data service now +hydrates each page in one batch instead of reading every record separately. +The native regression checks order, document bytes, and one batch read. +New caption indexing and selected-item retries take priority over the bulk +backlog, while provider rate limits still apply. + +The focused native batch and Library run passed 87 tests across 12 files. The +full social package run passed 244 tests across 22 files. Desktop main-process +typechecking and the production build passed. The full pre-push run passed +12,677 tests with four skipped, and all 101 workspace typecheck tasks passed. +Repository ESLint reported no errors and 469 existing warnings. The separate +renderer typecheck still reports six existing issues. The exploration-link scan +also finds 200 stale references in ignored worktrees and local agent memory, +with none in these edited documents. These checks do not complete the +whole-corpus enrichment or signed-app acceptance items above. Hosted checks at +`dd8ae0489` still fail the existing exploration-age gate (51 overdue documents +against a baseline of 41) and dependency audit (five high-severity advisories). +Neither baseline was raised. + +The native imports completed for all eight additional sources. A separate +read-only pass restaged the selected exports and compared every expected graph +record ID, schema, and nonempty text hash against the workspace database. + +| Source | Expected graph records | Source material | +| --------- | ---------------------: | ------------------------------------------ | +| GitHub | 5,768 | 1,514 starred repositories | +| TikTok | 14,122 | 5,055 content records; 38 collection names | +| Reddit | 8,006 | 3,734 posts and comments | +| X/Twitter | 32,098 | 12,632 content records; 8 AI conversations | +| Claude | 2,302 | 130 conversations and 1,762 messages | +| ChatGPT | 35,069 | 676 conversations and 7,419 messages | +| Grok | 23,565 | 766 conversations and 11,370 messages | +| Garden | 55 | 8 sources and 8 linked notes | + +The 120,985 records include archive records and source relationships, but exclude +per-run job records and raw source sidecars. All IDs and schemas matched, and all +37,518 nonempty text comparisons passed. No shorter text variant remained. +The checked visibility fields were private. TikTok's export names collections +without assigning saved videos to them; those missing memberships stay missing. +The garden reused two existing source records. + +The Library contains 55,973 resources. An explicit scan took 7,976 ms after the +batch-read fix. At the paused snapshot, metadata work had 5,213 complete results +and 721 partial results, with 48,037 queued or retrying. There were 5,834 saved +thumbnails and 25 retrieved transcripts, including video links cited in AI +conversations. Another 53,998 caption jobs were queued or retrying. These are +capability counts, not a claim that every link is enriched. The structured +database occupied about 3.9 GiB and the Library database about 448 MiB; +retained archives, blobs, and recovery copies use additional space. + +The native GitHub check recovered a 7,802-character repository description and +README, found its ending in search, and read a 106,016-byte cached image. Another +GitHub image request returned HTTP 429 and retained its cooldown. X returned +metadata through the local helper. A garden source returned a searchable public +page description while retaining its authored note. Reddit's public embed check +returned a partial preview, not a full post body or spoken transcript. + +With the renderer offline, the reopened TikTok details loaded their saved +540-pixel-wide image from a local blob and displayed the transcript. The app +returned online afterward. After the final restart on `dd8ae0489`, the same YouTube and TikTok transcripts +remained searchable and their saved image bytes were readable. A verified native +recovery copy, `1790990379981-e87f7137-4d58-4652-a025-287d71af6c5e`, covers 28 +workspace files. Enrichment was then resumed. This is an on-disk copy; restoring +this complete enlarged corpus on another Mac has not been tested. + +### Enrichment throughput and richer pages (2026-10-02) + +The next pass started with 55,973 resources, 7,508 fetched titles, 7,113 nonempty +fetched descriptions, 6,551 cached thumbnails, and 111 transcripts. These counts +include partial metadata and outbound citations from conversation archives. +They do not mean that 7,508 links have every field. Most network work remained +queued: GitHub had 73 fetched titles, Instagram 280, and YouTube 5,853. + +The bottleneck was partly local. One network-job claim took 421 ms on the real +queue: SQLite scanned pending work and sorted it against the resource table. +The revised query reads queue order from its existing partial index. On a +disposable copy of the same Library, twelve metadata claims took 0.13–0.64 ms. +The copy migrated in 4.0 seconds. These are local operation timings, not an +estimate of website throughput or total enrichment duration. + +Search updates also scanned all 127,761 passages to remove a resource's old +text. A reconciled row-ownership index now targets those passages directly. +The sampled lookup previously took 41 ms; a complete reindex on the migrated +copy took 0.95 ms. Restart reconciles ownership with FTS, including writes made +by an older app. Source records and fetched content are preserved. + +Network scheduling allows ten active jobs, with six metadata slots, two image +slots, two transcript slots, and at most three active jobs per source. Existing +provider pacing and persisted cooldowns still apply. Outbound web citations +use the destination hostname instead of their archive's platform for pacing; +the corpus contains 5,012 distinct provider/host keys. A throttled website no +longer delays every unrelated link imported from the same AI archive. Repeated +provider rate limits remain retryable beyond the ordinary per-resource attempt +limit. Successful metadata brings its dependent image/caption work forward. + +GitHub extraction now reads its embedded repository overview as well as the +rendered README. It retains public descriptions, README text, topics, website, +stars, forks, and license details without retaining viewer state or page tokens. +A current public xNet repository page yielded 5,482 README characters and 18 +topics. Public article pages contribute their visible article text; they remain +partial because the public page can omit content. Older GitHub and web previews +are queued once for this richer extraction without clearing their saved data or +refetching completed media. + +The focused Library suite passed 96 tests, including existing caption parsers, +host isolation, queue migration, passage replacement, provider concurrency, and +repeated throttling followed by recovery. Desktop main-process typechecking and +changed-file ESLint passed. Whole-corpus enrichment remains an open acceptance +item. Instagram written captions still do not establish spoken transcript +coverage, and automatic local transcription remains unconnected. + +In real Electron on `def125f38`, the first 120 seconds produced 260 fresh +metadata results across GitHub, Instagram, YouTube, TikTok, Reddit, X, and web +sources. Some refreshed existing previews. The snapshot held 7,676 fetched +titles, 7,257 descriptions, 6,637 thumbnails, and 124 transcripts; 46,524 metadata +jobs were still queued, retrying, or active. The app reported no Library error. + +Native search found the ending of a newly fetched 41,794-character GitHub README +and text from a 90,404-character web article. Cached images decoded for GitHub, +Instagram, YouTube, TikTok, and web samples. New YouTube and TikTok transcripts +contained 149 and 55 cues respectively, and their late passages were searchable. +The GitHub result also opened through the Library search UI. Full pre-push +verification passed 12,706 tests with four skipped, and all 101 workspace +TypeScript tasks passed. This is sample verification plus a progressing bulk +queue, not a completed enrichment pass. + +A later pass reached 9,989 fetched titles, 9,160 descriptions, 7,877 thumbnails, +and 268 transcripts. All 55,973 resources had completed local indexing; 43,681 +metadata jobs were queued or retrying. A GitHub thumbnail response exposed a +cross-host scheduling error: its 15-minute CDN cooldown also held repository +pages. HTTP errors now retain the final response hostname. A rate-limited image +host pauses the source's thumbnail lane when it differs from the source host; +source-host or unidentified throttling retains the whole-source pause. Existing +cooldowns are not shortened. The focused suite passed 100 tests, including +redirect attribution and both cooldown scopes surviving restart. + +### 3D link graph evidence (2026-10-02) + +Library now opens a full-window 3D graph of saved web links. The native run used +all 54,233 links in the 55,973-resource Library. It found 95,367 connections through +60 collections, 6,451 creator-name groups, 15,144 tags, and 80 categories. All +30,764 links without known connections stayed visible. The projection reported +no unreadable metadata and no edges pointing to missing nodes. The remaining +1,740 local text resources and conversations stay in Resources. + +Connections come from imported memberships, explicit topics and tags, source +hashtags, categories, and creator names scoped to a platform. Matching names do +not merge identities. Missing memberships are not invented. AI-generated +categories remain open work; the UI says so. + +```mermaid +flowchart LR + A[Imported social records] --> C[Read-only graph projection] + B[Saved Library metadata] --> C + C --> D[Serialized desktop snapshot] + D --> E[3D force layout in a worker] + E --> F[GPU points and relationship lines] + F --> G[Hover or pin a link] + G --> H[Load saved image and full metadata] +``` + +The overview sends compact labels and relationships, then loads source details +on demand. Its native boundary carries serialized JSON to avoid copying and +freezing tens of thousands of objects through Electron's context bridge. The +view uses [Three.js points](https://threejs.org/docs/pages/Points.html) and +[OrbitControls](https://threejs.org/docs/pages/OrbitControls.html), with +[d3-force-3d](https://github.com/vasturiano/d3-force-3d) in a disposable worker. +It adds no database schema or persisted relationship type. + +In the real Electron development profile, search focused an enriched YouTube +link, loaded its saved 1280-pixel thumbnail, and opened a 45-link neighborhood. +Hover picking showed the same title, image, and connections. The metadata +expander included source fields and enrichment provenance. The GitHub filter +included all 1,514 repositories and reached a settled layout. Orbit, zoom, +relationship filters, and keyboard search are available. Escape closed the view, +removed its canvas, and restored focus access to the rest of the app. Reopening +returned to all links. A 90-frame sample during the full layout measured 8 ms +median and 9 ms at the 95th percentile between animation callbacks on this Mac; +this is a local responsiveness observation, not a hardware-independent benchmark. + +The focused Library and graph checks passed: 92 tests across 13 files, including +complete pagination, duplicate memberships, isolated links, platform filtering, +malformed metadata, deterministic 3D seeds, and the recovery read barrier. The +main-process typecheck passed. The standalone renderer check still reports the +six pre-existing errors recorded above, with no new graph diagnostics. + +- [x] Add a desktop 3D view of all saved web links with source relationships, + hover metadata, source filters, and selectable neighborhoods. +- [x] Add keyboard autocomplete and browsable category, tag, playlist, and + creator filters with any/all membership matching. +- [ ] Add reviewed AI topic suggestions with visible evidence and provenance. + +**Graph navigation follow-up (2026-10-02):** Search now suggests ranked links +and groups below the input. Arrow keys and Enter navigate to a preview, including +when the inspector was hidden. Escape dismisses suggestions before closing the +graph. Matching supports separate title words, URLs, and accent-insensitive +labels. Suggestions stay within the current view and render at most twelve +matches, with the full match count shown. + +The group browser lists saved categories, tags, playlists, and creators by +unique link count. Counts reflect the source filter. Multiple selections use +either a union or intersection; hiding relationship lines does not remove +membership constraints. Filter chips are removable, and clearing filters restores +isolated links too. Neighborhoods stay within the active filters. + +```mermaid +flowchart LR + A[All saved web links] --> B[Source filter] + B --> C[Any or all selected groups] + C --> D[Optional neighborhood] + D --> E[Visible 3D graph] + E --> F[Ranked autocomplete] + F --> G[Focus camera and open preview] +``` + +Native verification used 54,233 links and 96,285 connections. Selecting JavaScript +and Ruby categories showed 661 links with **any**, and zero with **all**. A tag +selection showed seven links; a playlist selection showed 7,230. Keyboard preview, +no-result feedback, source filtering, relationship visibility, chip removal, +and clearing back to all 54,233 links were exercised in Electron. Eleven focused +model/navigation tests passed, including duplicate memberships, missing groups, +edge remapping, search ranking, Unicode, and source-scoped counts. Changed-file +lint passed. The standalone renderer typecheck still reports the same six +pre-existing diagnostics, with none in these changes. + +The final restart exposed a separate size-related failure: the data process's +whole-database inspection exceeded its ordinary 30-second request timeout twice. +Initialization now has a two-minute deadline, and the development launch probe +allows three minutes. Ordinary requests keep their 30-second deadline. The actual +manager test failed with the old timeout and passes with the longer startup +budget; it also verifies that integrity errors propagate and an unresponsive +initialization still times out. The native app reached its ready state with this +change. Startup recovery checks remain enabled. Recovery-copy work can still +block the main thread on this corpus; making that work responsive remains open. + +### C. Keep the library useful as new things arrive + +**Collection browsing evidence (2026-09-30):** Library now has Resources and +Collections entry points. The collection view uses the existing social schemas +and local query hook, keeps each membership separate, and opens the same cached +source cards and details as resource search. Both lists paginate at forty rows. +Unindexed members remain visible. Source-reported counts and actual local +membership counts have separate labels. + +In an isolated Electron development profile, a manual run paged through 41 +collections and a 42-entry playlist, searched for a collection, retained a +missing-source placeholder, preserved repeated entries, and opened resource +details. The fixture inserted memberships in reverse order; the final view +followed their archived sort keys. Nine Library storage tests passed, including +bounded card requests, missing entries, and omission of full transcripts and +notes from card responses. The broader collection/creator/neighborhood item +stays open; this does not prove full-corpus import or sharing. + +- [ ] Add Library entry points for Inbox, Resources, Collections, and Guides using existing social records and Pages, without making a blank Page for every import. +- [x] Add URL-plus-note capture in the shared workbench, with optional excerpt, duplicate handling, and failure-safe input retention. +- [ ] Add the explicit desktop capture shortcut; confirm focus returns and saving works offline. +- [ ] Link new captures to existing imported resources when they match, preserving independent personal notes and source provenance. +- [ ] Create two guide drafts from rediscovered sources; keep citations connected to the imported resource and its original URL. +- [ ] Verify search over new captures, imported/enriched text, transcripts, URLs, and Page bodies after restart; make each result open the intended item or timestamp. + +**Implementation evidence (2026-09-29):** The shared workbench now has a Save a link form with a URL, title, personal note, and optional excerpt. A native capture intent is persisted before source/Page writes. Retrying the same request reuses those records and preserves any later Page edits. The note is an ordinary private Page whose optional `sourceResources` relation cites the source. Known YouTube, Instagram, and Twitter/X aliases use the import adapters’ resource IDs; an existing Library URL can also resolve a saved GitHub source. Garden/GitHub overlap before Library scanning still needs the broader reconciliation check above. + +Seven capture tests cover retry identity, a failed body write and reopen, interrupted completion, preservation of later edits, reuse without changing source evidence, and refusal to index an unsupported body format as an empty note. Together with Library store/provider, Page schema, and navigation checks, the focused run passed 28 tests. A real Electron run in an isolated profile saved while offline, rendered the original URL, note, and excerpt in the Page editor, quit, reopened offline, found the source by a phrase in its note, and recognized a second capture of the same video. A subsequent run edited both the Page title and body, quit, reopened offline, and found the later body text; the edited title also persisted. It exposed and fixed a stale Lamport clock when the renderer edits a record written by the native importer. Local change allocation now reads the persisted clock, and native notifications refresh renderer subscribers. A focused data/storage/Library run passed 1,915 tests with one opt-in benchmark skipped; a separate commit check passed 2,962 tests with one skipped. The run also confirmed shortcut registration. Actual switching from another app and returning focus remains unverified, so that checkbox stays open. At that capture validation stage, unsaved form drafts remained outside native checkpoints; the later logical settings recovery evidence above extends coverage to drafts at completed checkpoints. A submitted capture’s durable intent and saved Page are covered. The two-guide and publication checks remain open. + +**Exit:** Chris saves a resource during normal browsing and later finds it using a phrase from the note. + +### D. Share something useful + +- [ ] Wire selected guide snapshots and selected assets into the existing static renderer; reject missing snapshots and private dependencies. +- [ ] Add preview and configurable export into a managed directory, with atomic replacement and stale-page cleanup limited to that directory. +- [ ] Try the proposed garden destination through its existing deployment pipeline; make export and publication status accurate. +- [ ] Share two reviewed guides and collect feedback on whether each reader understood why the resource was useful. + +**Exit:** A friend or client opens a useful guide on a phone without logging in. The page contains only what Chris approved. + +### E. Let use decide what comes next + +- [ ] Trial an optional selected-source helper with citations and a no-answer case; leave manual writing fully useful without it. +- [ ] Run a four-week founder trial after passes A through C; keep a short friction log without adding in-app streaks or scores. +- [ ] Ask two other people who collect and share resources to save, find, and share their own material; record where they need help. +- [ ] At review, decide whether to improve this loop, add a proven missing capability, or withdraw the direction; reconcile the roadmaps if continuing. + +**Exit:** There is evidence of voluntary use and useful output, or a clear reason to change course. + +## Validation checklist: proof that matters + +Recovery and usefulness need different evidence. The first requires controlled failures. The second requires real work. Passing a unit test cannot certify either whole experience. + +The release consumer is the existing Electron release workflow and the person promoting its artifact. Its pass condition is concrete: the same workspace content and identity survive a supported upgrade, and every failed migration preserves a recoverable original. Exercise the installed Mac app as well as pure storage tests. Keep deterministic failure fixtures isolated; any new scanner or gate script must include the repository's required in-memory negative-control selftest. + +| Validation | Required observation | +| ------------------------------------------ | ----------------------------------------------------------------------------------------- | +| Restart with real note content | Titles, URLs, body text, attachments, and identity match | +| Last-moment edit before update | An acknowledged saved edit exists after restart | +| Legacy and unknown database versions | No deletion; known old versions migrate; future formats refuse writes | +| Corrupt database or unavailable key | Original bytes remain; explicit recovery state; no fresh identity masquerading as success | +| Disk full during save or backup | No false saved/backup success; previous good recovery point retained | +| Kill during migration or promotion | Startup selects a complete generation or recovery state, never a half-migrated store | +| Incomplete export | Omitted blob/Yjs port or missing attachment fails completeness checks | +| Restore on a clean profile or another Mac | Notes and ownership recover with the recovery kit, without the original Keychain | +| Failed update followed by new notes | Recovery keeps those notes or clearly preserves them for later replay | +| Development alongside daily use | Distinct resolved data paths; sandbox cannot sync or publish | +| Body-only search after restart | Correct source note found without opening every Page by hand | +| Public snapshot missing | Export refuses rather than publishing the live draft | +| Private link and unselected asset in draft | Nothing private crosses the export boundary; preview explains omissions | +| Second export after unpublish | Managed output removes the page; unrelated garden files stay intact | + +Import validation must use the shapes in `.exports/` as well as small synthetic edge cases: + +| Import case | Required observation | +| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ | +| Known archive counts | Every selected source record reconciles; raw record counts, unique resources, and memberships remain separate measures | +| Instagram nested collections | Named collections and all available nested memberships survive; collection metadata is not turned into fake saved posts | +| Wrapped Instagram likes | The parser accepts the observed wrapper or reports it unsupported; it never reports zero successful records for an unread file | +| YouTube catalog and membership files | All 33 catalog rows and 32 membership files are accounted for, including repeated videos and unexplained gaps | +| Twitter likes without bookmarks or dates | Likes remain likes; missing bookmark coverage and save timestamps are visible | +| Same resource saved and liked | One resolved resource exposes both actions and all source records; notes remain intact | +| Reimport, changed ordering, overlapping exports | Stable items and memberships do not multiply; fresh observations retain provenance; no inferred deletions | +| GitHub repository rename or unavailable repo | Stable IDs preserve identity and URL history; missing metadata remains explicit | +| Interrupted or rate-limited GitHub fetch | The snapshot remains partial, resumes or retries safely, and never clears unseen stars | +| Restart or cancellation during import | Only committed chunks advance the checkpoint; progress never claims an incomplete run finished | +| Restore without `.exports/` | Imported data, retained evidence, memberships, checkpoints, and personal notes remain available | +| Cross-source question | The answer names the evidence for each connection; sparse titles never stand in for unseen video or article content | + +Enrichment must pass its own checks before the Library claims useful coverage: + +| Enrichment case | Required observation | +| ---------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------- | +| Never-opened resource beyond the first query page | Metadata, thumbnail, and applicable transcript jobs run without scrolling its card into view | +| Title succeeds, description fails | Field coverage remains partial; a resolved title does not hide the missing description | +| Provider throttling, missing credentials, or malformed caption payload | Retry/actionable failure, never a false no-captions or all-done result | +| Non-English or automatic caption track | Correct language and generation source retained; an English miss does not end discovery | +| Video without usable captions but with accessible audio | Local transcription yields indexed text and timestamps, or reports an explicit failure/partial result | +| Instagram written caption and speech differ | Both remain separate and searchable; neither is substituted for the other | +| Long transcript or a single oversized cue | Text beyond preview/index-field limits remains searchable; no dropped tail or duplicated stale segments | +| Expired thumbnail URL or offline restart | Cached image still renders; a missing image displays a labeled fallback and accurate coverage | +| Invalid image response or oversized asset | Download is rejected safely; the prior good image remains available | +| App update during enrichment | Completed work survives; pending jobs resume without repeating every provider request | +| Restore without provider credentials or network | Stored descriptions, complete transcripts, and thumbnail blobs remain usable | +| Full-corpus report | Unique-resource totals reconcile by capability, including incomplete, deferred, and unsupported work | + +- [ ] Use fixtures from the previous installed release and the oldest supported storage version, plus unversioned and future-version fixtures. +- [ ] Add sanitized fixtures for the observed social archive shapes and GitHub star snapshots; prove fidelity, source reconciliation, idempotence, cancellation/resume, and private defaults. +- [ ] Run a read-only dry run against the actual selected archives, followed by a recoverable full import after pass A; record counts, time, storage growth, and every unsupported category locally. +- [ ] Test metadata, thumbnail, and transcript providers with deterministic failure fixtures, then verify live behavior on a representative sample of Chris's links; separate network limits from parser defects. +- [ ] Prove full-corpus scheduling, field-level coverage, durable retry, offline thumbnail rendering, transcript retrieval, and backup/restore of fetched content in the real desktop app. +- [ ] Run storage and bundle tests that exercise the failure cases above, including a deliberately incomplete backup that the verifier rejects. +- [ ] Drive the real packaged Mac app through archive import, graph browsing, capture, restart, upgrade, rollback/recovery, and external-backup restore. +- [ ] Verify the complete publishing path in a clean browser session and at a phone viewport. +- [x] Record actual command results and manual observations alongside completed checklist items; leave unknowns unchecked. + +For the founder trial, look for use on at least eight of ten days when Chris naturally does relevant reading or sharing. This is a research measure, not a demand to manufacture daily activity. Ask Chris to find five saved resources from remembered context, including an old social save and a starred repository, aiming for under 30 seconds each. Trace three useful connections between imported resources and garden notes, with source evidence. Produce and share two useful guides. Record each update that requires a terminal, loses context, or creates doubt about stored data; those failures outrank adding features. + +For the two outside authors, success means completing save → find → share with their own material and little assistance. Reader feedback alone cannot establish demand for an authoring app. If Chris still prefers existing tools after the trial, inspect which step failed before adding more AI or a larger import pipeline. + +### Final preview checks (2026-09-29) + +| Check | Observed result | +| ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------- | +| `pnpm exec vitest run --project unit packages/data/src --project electron apps/electron/src/storage apps/electron/src/library apps/electron/src/renderer/shell/desktop-platform.test.ts` | 1,915 passed; one opt-in benchmark skipped | +| `pnpm turbo run typecheck` | 101 tasks successful | +| `pnpm exec tsc -p apps/electron/tsconfig.node.json --noEmit` | Passed | +| `pnpm --filter xnet-desktop build` | Passed | +| Changed-file ESLint | No errors | +| Source Electron capture and offline Library runs | Passed as described above; isolated temporary profiles, not the signed installed application | +| Standalone `tsconfig.web.json` check | Still fails on existing deep-link, FormView, header ref, plugin cleanup, and IPC interface typing; not covered by root Turbo | + +## Risks and decisions still open + +The largest risk is spending months on a general backup platform before saving a useful note. Keep pass A focused on the desktop paths that already exist, with one complete checkpoint format, one portable recovery path, and one installed upgrade test. These are justified by observed failure paths. A new multi-device service, background research fleet, or universal ingestion system is outside scope. + +Storage cost may make the proposed checkpoint cadence too expensive if unchanged archive blobs are copied repeatedly. Measure the actual seed corpus and its imported graph, and share immutable blobs across recovery points. A cap can change retention, but it must not quietly weaken the displayed recovery promise. The requested social corpus and its source evidence are core backup scope. Larger media categories need an explicit inclusion decision and a clear coverage report. + +The export shapes will change. Preserve the archive fingerprint and parser version so later repairs can reprocess retained records. The observed Twitter archive does not establish bookmark coverage, and sparse YouTube records cannot supply absent content. The GitHub snapshot covers the authenticated account's currently accessible stars. Report each gap without inventing relationships or asking Chris to reorganize the collection by hand. + +The signing configuration, supported historical database versions, and complete key inventory need implementation-time confirmation. Resolve them at the start of pass A. Do not paper over uncertainty with “backup successful.” An off-device destination and recovery-secret storage also require Chris's choice during setup; no service is chosen or provisioned here. + +The garden export destination remains a proposed default. The library and public renderer should work if Chris chooses a different site. Private collaboration, a coaching portal, full-content clipping, mobile authoring, automatic topic inference, and a new business model wait for evidence from this loop. The source-backed graph of saved items, collections, creators, and notes is part of the first library. + +**Recommended next work:** verify an installed upgrade and restore the enlarged corpus before relying on it for irreplaceable notes. Let the existing enrichment queue run, review its recorded gaps, and close the caption and navigation checks in pass B. Full website import and overlapping-export reconciliation remain open. Use the library in ordinary work before expanding publishing or AI. diff --git a/docs/explorations/0467_[_]_DURABLE_LIBRARY_CONTENT_AS_XNET_NODES.md b/docs/explorations/0467_[_]_DURABLE_LIBRARY_CONTENT_AS_XNET_NODES.md new file mode 100644 index 000000000..24304304a --- /dev/null +++ b/docs/explorations/0467_[_]_DURABLE_LIBRARY_CONTENT_AS_XNET_NODES.md @@ -0,0 +1,266 @@ +--- +title: Durable Library content as xNet nodes and referenced blobs +status: draft +last_updated: 2026-10-03 +review: 2026-10-24 +decider: Chris Smothers +door: one-way +tags: [personal-library, storage, nodes, blobs, sync, migration, durability] +--- + +# Durable Library content as xNet nodes and referenced blobs + +> [!TIP] +> Make every saved Library result recoverable from xNet nodes and their referenced blobs. Keep fast local indexes and the durable enrichment queue, but remove their role as the only home for collected content. Prove this by rebuilding the Library offline in a fresh profile without the original `library.db`. + +## Problem statement + +The personal Library in [exploration 0466](./0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md) is collecting titles, descriptions, GitHub READMEs, thumbnails, and available captions from imported links. Chris wants to use that collection every day while continuing to develop xNet. A rebuild, database change, or move to another device should preserve the work already invested in fetching and organizing it. + +Imported resources and relationships are already xNet nodes. Enrichment is only partly represented there: bounded display metadata and transcript passages reach NodeStore, while full provider results remain in a desktop SQLite database. Native recovery includes that database, but nodes alone cannot reconstruct the enriched Library. + +The question is therefore about **which records own the content**, not replacing SQLite. xNet nodes and blobs already use SQLite on desktop. The goal is one durable content model, with local tables for efficient browsing and background work. + +## Executive summary + +The recommended design has three parts: + +1. **Nodes describe resources and observations:** identities, relationships, titles, authors, coverage, provenance, and references to saved content. +2. **Immutable blobs preserve larger payloads:** full text, README bodies, normalized and original captions, provider evidence, images, and retained archives. +3. **Device-local tables serve the app:** search, card projections, graph caches, leases, retries, and scheduling. Content projections can be rebuilt. Operational queue state remains durable and must survive updates. + +Existing resources keep their IDs. User writing stays separate from provider output. Backfill uses already saved bytes; it does not refetch the internet. The old Library database and backups remain intact until reconstruction has been proven. + +**Status:** proposal only. No storage migration is performed by this exploration. The ongoing enrichment pass can continue on the existing implementation. + +The review date allows three weeks to decide on the first migration slice and its evidence. It is not a delivery promise. `door: one-way` reflects the persistent schema and payload-format commitments involved. Before implementing those commitments, record an ADR with a `Tripwire:` in the [canonical decision log](../../site/src/content/docs/docs/architecture/decisions.mdx). Names and example formats below remain provisional until that decision. + +## Current state in the repository + +The development workspace lives under the Electron profile's `xnet-data` directory, separate from the checkout and application build. Recovery generations live beside it in `xnet-recovery`. The source-build profile is distinct from the installed application's profile; opening a different profile does not migrate data between them. + +| Data | Current home | Status | Consequence | +| ---------------------------------------------------------------------- | ------------------------------------------------------------------- | ------------------------------------ | --------------------------------------------------------------------------------------------------------------------- | +| Imported links, posts, collections, memberships, and notes | NodeStore in `data.db` | ✅ Canonical nodes | Preserve existing IDs and source relationships. | +| Display enrichment | `SocialEnrichment` nodes | 🚧 Bounded projection | Title is capped at 1,000 characters and description at 5,000. Full results cannot be reconstructed from these fields. | +| Full metadata, README text, provider evidence, caption tracks and cues | JSON resource payloads in `library.db` | 🚧 Desktop-owned content | Valuable content currently depends on this additional database. | +| Transcript passages | `SocialContent` nodes plus full caption payload in Library | 🚧 Both representations exist | Preserve timing, language, raw evidence, and passage relationships through migration. | +| Downloaded thumbnails | Content-addressed bytes in the data-process blob table in `data.db` | ✅ Local bytes saved | A CID reference still needs proven transfer, authorization, and retention behavior. | +| Other editor attachments | Main-process blob store in `xnet.db` | ✅ Separate existing store | A recovery manifest must cover both stores; this proposal does not assume they are interchangeable. | +| Original ZIP/JSON exports | Retained, hash-verified files in `import-sources/` | ✅ Exact local copies | Include all retained bytes in recovery, even when only some archive categories were imported. | +| Search and card data | FTS5 and resource tables in `library.db` | 🚧 Mixed with authoritative payloads | Separate rebuildable views from content ownership. | +| Work attempts, cooldowns, capture intents | `library.db` | ✅ Durable local state | Preserve restart behavior and unfinished user saves during migration. | +| Native recovery | Verified copies of `xnet-data` | ✅ Local coverage | Includes Library and archives; remains on the same disk. | + +An aggregate observation on 2026-10-03 found 55,973 Library resources. The primary database was about 7.8 GiB, Library about 3.6 GiB, and ten retained archives about 0.63 GiB. These are changing file sizes, not a permanent capacity estimate or a claim that all resources are fully enriched. No archive contents or private conversations are reproduced here. + +```mermaid +flowchart LR + Import[Imported resources and relationships] --> Nodes[NodeStore in data.db] + Fetch[Provider retrieval] --> Library[Full results in library.db] + Library --> Summary[Bounded enrichment summaries] + Summary --> Nodes + Library --> Passages[Transcript passage nodes] + Passages --> Nodes + Fetch --> Images[Thumbnail blobs in data.db] + Library --> Search[Local search and card views] + Nodes --> Recovery[Native recovery generation] + Library --> Recovery + Images --> Recovery + Archives[Retained archives] --> Recovery +``` + +
+Code paths inspected for this proposal + +- [Library service](../../apps/electron/src/library/service.ts): `project` writes bounded node fields; `retain` saves full results locally; `execute` handles metadata, blobs, captions, and attempts. Successful projections use signed deterministic node imports. +- [Library store](../../apps/electron/src/library/store.ts): resource JSON, FTS5, versioned work records, persistent provider pauses, and capture intents. It uses WAL and `synchronous=FULL`; interrupted jobs reopen queued. +- [Shared Library types](../../apps/electron/src/shared/library.ts): metadata evidence, per-field coverage, caption cues, language, and provider provenance. +- [Enrichment schema](../../packages/social/src/schemas/enrichment.ts): stable platform/content identity and bounded fields, including a thumbnail CID stored as text. +- [Transcript node construction](../../packages/social/src/transcripts/nodes.ts): passage representation and source relationships. +- [Data service](../../apps/electron/src/data-process/data-service.ts): node persistence, blob reads/writes, and existing peer blob exchange. +- [File properties](../../packages/data/src/schema/properties/file.ts): existing typed file references with CID, MIME type, byte size, and optional thumbnail reference. +- [Archive storage policy](../../packages/social/src/import/storage.ts): existing single-archive, selected-entry, and manifest-only modes. These modes are not proof of byte-for-byte archive recovery. +- [Retained sources](../../apps/electron/src/storage/import-sources.ts) and [import journals](../../apps/electron/src/storage/import-journal.ts): exact input preservation and acknowledged progress. +- [Native recovery](../../apps/electron/src/main/recovery.ts), [checkpoints](../../apps/electron/src/storage/checkpoints.ts), and [migration staging](../../apps/electron/src/storage/migrations.ts): writer barriers, database normalization, verification, and preserved originals. +- [Storage inventory](../reference/desktop-storage.md): current coverage, identity requirements, portable exports, and known exclusions. + +
+ +## External research + +SQLite's online backup API supports consistent database snapshots and incremental copying. This is useful for a migration snapshot, but a snapshot of one database does not establish consistency across several stores and files. xNet still needs its workspace-wide writer barrier and manifest verification. [SQLite backup API](https://www.sqlite.org/backup.html). + +SQLite WAL durability depends on the journal and synchronization settings; the main database file alone may not contain the latest committed state. Preserve the existing recovery path instead of copying a live `.db` file with an ordinary file-copy command. [SQLite WAL documentation](https://www.sqlite.org/wal.html). + +IPFS distinguishes identifying content from retaining it. Its pinning model illustrates why a content address by itself is insufficient: collection rules and retained roots determine whether bytes remain available. The relevant lesson for xNet is explicit ownership and reachability. This is not a proposal to introduce IPFS or a paid pinning service. [IPFS persistence documentation](https://docs.ipfs.tech/concepts/persistence/). + +## Options and tradeoffs + +| Option | Strength | Cost | Decision | +| --------------------------------------------------------------------- | ------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ----------------------------- | +| Keep full results only in Library and back up both databases | Smallest immediate change; works locally today | Desktop-specific content ownership; nodes cannot restore the collection | 🟡 Transitional only | +| Put complete payloads directly in node properties | One obvious record model | Large changes, repeated payloads in history, existing field limits, expensive common reads | 🔴 Avoid for large payloads | +| Store each provider result as a versioned blob referenced by a node | Complete evidence, small node updates, byte deduplication, bounded reads | Requires reliable reference enumeration, retention, transfer, and migration | 🟢 Recommended | +| Keep full results in an unrelated file directory or external database | Flexible payload storage | Another independent recovery and sharing contract | 🔴 No additional store needed | + +The node-and-blob option still needs a local SQL projection for large collections. The recent browsing fix added ordered indexes and stopped overlapping polling; moving ownership must preserve those gains. A Library card request should not parse every README or caption track. + +## Proposed content model + +### Nodes carry meaning and provenance + +Keep the existing imported resource and collection nodes. Extend the enrichment model additively, or introduce a closely related observation schema if its versioning and authorization fit better. Choose between those alternatives in the ADR; do not silently replace existing schema IDs. + +An observation identifies its resource, provider and extractor version, retrieval time, format version, per-field coverage, and typed content references. It distinguishes full text, previews, authored captions, retrieved spoken caption tracks, and future local transcription. A failed retrieval may create an explicit outcome record; it must not overwrite a successful payload or pretend that unavailable text was collected. + +User titles, notes, annotations, and tags have separate ownership. A refresh cannot replace them. Provider-generated labels remain distinguishable from an authored video title. The effective display value can be computed from an explicit user override and the selected source observation. + +### Blobs hold complete payloads + +| Blob role | Required content | +| ----------------- | --------------------------------------------------------------------------------------------- | +| Normalized text | Complete description, README, or article text; format and encoding | +| Provider evidence | Saved response or extracted evidence actually available, with its fidelity labeled | +| Caption track | Every cue with timing, language, automatic-caption flag, and original response where retained | +| Thumbnail | Original verified raster bytes and content type | +| Source archive | Exact retained archive bytes, hash, and import provenance | + +Use xNet's existing content-ID implementation. Do not invent a competing CID or infer that an archive's existing SHA-256 fingerprint is already an xNet CID. Prefer the typed file-reference mechanism where suitable. References hidden only inside arbitrary JSON will not be enough unless backup, transfer, authorization, and garbage collection can enumerate them. + +For large archives, define a bounded chunk manifest if existing transport and memory limits require one. A manifest must account for every retained byte, preserve order and lengths, and verify the reconstructed archive hash. Keep the current retained file until that round trip is proven. Do not load a large ZIP into one node property or assume the existing selected-entry policy preserves the whole ZIP. + +Payload identity should exclude incidental retry counters and scheduling data. Re-fetching identical content should reuse its blob and semantic revision. Retrieval observations can record that it was seen again without manufacturing another full content copy. + +```mermaid +erDiagram + RESOURCE ||--o{ OBSERVATION : has + OBSERVATION ||--o{ BLOB_REFERENCE : retains + BLOB_REFERENCE }o--|| IMMUTABLE_BLOB : identifies + RESOURCE ||--o{ USER_NOTE : contextualized_by + COLLECTION ||--o{ MEMBERSHIP : contains + RESOURCE ||--o{ MEMBERSHIP : belongs_through + ARCHIVE ||--o{ BLOB_REFERENCE : preserves +``` + +### Local tables remain useful + +Search indexes, bounded cards, graph layouts, and parsed payload caches can be reconstructed from nodes and blobs. The 3D graph remains a derived view of stored relationships and metadata; it does not require a separate graph database. + +Queue leases, retries, attempt logs, and provider cooldowns remain local operational state. They must survive normal restarts and upgrades, even though they do not define the saved collection. Losing a cache must not reset a live queue or trigger a flood of provider requests. A fresh profile should rebuild from saved observations first; any later network scheduling must respect known outcomes and apply conservative pacing. + +Unfinished captures need separate treatment: their original text and write intents are not disposable. Preserve and reconcile them before describing any portion of `library.db` as safe to remove. + +## Write, refresh, and read behavior + +```mermaid +sequenceDiagram + participant Worker as Enrichment worker + participant Blobs as Durable blob storage + participant Nodes as NodeStore + participant Cache as Local Library projection + Worker->>Blobs: Write payload and verify content ID + Blobs-->>Worker: Durable acknowledgement + Worker->>Nodes: Commit observation and references + Nodes-->>Worker: Durable node acknowledgement + Worker->>Cache: Update card and search data + alt Projection interrupted + Nodes-->>Cache: Replay committed observation after restart + end +``` + +The durability acknowledgement is after both bytes and their owning record exist. A crash after the blob write can leave an unreferenced object; a crash after the node commit can leave an outdated index. Both are repairable. A referenced but missing blob is an explicit incomplete result, never an empty successful description. + +Use an observation or manifest reference to select a coherent result. Avoid independently updating several payload CIDs in a way that can merge into a mixture of two revisions. Concurrent device fetches should preserve their provenance, deduplicate identical bytes, and follow a documented selection rule. A stale backfill must not win merely because its migration write has a later Lamport time. Fetch time alone also does not make a truncated preview better than an existing complete result. + +Readers fetch bounded node/card data for browsing and load large payloads for details or background indexing. Missing remote blobs need a visible availability state and resumable transfer. Node sync, local blob availability, remote retention, and off-device backup are separate facts. + +### Sharing and retention + +An imported archive may contain private conversations and unselected categories. Sharing a single link must not grant access to its entire archive or raw private evidence. Blob delivery must enforce the owning record's authorization; possession of a CID is not a permission check. Shared views should reference only the selected content the recipient is allowed to receive. + +Retained roots include live nodes, retained revisions, in-progress commits, and retained recovery generations. Deleting a current observation must not make a still-restorable checkpoint lose its bytes. Existing peer blob exchange is a seam to audit, not proof that the proposed authorization and retention contract already holds. + +The first migration can keep existing native checkpoints unchanged. Moving blobs out of monolithic databases to reduce backup duplication is a separate storage decision. Content addressing alone will not stop full SQLite snapshots from consuming disk. Measure actual disk growth, preserve retention, and require a separate verified change before introducing shared backup object storage. + +## Migration without losing the current collection + +> [!IMPORTANT] +> Backfill from the saved Library and retained archives. Preserve successful retrieval, incomplete results, private records, capture intents, source bytes, identity, and recovery generations. No bulk refetch and no reset of all partial jobs. + +1. **Inventory and checkpoint.** Wait for active recovery to finish. Use the native barrier and verified checkpoint path. Record counts and content hashes by role, existing node IDs, unresolved references, and disk headroom. Keep private inventories out of Git. +2. **Add compatible readers and writers.** New enrichment commits canonical observations and blobs, then updates local projections. Keep legacy reads available until their rows are verified. Unsupported formats fail visibly and preserve the bytes. +3. **Backfill in bounded batches.** Assign deterministic migration identities based on the resource, provider, format, and content digest. Reuse thumbnail blobs and existing transcript passages where possible. Persist a migration cursor only after acknowledged writes. Preserve exact stored evidence rather than claiming it is a raw response when it is not. +4. **Reconcile concurrent work.** Begin from a consistent snapshot, track row content digests, and catch results written during the backfill. Do not rely on an offset through mutable rows as the sole completion proof. Use a short final writer barrier for reconciliation; never freeze the app for the whole corpus. Old app versions that cannot respect the transition must be detected rather than silently continuing legacy writes after cutover. +5. **Compare representations.** Verify every migrated field, payload digest, thumbnail reference, caption cue and relationship. Report mismatches and explicit source gaps separately. A nonempty title is not evidence that the full description or transcript survived. +6. **Prove reconstruction.** Use a separate profile with an authorized copy of the same workspace identity, no legacy Library content database, no access to the original archive paths, and source-network access disabled. Build views entirely from canonical records and referenced bytes. +7. **Switch the content reader.** Promote the verified canonical path. Preserve rollback data. Removing duplicate legacy payloads, deleting archives, or changing backup retention is outside the initial migration. + +```mermaid +stateDiagram-v2 + [*] --> Inventoried + Inventoried --> Protected: Verified checkpoint + Protected --> Backfilling: Compatible writes enabled + Backfilling --> Backfilling: Resume acknowledged batches + Backfilling --> Reconciled: Concurrent changes accounted for + Reconciled --> Verified: Offline reconstruction matches + Verified --> CanonicalReads: Reader cutover + Backfilling --> PreservedFailure: Missing bytes or rejected write + Reconciled --> PreservedFailure: Inventory mismatch + PreservedFailure --> Backfilling: Repair and resume +``` + +## Recovery contract and limits + +The acceptance claim is **content completeness from nodes plus all referenced bytes**. It includes source evidence and user writing, not just a similar-looking list of cards. It does not imply recovery of OS-bound sign-in sessions, arbitrary plugin state, or settings outside the desktop recovery contract. + +Native backups already include `library.db`; keep that protection during transition. Local copies remain on the same disk. Portable encrypted exports exist, but this proposal neither configures an off-device destination nor claims that node sync is a backup. Local application data and native recovery copies are not application-encrypted; changing their encryption model is separate work. + +Local ASR remains unconnected. Preserving written Instagram captions or provider previews does not create spoken transcripts. Existing unavailable, blocked, and partial outcomes must remain distinguishable after migration. + +## Risks and open questions + +| Question | Recommended starting position | Evidence needed | +| --------------------------------------------------------- | -------------------------------------------------------------------------------- | --------------------------------------------------------------- | +| Extend enrichment nodes or add observation nodes? | Keep existing summaries compatible; prototype immutable observations | Concurrent refresh and old-client fixtures | +| One payload blob or separate text/evidence/caption blobs? | Separate large roles when it permits bounded retrieval; keep a coherent manifest | Size, transfer, and reindex measurements | +| How much source history should be retained? | Preserve everything already collected during this migration | Later explicit retention policy with checkpoint reachability | +| Can existing blob transport handle all roles? | Reuse it only after auditing limits and authorization | Cross-profile transfer, missing-object and denied-access checks | +| Will the migration fit on disk? | Budget source, new blobs, journals, and recovery copies before starting | Measured peak growth on a representative copied workspace | +| What does a downgrade do? | Preserve originals and reject incompatible writes explicitly | Supported-version and unsupported-version restart exercises | +| Should full archives sync to every device? | Preserve them as private owned objects; make download policy explicit | Offline restoration and selective-sharing evidence | + +## Implementation checklist + +All items are proposed work; no migration is marked complete here. + +- [ ] Write the ADR for observation identity, payload formats, reference ownership, compatibility, and a `Tripwire:` that reopens the design if reconstruction, privacy, or measured sync costs fail its contract. +- [ ] Define additive schemas and strict versioned payload validators, reusing existing file references where appropriate; update seeds and package changesets when implementing. +- [ ] Implement complete reference enumeration, durable blob acknowledgement, hash verification, and ownership/retention roots. +- [ ] Audit and wire authorized, bounded, resumable blob transfer for Library content and archive manifests. +- [ ] Write canonical observations before updating local views; preserve user edits and make concurrent observation selection explicit. +- [ ] Build the idempotent backfill journal and dry-run inventory, including capture intents, private records, and retained archive bytes. +- [ ] Add reconciliation for results arriving during backfill and a compatible recovery/downgrade path. +- [ ] Rebuild card, search, transcript, and graph projections from canonical content without network requests. +- [ ] Expose separate content, local-availability, migration, and retrieval-coverage states without labeling missing content complete. +- [ ] Update storage and recovery documentation with measured coverage; retain legacy payloads through verified cutover. + +## Validation checklist + +- [ ] A fresh isolated profile reconstructs the full saved Library from nodes and blobs with provider networking disabled and the old Library database unavailable. +- [ ] Every payload and retained archive matches its inventory hash; caption timing/language, relationships, notes, and coverage states match, not only total counts. +- [ ] Repeating backfill adds no duplicate semantic observations, transcript passages, blobs, or collection memberships. +- [ ] Crash injection before/after blob commit, node commit, cursor acknowledgement, and projection write preserves data and resumes deterministically. +- [ ] Refresh and migration races preserve newer complete results and all user edits; simultaneous device observations cannot assemble mixed manifests. +- [ ] Missing/corrupted payloads, unknown versions, denied access, and disk exhaustion produce explicit incomplete/error states while preserving originals. +- [ ] A selective share can fetch its authorized payloads but cannot fetch a private archive or unrelated evidence, even when its CID is known. +- [ ] Retained recovery points restore their referenced blobs after later edits/deletions; successful node synchronization alone is never reported as complete backup coverage. +- [ ] Compare large-library card/search latency, memory, signed-history growth, sync bytes, and peak disk use against a recorded baseline before cutover. +- [ ] Verify restart, offline details, thumbnail decoding, and timed-caption search in the real Electron app using an isolated profile; never replace the user's identity or enable test bypass in the real profile. +- [ ] A documented reconstruction gate has a named consumer and decisive result: the Library migration release check passes only on full inventory equality. Its in-memory negative controls omit a blob, truncate a caption, and drop a membership; each must fail the check. + +## Recommendation + +Implement a small vertical slice first: one GitHub README, one YouTube caption track, one Instagram written caption, one thumbnail, and one retained archive, each with its resource relationships and explicit coverage. Prove offline reconstruction, interruption recovery, and selective access before backfilling the full collection. + +Keep the current enrichment work and recovery protection operating while preparing that change. Completion means the collected content can outlive the desktop Library implementation: a fresh profile can recover it from xNet nodes and their referenced blobs without depending on the original websites still being available. diff --git a/docs/explorations/0468_[_]_VISION_PRO_SPATIAL_LIBRARY_GRAPH.md b/docs/explorations/0468_[_]_VISION_PRO_SPATIAL_LIBRARY_GRAPH.md new file mode 100644 index 000000000..967cc88ec --- /dev/null +++ b/docs/explorations/0468_[_]_VISION_PRO_SPATIAL_LIBRARY_GRAPH.md @@ -0,0 +1,301 @@ +--- +title: Exploring the Library graph on Apple Vision Pro +status: draft +last_updated: 2026-10-03 +review: 2026-10-24 +decider: Chris Smothers +door: two-way +tags: [personal-library, graph, visionos, webxr, spatial-computing, interaction] +--- + +# Exploring the Library graph on Apple Vision Pro + +> [!TIP] +> Start with a WebXR experiment using the existing Three.js graph. Prove headset rendering and PlayStation VR2 Sense input separately, then add controller-directed flight. If Safari cannot supply tracked controller poses and analog triggers, use a small native visionOS viewer for that experience. + +## What we want to do + +Chris wants to stand inside the Library's three-dimensional link graph, explore its neighborhoods, and inspect the things saved there. A controller becomes a flight instrument: point it through space, squeeze the trigger to accelerate, and look around independently while moving. Search and filters should make a collection of roughly 54,000 links navigable rather than merely impressive to look at. + +Useful moments include finding a saved video, seeing its title and thumbnail, following a playlist into related resources, and discovering an unexpected connection between topics. The graph should explain whether an edge comes from a playlist, creator, hashtag, or another source. Future AI suggestions need their own evidence label; spatial proximity alone must not imply a factual relationship. + +This builds on the [personal Library](./0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md), the [durable content proposal](./0467_[_]_DURABLE_LIBRARY_CONTENT_AS_XNET_NODES.md), and the earlier [immersive recommendation space exploration](./0151_[_]_SELF_ORGANIZING_SOCIAL_GRAPH_IMMERSIVE_RECOMMENDATION_SPACE.md). The earlier immersive proposal supplies product ideas; the code inventory below describes today's implementation. + +**Status: research and design only.** No XR renderer, headset connection, or controller implementation is added here. No physical Vision Pro test has been performed for this exploration. + +The review date gives three weeks to decide whether the hardware experiment justifies further work. It is a decision checkpoint, not a delivery promise. The initial read-only viewer is reversible, so this exploration is `two-way`. A durable public pairing protocol or schema commitment would need an ADR with a `Tripwire:` in the [decision log](../../site/src/content/docs/docs/architecture/decisions.mdx) before implementation. + +## Current state in the repository + +We already own a Three.js renderer. We are not using `3d-force-graph` or `react-force-graph`, and a library replacement is not a prerequisite for XR. + +| Part | Status | Current implementation and implication | +| ------------------------- | ---------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Rendering | ✅ Existing | [scene.ts](../../apps/electron/src/renderer/components/library-graph/scene.ts) batches nodes into `Points` and edges into `LineSegments`. It uses a custom point shader, a perspective camera, and `OrbitControls`. | +| Layout | ✅ Existing | [layout.worker.ts](../../apps/electron/src/renderer/components/library-graph/layout.worker.ts) runs `d3-force-3d` off the main thread and transfers position buffers. The layout can be paused. | +| Scene lifecycle | 🚧 Desktop assumptions | [GraphCanvas.tsx](../../apps/electron/src/renderer/components/library-graph/GraphCanvas.tsx) owns the worker and scene, including automatic framing when layout finishes. XR needs a separate framing policy. | +| Search, groups, details | 🚧 DOM interface | [LibraryGraphView.tsx](../../apps/electron/src/renderer/components/LibraryGraphView.tsx) supplies search, filters, selection, and details. These controls do not automatically become visible in an immersive session. | +| Graph data | ✅ Reusable shape | [library-graph.ts](../../apps/electron/src/shared/library-graph.ts) defines nodes, indexed edges, relationship evidence, counts, and warnings. Positions are not currently part of that snapshot. | +| Data access | 🚧 Electron dependency | The view calls `window.xnet.libraryGraph()` and related native APIs. [library-ipc.ts](../../apps/electron/src/main/library-ipc.ts) wires desktop access; Safari cannot use Electron's preload. | +| Relationship construction | ✅ Source evidence | [graph.ts](../../apps/electron/src/library/graph.ts) derives the graph from Library records. Preserve resource identities and evidence when presenting it elsewhere. | +| XR input and presentation | ⬜ Proposed | No headset session, spatial panels, controller flight, or native visionOS target is established by these files. | + +The desktop [package manifest](../../apps/electron/package.json) pins Three.js `0.186.1` and `d3-force-3d` `3.0.6`. Test changes against those versions before assuming an upgrade is necessary. + +There are several concrete desktop assumptions to remove. Rendering is currently invalidation-driven through ordinary `requestAnimationFrame`; labels are twelve projected HTML elements; picking uses a screen-space pointer; and camera distances are arbitrary graph units. Both points and lines disable frustum culling. These choices can work on a monitor without establishing acceptable headset behavior. + +## What the platforms actually support + +The following separates documented platform capabilities from combinations that still need a device test. Research was checked on **2026-10-03**. + +| Capability | Evidence | Consequence for xNet | +| ----------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------- | +| Immersive WebXR in Safari on Vision Pro | ✅ WebKit documented `immersive-vr` with visionOS 2. [Safari 18 announcement](https://webkit.org/blog/15443/news-from-wwdc24-webkit-in-safari-18-beta/) | A browser-based headset renderer is a credible first experiment. Query session support on the actual device. | +| Look-and-pinch input | ✅ WebKit documents transient pointer input and optional hand tracking. [Natural input for WebXR](https://webkit.org/blog/15162/introducing-natural-input-for-webxr-in-apple-vision-pro/) | Support pinch selection. Do not assume persistent mouse hover or access to a continuous eye-gaze stream. | +| Newer WebXR rendering features | ✅ Safari 27 documents texture-array projection layers. [Safari 27 release notes](https://webkit.org/blog/18325/webkit-features-for-safari-27-0/) | Feature-detect improvements later; they do not remove the need to measure our shaders and scene. | +| Native PS VR2 Sense tracking | ✅ Apple documents spatial accessories in visionOS 26, including PlayStation VR2 Sense controllers. [WWDC25 spatial accessories](https://developer.apple.com/videos/play/wwdc2025/289/) | Native visionOS has a documented route to tracked spatial controllers. | +| Sense poses and analog triggers in Safari WebXR | ❓ Unverified | Native support does not establish browser exposure. This is the first hardware gate, not a promise. | +| WebXR controller button values | ✅ The Gamepads Module defines an optional gamepad on an XR input source. [W3C module](https://www.w3.org/TR/webxr-gamepads-module-1/) | Inspect mapping and capabilities at runtime. A standard's existence does not prove a particular browser/controller combination ships it. | +| Passthrough, immersive AR, and DOM overlays | ❓ Separate probes | Do not derive support from `immersive-vr`. Keep the first web design usable without them. | +| Mac-to-headset spatial preview | ✅ Apple documents USD-based Spatial Preview in its 2026 material. [WWDC26 Spatial Preview](https://developer.apple.com/videos/play/wwdc2026/282/) | An alternative for inspecting a graph exported as a scene, not an automatic bridge for this interactive Three.js canvas. | + +Use the precise controller name: **PlayStation VR2 Sense**. An ordinary DualSense gamepad is a different input device and should not be presented as an equivalent tracked flight controller. + +Native support uses Game Controller for device/input integration and spatial tracking APIs for poses. Apple documents the spatial gamepad declaration and accessory tracking usage description in its [Game Controller updates](https://developer.apple.com/documentation/updates/gamecontroller). A native implementation must handle the corresponding permission and disconnect states, rather than treating a connected button device as a tracked pose source. + +## Options and tradeoffs + +| Option | Reuse | Main cost or uncertainty | Decision | +| -------------------------------------------------------------------------------- | ------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------- | +| **Safari + Three.js WebXR** | Graph model, layout, geometry, filtering logic, much of rendering | Spatial interface, browser data access, and unverified Sense exposure | 🟢 First experiment | +| **Native visionOS viewer** using SwiftUI, RealityKit, Game Controller, and ARKit | Graph snapshots, IDs, layout results, content, interaction specification | A second renderer and client; native build/distribution workflow | 🟡 Preferred fallback when tracked controller flight cannot work on the web | +| **Spatial Preview from the Mac** | Graph data and exported positions | Convert the scene to USD and integrate native Mac APIs; custom flight is not established by the preview documentation | 🟡 Useful separate tabletop inspection experiment | +| **Mac native rendering or streaming** | Mac-side computation and data locality | Metal/Compositor Services or an appropriate streaming pipeline, input return path, latency, and an awake Mac | ⏸ Defer until measurements justify it | +| **Mac Virtual Display alone** | Existing app unchanged | The graph remains inside a flat desktop window | ✅ Convenient existing access, but does not satisfy immersive exploration | + +Apple also describes native and streamed rendering paths in its [visionOS 27 overview](https://developer.apple.com/videos/play/wwdc2026/287/) and a [RemoteImmersiveSpace](https://developer.apple.com/documentation/swiftui/remoteimmersivespace). These deserve a fresh implementation-specific investigation if we choose them. They are not evidence that Electron can send its existing canvas to the headset with a configuration switch. + +For a native viewer, start with RealityKit and batched geometry. Do not create a heavyweight entity, label, and thumbnail for every saved link. Consider a custom Metal renderer only after proving that the simpler scene representation misses the measured budget. Adding Unity or Unreal would introduce a substantial dependency without resolving the initial input and data questions. + +## Architecture: one Library, several ways to view it + +The headset should read a projection of the existing Library. It should not become a second canonical database or run another bulk enrichment pass. + +```mermaid +flowchart LR + Content[Existing Library content and blobs] --> Projection[Graph snapshot and detail projection] + Projection --> Desktop[Electron data adapter] + Projection --> ReadBridge[Paired read-only access] + Fixture[Private offline snapshot] --> Browser[Browser data adapter] + ReadBridge --> Browser + Desktop --> Core[Graph model and scene construction] + Browser --> Core + Core --> Flat[Desktop camera and DOM interface] + Core --> XR[WebXR rig and spatial interface] + ReadBridge -. if needed .-> Native[Native visionOS viewer] +``` + +Extract a small data boundary around fetching a graph snapshot, looking up a resource, searching, and reading a thumbnail. This is a proposed boundary, not an existing browser API. Keep parsing, graph filtering, and flight state transitions as pure functions where practical; adapters own sessions, networking, and rendering effects. + +Snapshots need a revision identifier, stable resource IDs, and a matching layout revision. Edges currently address node-array offsets: a client must never combine one snapshot's edges with another snapshot's positions. Cancel stale detail requests, retain the selected ID across refreshes, and install a new snapshot atomically. Keep saved viewpoints relative to a graph revision or a named resource neighborhood so they can be re-established after layout changes. + +Start the rendering experiment with a small private fixture containing no personal data. Then prove read-only access to the real Library. On Vision Pro, `localhost` means the headset, not the Mac. WebXR requires a secure context, and a secure page cannot be assumed to fetch an arbitrary insecure LAN endpoint. Establish a trusted HTTPS path as part of the connection design. [WebXR Device API](https://www.w3.org/TR/webxr/). + +The connection should use explicit pairing, a revocable read-only capability, and bounded endpoints for graph snapshots, details, and authorized blob reads. Restrict origins and resource scope. Do not expose the development debugger, test-auth bypass, arbitrary SQL, or identity keys. Merely changing a development server's bind address is not the proposed transport. + +For an early offline trial, an explicitly selected private export is also viable. Keep it out of public static hosting and avoid bundling the entire archive corpus. Longer-term independent headset use should follow exploration 0467's node-and-blob ownership work, including actual blob availability. A viewer connected to a Mac is not yet an independently synced Library. + +## Moving the existing scene into XR + +Three.js documents enabling XR, entering a session, and using `setAnimationLoop` in its [VR guide](https://threejs.org/manual/pages/how-to-create-vr-content.html). Its [WebXRManager](https://threejs.org/docs/pages/WebXRManager.html) exposes the session, reference space, cameras, and controller objects. These are the rendering foundation, not the whole feature. + +The renderer needs two explicit lifecycles: + +- **Desktop:** preserve the current on-demand rendering, OrbitControls, DOM panels, and keyboard navigation. +- **Immersive:** run the XR animation loop, suspend OrbitControls, use spatial input and scene-resident labels, and restore desktop behavior cleanly when the session ends. + +Place the XR camera and controller objects under a **locomotion rig**. The runtime still supplies the viewer's tracked head pose. Move the rig for virtual travel; do not overwrite the tracked camera with a controller quaternion. Head movement remains independent of steering. + +```mermaid +flowchart TD + World[World coordinates in metres] --> Rig[Virtual locomotion rig] + Rig --> Head[Runtime-tracked head and stereo cameras] + Rig --> Hands[Runtime-tracked controllers or hands] + World --> GraphRoot[Graph root: origin and display scale] + GraphRoot --> Layout[Stable layout coordinates] + World --> Panels[Pinned spatial cards and navigation] +``` + +Choose a graph-to-metre scale deliberately. Desktop camera distances and force-layout coordinates cannot be interpreted as metres unchanged. Separate graph scaling from locomotion speed. Keep coordinates near a local origin, and transform controller poses exactly once between reference space and world space. + +Freeze the force layout during flight. A worker keeps computation off the render thread, but changing every node's position still causes buffer uploads and makes the world move around the viewer. Build or settle the layout before entry; offer an explicit refresh at rest. The existing automatic fit must never relocate the XR rig when a worker finishes. + +DOM labels, autocomplete, and the detail inspector will not automatically appear inside the immersive render. Add a bounded set of spatial labels and cards. For the earliest experiment, search in the flat page before entry is acceptable, provided it is described as a temporary limitation. A useful follow-up needs an in-headset search panel or an explicit pause-to-search flow; do not assume DOM Overlay support. System text input or dictation is an optional platform integration, not a reason to invent a full virtual keyboard first. + +## Controller flight + +### Point, squeeze, and fly + +The default mode should closely match Chris's request: + +1. Choose a flight hand and enter flight mode explicitly. +2. Point the controller in the desired direction, including up, down, or backward. +3. Squeeze the analog trigger to apply acceleration along that direction. +4. Release it to decelerate; use a separate brake action for an immediate stop. +5. Look around freely while travelling. The other hand can select a link or open navigation. + +**Acceleration is not speed.** Holding the trigger should increase velocity up to a configurable cap. Trigger release should apply drag by default, so letting go reliably brings the user to rest. An optional coast mode can come later. Tune the speed cap against the displayed graph scale, not the raw layout radius. + +Keep the virtual horizon stable in this default mode. Pointing in a new direction changes thrust without forcing the viewer's head or rolling the entire world. Opening a menu or inspecting a result brakes flight. Selecting a card must never double as a throttle command. + +### What full six-degree-of-freedom control means + +Six degrees of freedom means **three axes of position and three axes of rotation**. Orientation alone has three rotational degrees of freedom. A tracked Sense controller can therefore support more than a forward pointing vector, if the client receives its full pose. + +Offer an advanced **full-pose flight mode** as a separate, deliberate choice. A clutch captures a neutral hand pose. Relative hand translation commands sideways, vertical, and forward/backward movement; relative yaw, pitch, and roll command bounded angular rates. The trigger scales translational thrust. Releasing the clutch recenters the control reference so the user does not need to hold an awkward arm pose. + +This makes roll and arbitrary vehicle orientation available without forcing them into the first experience. Rotating the locomotion rig is virtual vehicle motion; the headset's physical head tracking still composes with that rig. Specify the controller reference frame carefully so rotating the vehicle does not feed back into its own steering command. + +| Input | Proposed behavior | Constraint | +| ------------------------- | ---------------------------------------------------------------------- | ----------------------------------------------------------------------- | +| Flight-hand pose | Aim thrust; optionally control all six axes relative to a neutral pose | Full pose must actually be tracked; expose mode and handedness clearly. | +| Analog trigger | Thrust with dead zone and adjustable response curve | Confirm analog range and input mapping on hardware. | +| Brake / clutch action | Stop, or recapture neutral pose in advanced mode | Choose from available non-system controls after the probe. | +| Other-hand ray and select | Inspect a link, pin a card, choose a destination | Use a dedicated interaction role; support swapping hands. | +| Single-controller mode | Explicit switch between navigation and inspection | No simultaneous ambiguous trigger action. | +| Hands only | Pinch selection, graph manipulation, destination jumps | Remains useful without imitating a nonexistent analog trigger. | + +Do not reserve platform Home or other system gestures. Haptics may confirm selection if the chosen API supports them; adaptive trigger effects are not a requirement. + +
+Frame-independent flight integration and coordinate rules + +An illustrative default-mode model, with all quantities expressed in an agreed world frame: + +```text +u = responseCurve(deadZone(trigger)) +forward = rotate(rigRotation * controllerReferenceRotation * aimCalibration, [0, 0, -1]) +vNext = clampLength((velocity + maxAcceleration * u * forward * dt) * exp(-drag * dt), maxSpeed) +pNext = position + vNext * dt +headWorldPose = rigTransform * runtimeHeadPose +``` + +The controller quaternion above is in XR reference space. If an adapter has already supplied a world-space aim vector, do not multiply the rig transform again. Use the API's target ray for pointing and grip pose for a physical hand-relative flight instrument; calibrate their difference rather than assuming they are interchangeable. + +Clamp elapsed time after a suspended frame and reset accumulated velocity on session interruption. Treat unavailable or non-finite poses as a stopped state, not a zero-position input. Advanced-mode angular rates need their own cap, damping, and neutral-pose transform; they are not implemented by this translational formula. + +Keep the update function deterministic over state, sampled input, and `dt`. Test equivalent elapsed time at several frame rates, dead zones, braking, speed limits, coordinate conversion, and loss of tracking without needing a headset. + +
+ +### Input loss and comfortable movement + +```mermaid +stateDiagram-v2 + [*] --> Inspecting + Inspecting --> Armed: Explicit flight choice and valid tracking + Armed --> Flying: Fresh trigger press + Flying --> Armed: Release and decelerate + Flying --> Inspecting: Brake or open panel + Armed --> Stopped: Tracking or session interrupted + Flying --> Stopped: Tracking or session interrupted + Stopped --> Inspecting: Tracking returns and trigger released + Inspecting --> [*]: Leave session +``` + +On disconnect, lost tracking, hidden session, or an invalid pose, stop both translation and rotation immediately. Require trigger release and a fresh arming action before movement resumes. Avoid a sudden jump after reconnecting a controller or returning from a system panel. + +Provide a stationary overview, a recoverable home viewpoint, and explicit destination jumps from the beginning. Smooth flight is optional. Advanced roll and pitch need separate opt-in, conservative limits, and an easy exit. The design follows Apple's emphasis on comfortable, user-controlled immersive motion; actual comfort still needs personal testing. [Immersive experiences](https://developer.apple.com/design/human-interface-guidelines/immersive-experiences/), [motion](https://developer.apple.com/design/human-interface-guidelines/motion). + +## Finding things while inside the graph + +Use three complementary views: a compact overview that can be rotated and scaled, a focused neighborhood around a selected resource, and flight through the larger graph. A virtual tabletop in WebXR is not a promise that it appears anchored to a real table through passthrough; a native shared-space presentation is a separate platform option. + +Autocomplete should highlight a result and its neighborhood first. **Travel there** is a separate action, with a controlled transition and a return breadcrumb. The desktop's current select-and-focus behavior must not automatically become headset camera travel. + +Filters should retain stable resource IDs, show matching counts, and explain collapsed neighborhoods. Category, platform, creator, playlist, and tag controls can all operate on the same graph projection. Keep the current view stable while choosing a filter, then make changes explicit. A cluster summary must disclose that it represents more links than are individually drawn. + +For inspection, a tracked controller can provide a ray and intentional hover; pinch can select a result without requiring hover. WebKit's transient input sources can appear only during a gesture, and hand-tracking sources can coexist with them. Iterate and classify sources rather than hard-coding the first two array entries. [WebKit input model](https://webkit.org/blog/15162/introducing-natural-input-for-webxr-in-apple-vision-pro/). + +Pin a small card near the selected resource with its thumbnail, title, description, source, tags, memberships, and relationship evidence. Offer longer README text or transcript content on demand. Show retrieval coverage: a preview is not full content, and a written caption is not a spoken transcript. Fetch thumbnail bytes through the authorized data adapter and retain useful placeholders when enrichment is unavailable. Avoid a wall of constantly head-locked panels. + +## Rendering a large Library + +The batched points and lines are a useful starting point. However, acceptable desktop performance is not evidence of acceptable stereo rendering, input latency, or thermal behavior. + +Start at 1,000 synthetic links, move to 10,000, then test a representative graph near the Library's current size. Keep the whole collection addressable while drawing a useful subset of labels, detailed nodes, thumbnails, and edges. Render distant regions as cluster summaries, show nearby resources individually, and emphasize selected relationships. Every resource must remain reachable through search even when its region is collapsed. + +| Cost | Proposed treatment | Evidence to collect | +| ------------------------ | ---------------------------------------------------------------------- | ----------------------------------------------------------------------- | +| Layout changes | Settle or cache a versioned layout before flight | No continuous full-position uploads during a stable session. | +| Geometry and overdraw | Preserve batching; limit visible edge density; consider spatial chunks | Stereo frame timing and transparent fragment cost at each graph size. | +| Labels and images | A bounded label budget, a few pinned cards, demand-loaded thumbnails | Texture memory and readability at actual viewing distances. | +| Picking | Spatially indexed candidates or a measured GPU picking path | No full 54,000-node raycast on every XR frame. | +| Shader sizing | Review the current pixel-sized point shader against each XR view | Nodes remain legible and targetable without filling the scene. | +| Resolution and foveation | Feature-detect supported controls and tune after measurement | Measured quality and frame-time tradeoff, not a desktop DPR assumption. | + +Measure delivered session cadence, p95/p99 frame intervals, missed frames, memory, and sustained behavior. Use GPU timing only where the runtime exposes a suitable facility. Derive a frame budget from actual cadence: 90 Hz gives about 11.1 ms and 120 Hz about 8.3 ms. These are examples, not promises about a particular headset session. + +Keep WebGL for the first experiment. A WebGPU migration or graph-library replacement adds scope before we know the bottleneck. If the large graph cannot meet the measured budget after bounded detail and stable layout, compare native rendering and Mac-assisted paths using the same fixture and interaction tasks. + +## Risks and decisions still open + +The largest uncertainty is Safari's exposure of the exact controller combination. Record headset model, visionOS version, Safari build, and controller firmware when testing. Documentation establishes native support; only a device result can settle the proposed web flight path. + +Spatial browsing also changes what a good graph layout means. A useful monitor layout can be too dense to inhabit. We may need a room-sized overview and local neighborhoods even when rendering every point is technically affordable. Evaluate whether flight helps Chris find and connect resources, rather than treating time spent flying as the success metric. + +The data connection is a separate deliverable. A headset that cannot read the Mac's authenticated Library will only show a demo. Mac sleep, revoked pairing, unavailable thumbnails, and interrupted requests must produce understandable states without losing the selected resource. Offline use needs a deliberate local content policy and must not be claimed from graph metadata alone. + +Use the existing desktop interface as the accessible fallback. In-headset controls need adjustable text, handedness, and stationary alternatives; color alone should not distinguish edge evidence. Refer to Apple's [accessibility guidance](https://developer.apple.com/design/human-interface-guidelines/accessibility) when implementing the native or spatial interface. + +## Implementation checklist + +### 1. Establish the hardware facts + +- [ ] Build a minimal secure WebXR probe using synthetic geometry and explicit session entry/exit. +- [ ] Record actual device/software versions, `immersive-vr` support, granted features, and session cadence. +- [ ] Pair PS VR2 Sense controllers through the system and inspect input-source profiles, handedness, ray mode, grip pose, and gamepad mapping. +- [ ] Verify independent position and rotation tracking, analog trigger values across the squeeze range, and disconnect/reconnect behavior. +- [ ] Verify hands-only selection and coexistence of transient pointers with persistent input sources. +- [ ] Decide WebXR or native for controller flight from these results. A missing browser capability remains missing; do not silently substitute a button-only demo. + +### 2. Make one small graph useful in the headset + +- [ ] Extract the graph data boundary and share model/layout logic without making the browser depend on Electron preload. +- [ ] Add XR session lifecycle, metre scaling, a locomotion rig, and stable precomputed positions; restore desktop controls on exit. +- [ ] Add stationary overview, selected-resource highlighting, spatial labels, one metadata card, and a home action. +- [ ] Implement point-and-accelerate flight, configurable handedness, braking, and interruption handling as pure state transitions with tested math. +- [ ] Add explicit search destination travel and a return trail before enabling long-distance free flight. +- [ ] Prototype full-pose flight behind a separate opt-in only after the default flight interaction passes hardware review. + +### 3. Connect the real Library and measure scale + +- [ ] Design and verify trusted, paired read-only access for snapshots, search, details, and thumbnails; keep private data off public hosting. +- [ ] Use stable IDs and revision-matched positions; handle stale responses, disconnection, and missing enrichment explicitly. +- [ ] Add playlist, platform, tag, and category filtering with visible counts and evidence labels. +- [ ] Measure 1,000-link, 10,000-link, and representative full-Library fixtures on the headset; document the chosen frame and memory budgets. +- [ ] Add bounded labels, edge detail, thumbnail loading, and picking acceleration where measurements require them. +- [ ] Decide whether native visionOS or Mac-assisted rendering is justified by an observed capability or performance gap. + +## Validation checklist + +The named consumer of this validation is the **spatial Library release review**, owned by Chris. It can accept the browser viewer, require the native fallback, or stop the experiment. Physical-device results are required; a simulator or desktop browser is insufficient evidence for tracked flight or comfort. + +- [ ] With real hardware, find three known resources by search and by navigating a neighborhood; inspect saved metadata and return to the previous location. +- [ ] Demonstrate that trigger pressure changes acceleration, the speed cap holds, the head remains independent, and braking does not require a precise gesture. +- [ ] Demonstrate full position/rotation input separately from ordinary gamepad button support; record which modes the tested platform supports. +- [ ] Exercise tracking loss, invalid poses, controller disconnect, session suspension, and re-entry with a held trigger; all must stop motion until explicitly rearmed. +- [ ] Verify deterministic flight integration with in-memory inputs at several frame rates, including large time gaps and non-finite input. If a new automated gate is added, include negative controls that make it fail when braking or limits are broken. +- [ ] Confirm selection and card inspection never cause unintended acceleration or automatic viewpoint jumps. +- [ ] Complete a proposed 20-minute browsing trial, record Chris's comfort and task feedback, and verify the stationary mode remains useful. This is a product trial, not a universal comfort guarantee. +- [ ] Meet the declared frame-time and memory budgets on the full-size fixture, including selection and filtering; retain every resource's searchability under level-of-detail reduction. +- [ ] Revoke pairing and interrupt connectivity; prove unauthorized reads fail, UI errors are visible, and original Library content is unchanged. +- [ ] Verify exit restores the desktop graph without duplicate loops, stale input handlers, or a moved desktop camera caused by XR cleanup. + +## Recommendation + +Build the **hardware capability probe first**, then a small read-only WebXR graph. The existing Three.js scene makes that a focused experiment, but the browser's Sense support and access to real Library data must be proven before promising the complete experience. + +If the browser supplies tracked poses and analog triggers, implement the requested flight model there and measure it with the real graph. If it does not, retain a useful hands-based browser viewer and choose a native visionOS client for full controller flight. Keep the graph model, saved content, relationship evidence, and interaction rules common across those paths. + +The first milestone is simple: put on the headset, find a saved link, fly deliberately to its neighborhood, inspect its thumbnail and metadata, and return home comfortably. That proves more than a large cloud of points alone. diff --git a/docs/explorations/STALE.md b/docs/explorations/STALE.md index 9931f663a..a94c8a3bd 100644 --- a/docs/explorations/STALE.md +++ b/docs/explorations/STALE.md @@ -14,7 +14,7 @@ review: 2027-02-01 # renew the claim status: withdrawn # release it; the document stays exactly where it is ``` -**41** stale of 307 undecided. +**41** stale of 310 undecided. ## How this backlog retires @@ -24,16 +24,16 @@ cannot drag the curve down. | Days since written | Cohort | Still unshipped | | --- | --- | --- | -| 1 | 445 | 57% | +| 1 | 448 | 57% | | 7 | 445 | 56% | -| 14 | 444 | 55% | +| 14 | 445 | 56% | | 30 | 444 | 55% | -| 60 | 400 | 53% | -| 90 | 247 | 51% | -| 120 | 136 | 61% | +| 60 | 428 | 54% | +| 90 | 280 | 50% | +| 120 | 156 | 62% | The curve does not fall: 57% of documents at least a day old are -unshipped, and 61% at 120 days. An exploration is checked off +unshipped, and 62% at 120 days. An exploration is checked off within days of being written, or never — so an old `[_]` is not a pending decision, it is a decision already made by inaction. Renew it deliberately, or withdraw it; both are one line and neither renames the file. @@ -42,47 +42,47 @@ or withdraw it; both are one line and neither renames the file. | Exploration | Due | Overdue | Decider | | --- | --- | --- | --- | -| [0079_[_]_AUTH_SCHEMA_DSL_VARIATIONS.md](0079_%5B_%5D_AUTH_SCHEMA_DSL_VARIATIONS.md) | 2026-05-09 *(default)* | 140d | — | -| [0080_[_]_UCAN_HYBRID_AUTHORIZATION_INTEGRATION.md](0080_%5B_%5D_UCAN_HYBRID_AUTHORIZATION_INTEGRATION.md) | 2026-05-10 *(default)* | 139d | — | -| [0081_[_]_NODE_PERMISSIONS_UCAN_EVALUATION.md](0081_%5B_%5D_NODE_PERMISSIONS_UCAN_EVALUATION.md) | 2026-05-10 *(default)* | 139d | — | -| [0082_[_]_GLOBAL_NAMESPACE_AUTHORIZATION.md](0082_%5B_%5D_GLOBAL_NAMESPACE_AUTHORIZATION.md) | 2026-05-10 *(default)* | 139d | — | -| [0083_[_]_UNIFIED_AUTHORIZATION_ARCHITECTURE.md](0083_%5B_%5D_UNIFIED_AUTHORIZATION_ARCHITECTURE.md) | 2026-05-10 *(default)* | 139d | — | -| [0084_[_]_GROUPS_AS_RELATIONS.md](0084_%5B_%5D_GROUPS_AS_RELATIONS.md) | 2026-05-10 *(default)* | 139d | — | -| [0086_[_]_NATIVE_REWRITE_ZIG_RUST.md](0086_%5B_%5D_NATIVE_REWRITE_ZIG_RUST.md) | 2026-05-12 *(default)* | 137d | — | -| [0088_[_]_DATABASE_UI_COMPETITIVE_ARCHITECTURE.md](0088_%5B_%5D_DATABASE_UI_COMPETITIVE_ARCHITECTURE.md) | 2026-05-13 *(default)* | 136d | — | -| [0089_[_]_REST_GRAPHQL_INTEROPERABILITY_BOUNDARY.md](0089_%5B_%5D_REST_GRAPHQL_INTEROPERABILITY_BOUNDARY.md) | 2026-05-18 *(default)* | 131d | — | -| [0090_[_]_ELECTRON_P2P_REMOTE_SHARE_OPTIONS.md](0090_%5B_%5D_ELECTRON_P2P_REMOTE_SHARE_OPTIONS.md) | 2026-05-21 *(default)* | 128d | — | -| [0091_[_]_GLOBAL_SCHEMA_FEDERATION_MODEL.md](0091_%5B_%5D_GLOBAL_SCHEMA_FEDERATION_MODEL.md) | 2026-05-21 *(default)* | 128d | — | -| [0093_[_]_NODE_NATIVE_GLOBAL_SCHEMA_FEDERATION_MODEL.md](0093_%5B_%5D_NODE_NATIVE_GLOBAL_SCHEMA_FEDERATION_MODEL.md) | 2026-05-21 *(default)* | 128d | — | -| [0095_[_]_PACKAGE_PORTFOLIO_CLEANUP_AND_API_SIMPLIFICATION.md](0095_%5B_%5D_PACKAGE_PORTFOLIO_CLEANUP_AND_API_SIMPLIFICATION.md) | 2026-05-30 *(default)* | 119d | — | -| [0096_[_]_PLAN03_ERP_REALITY_CHECK_AND_EXECUTION_RESET.md](0096_%5B_%5D_PLAN03_ERP_REALITY_CHECK_AND_EXECUTION_RESET.md) | 2026-05-30 *(default)* | 119d | — | -| [0098_[_]_OPENCLAW_INTEGRATION.md](0098_%5B_%5D_OPENCLAW_INTEGRATION.md) | 2026-06-01 *(default)* | 117d | — | -| [0099_[_]_DATABASE_EDITING_UX_AND_UNDO_REDO_REMEDIATION_PLAN.md](0099_%5B_%5D_DATABASE_EDITING_UX_AND_UNDO_REDO_REMEDIATION_PLAN.md) | 2026-06-01 *(default)* | 117d | — | -| [0100_[_]_NPM_PUBLISH_WORKFLOW_FOR_XNETJS.md](0100_%5B_%5D_NPM_PUBLISH_WORKFLOW_FOR_XNETJS.md) | 2026-06-02 *(default)* | 116d | — | -| [0101_[_]_END_TO_END_NPM_TRUSTED_PUBLISHING_PLAYBOOK.md](0101_%5B_%5D_END_TO_END_NPM_TRUSTED_PUBLISHING_PLAYBOOK.md) | 2026-06-03 *(default)* | 115d | — | -| [0102_[_]_AFFINE_BLOCKSUITE_INTEGRATION_FEASIBILITY.md](0102_%5B_%5D_AFFINE_BLOCKSUITE_INTEGRATION_FEASIBILITY.md) | 2026-06-03 *(default)* | 115d | — | -| [0103_[-]_TASKS_EMBEDDED_IN_PAGES_BACKED_BY_NODES_MENTIONS_DUE_DATES_NESTED_SUBTASKS_DATABASES_CANVASES_AND_CROSS_SURFACE_TASK_MODEL.md](0103_%5B-%5D_TASKS_EMBEDDED_IN_PAGES_BACKED_BY_NODES_MENTIONS_DUE_DATES_NESTED_SUBTASKS_DATABASES_CANVASES_AND_CROSS_SURFACE_TASK_MODEL.md) | 2026-06-04 *(default)* | 114d | — | -| [0104_[-]_EXPLORE_DRAMATICALLY_SIMPLIFYING_THE_UX_AROUND_A_CANVAS_FIRST_PRIMARY_APP_INSPIRED_BY_AFFINE_MINIMIZING_BUTTONS_AND_CHROME_WITH_ZOOM_IN_DOCUMENTS_AND_DATABASES.md](0104_%5B-%5D_EXPLORE_DRAMATICALLY_SIMPLIFYING_THE_UX_AROUND_A_CANVAS_FIRST_PRIMARY_APP_INSPIRED_BY_AFFINE_MINIMIZING_BUTTONS_AND_CHROME_WITH_ZOOM_IN_DOCUMENTS_AND_DATABASES.md) | 2026-06-04 *(default)* | 114d | — | -| [0105_[_]_WHAT_TO_WORK_ON_NEXT_AFTER_OPEN_SOURCE_LAUNCH.md](0105_%5B_%5D_WHAT_TO_WORK_ON_NEXT_AFTER_OPEN_SOURCE_LAUNCH.md) | 2026-06-05 *(default)* | 113d | — | -| [0106_[_]_CI_PERF_TESTING_OPTIONS.md](0106_%5B_%5D_CI_PERF_TESTING_OPTIONS.md) | 2026-06-05 *(default)* | 113d | — | -| [0106_[_]_JOIN_QUERIES_MULTI_TYPE_AGGREGATES_QUERY_PLANNING_API.md](0106_%5B_%5D_JOIN_QUERIES_MULTI_TYPE_AGGREGATES_QUERY_PLANNING_API.md) | 2026-06-05 *(default)* | 113d | — | -| [0107_[_]_STORYBOOK_PERFORMANCE_PANEL_AND_ELECTRON_IDE_WORKSHOP.md](0107_%5B_%5D_STORYBOOK_PERFORMANCE_PANEL_AND_ELECTRON_IDE_WORKSHOP.md) | 2026-06-06 *(default)* | 112d | — | -| [0108_[_]_CANVAS_V1_PAGES_DATABASES_AND_INFINITE_CANVAS_DEEP_DIVE.md](0108_%5B_%5D_CANVAS_V1_PAGES_DATABASES_AND_INFINITE_CANVAS_DEEP_DIVE.md) | 2026-06-07 *(default)* | 111d | — | -| [0108_[_]_EXPO_APP_PARITY_WITH_ELECTRON_AND_WEB.md](0108_%5B_%5D_EXPO_APP_PARITY_WITH_ELECTRON_AND_WEB.md) | 2026-06-07 *(default)* | 111d | — | -| [0108_[_]_TIMING_FOR_INTEGRATING_CHAT_AND_VIDEO_INTO_XNET_NOW_VS_LATER.md](0108_%5B_%5D_TIMING_FOR_INTEGRATING_CHAT_AND_VIDEO_INTO_XNET_NOW_VS_LATER.md) | 2026-06-07 *(default)* | 111d | — | -| [0108_[_]_USEQUERY_UPGRADE_TIMING_AND_INTEGRATION_SEQUENCING.md](0108_%5B_%5D_USEQUERY_UPGRADE_TIMING_AND_INTEGRATION_SEQUENCING.md) | 2026-06-07 *(default)* | 111d | — | -| [0109_[_]_REPOSITORY_AND_PROJECT_SUMMARY_FOR_NON_TECHNICAL_USERS.md](0109_%5B_%5D_REPOSITORY_AND_PROJECT_SUMMARY_FOR_NON_TECHNICAL_USERS.md) | 2026-06-08 *(default)* | 110d | — | -| [0110_[_]_XNET_AS_A_VIABLE_WIKIPEDIA_ALTERNATIVE.md](0110_%5B_%5D_XNET_AS_A_VIABLE_WIKIPEDIA_ALTERNATIVE.md) | 2026-07-04 *(default)* | 84d | — | -| [0111_[_]_UNIFIED_WORKBENCH_ARCHITECTURE_FOR_XNET.md](0111_%5B_%5D_UNIFIED_WORKBENCH_ARCHITECTURE_FOR_XNET.md) | 2026-07-04 *(default)* | 84d | — | -| [0112_[_]_UNIVERSAL_CLIPPER_AND_AI_KNOWLEDGE_GRAPH_INGESTION.md](0112_%5B_%5D_UNIVERSAL_CLIPPER_AND_AI_KNOWLEDGE_GRAPH_INGESTION.md) | 2026-07-04 *(default)* | 84d | — | -| [0113_[_]_OTHER_INTERNET_INFRASTRUCTURE_ROLES_FOR_XNET.md](0113_%5B_%5D_OTHER_INTERNET_INFRASTRUCTURE_ROLES_FOR_XNET.md) | 2026-07-04 *(default)* | 84d | — | -| [0114_[_]_DECENTRALIZED_ALTERNATIVES_FOR_NON_XNET_INTERNET_LAYERS.md](0114_%5B_%5D_DECENTRALIZED_ALTERNATIVES_FOR_NON_XNET_INTERNET_LAYERS.md) | 2026-07-04 *(default)* | 84d | — | -| [0115_[_]_ARCHITECTING_FULLY_DECENTRALIZED_GLOBAL_WEB_SEARCH.md](0115_%5B_%5D_ARCHITECTING_FULLY_DECENTRALIZED_GLOBAL_WEB_SEARCH.md) | 2026-07-06 *(default)* | 82d | — | -| [0116_[_]_ARCHITECTING_DECENTRALIZED_TWITTER_X_ON_XNET.md](0116_%5B_%5D_ARCHITECTING_DECENTRALIZED_TWITTER_X_ON_XNET.md) | 2026-07-06 *(default)* | 82d | — | -| [0117_[_]_ARCHITECTING_DECENTRALIZED_AI_ON_XNET.md](0117_%5B_%5D_ARCHITECTING_DECENTRALIZED_AI_ON_XNET.md) | 2026-07-06 *(default)* | 82d | — | -| [0118_[_]_ARCHITECTING_A_DECENTRALIZED_OSS_FORGE_ON_XNET.md](0118_%5B_%5D_ARCHITECTING_A_DECENTRALIZED_OSS_FORGE_ON_XNET.md) | 2026-07-06 *(default)* | 82d | — | -| [0119_[_]_XNET_AS_A_COMPELLING_WEB_AND_MOBILE_DEVELOPER_TOOL.md](0119_%5B_%5D_XNET_AS_A_COMPELLING_WEB_AND_MOBILE_DEVELOPER_TOOL.md) | 2026-07-06 *(default)* | 82d | — | -| [0120_[_]_XNET_PACKAGE_SECURITY_AND_RELIABILITY_EXPLORATION.md](0120_%5B_%5D_XNET_PACKAGE_SECURITY_AND_RELIABILITY_EXPLORATION.md) | 2026-07-06 *(default)* | 82d | — | +| [0079_[_]_AUTH_SCHEMA_DSL_VARIATIONS.md](0079_%5B_%5D_AUTH_SCHEMA_DSL_VARIATIONS.md) | 2026-05-09 *(default)* | 150d | — | +| [0080_[_]_UCAN_HYBRID_AUTHORIZATION_INTEGRATION.md](0080_%5B_%5D_UCAN_HYBRID_AUTHORIZATION_INTEGRATION.md) | 2026-05-10 *(default)* | 149d | — | +| [0081_[_]_NODE_PERMISSIONS_UCAN_EVALUATION.md](0081_%5B_%5D_NODE_PERMISSIONS_UCAN_EVALUATION.md) | 2026-05-10 *(default)* | 149d | — | +| [0082_[_]_GLOBAL_NAMESPACE_AUTHORIZATION.md](0082_%5B_%5D_GLOBAL_NAMESPACE_AUTHORIZATION.md) | 2026-05-10 *(default)* | 149d | — | +| [0083_[_]_UNIFIED_AUTHORIZATION_ARCHITECTURE.md](0083_%5B_%5D_UNIFIED_AUTHORIZATION_ARCHITECTURE.md) | 2026-05-10 *(default)* | 149d | — | +| [0084_[_]_GROUPS_AS_RELATIONS.md](0084_%5B_%5D_GROUPS_AS_RELATIONS.md) | 2026-05-10 *(default)* | 149d | — | +| [0086_[_]_NATIVE_REWRITE_ZIG_RUST.md](0086_%5B_%5D_NATIVE_REWRITE_ZIG_RUST.md) | 2026-05-12 *(default)* | 147d | — | +| [0088_[_]_DATABASE_UI_COMPETITIVE_ARCHITECTURE.md](0088_%5B_%5D_DATABASE_UI_COMPETITIVE_ARCHITECTURE.md) | 2026-05-13 *(default)* | 146d | — | +| [0089_[_]_REST_GRAPHQL_INTEROPERABILITY_BOUNDARY.md](0089_%5B_%5D_REST_GRAPHQL_INTEROPERABILITY_BOUNDARY.md) | 2026-05-18 *(default)* | 141d | — | +| [0090_[_]_ELECTRON_P2P_REMOTE_SHARE_OPTIONS.md](0090_%5B_%5D_ELECTRON_P2P_REMOTE_SHARE_OPTIONS.md) | 2026-05-21 *(default)* | 138d | — | +| [0091_[_]_GLOBAL_SCHEMA_FEDERATION_MODEL.md](0091_%5B_%5D_GLOBAL_SCHEMA_FEDERATION_MODEL.md) | 2026-05-21 *(default)* | 138d | — | +| [0093_[_]_NODE_NATIVE_GLOBAL_SCHEMA_FEDERATION_MODEL.md](0093_%5B_%5D_NODE_NATIVE_GLOBAL_SCHEMA_FEDERATION_MODEL.md) | 2026-05-21 *(default)* | 138d | — | +| [0095_[_]_PACKAGE_PORTFOLIO_CLEANUP_AND_API_SIMPLIFICATION.md](0095_%5B_%5D_PACKAGE_PORTFOLIO_CLEANUP_AND_API_SIMPLIFICATION.md) | 2026-05-30 *(default)* | 129d | — | +| [0096_[_]_PLAN03_ERP_REALITY_CHECK_AND_EXECUTION_RESET.md](0096_%5B_%5D_PLAN03_ERP_REALITY_CHECK_AND_EXECUTION_RESET.md) | 2026-05-30 *(default)* | 129d | — | +| [0098_[_]_OPENCLAW_INTEGRATION.md](0098_%5B_%5D_OPENCLAW_INTEGRATION.md) | 2026-06-01 *(default)* | 127d | — | +| [0099_[_]_DATABASE_EDITING_UX_AND_UNDO_REDO_REMEDIATION_PLAN.md](0099_%5B_%5D_DATABASE_EDITING_UX_AND_UNDO_REDO_REMEDIATION_PLAN.md) | 2026-06-01 *(default)* | 127d | — | +| [0100_[_]_NPM_PUBLISH_WORKFLOW_FOR_XNETJS.md](0100_%5B_%5D_NPM_PUBLISH_WORKFLOW_FOR_XNETJS.md) | 2026-06-02 *(default)* | 126d | — | +| [0101_[_]_END_TO_END_NPM_TRUSTED_PUBLISHING_PLAYBOOK.md](0101_%5B_%5D_END_TO_END_NPM_TRUSTED_PUBLISHING_PLAYBOOK.md) | 2026-06-03 *(default)* | 125d | — | +| [0102_[_]_AFFINE_BLOCKSUITE_INTEGRATION_FEASIBILITY.md](0102_%5B_%5D_AFFINE_BLOCKSUITE_INTEGRATION_FEASIBILITY.md) | 2026-06-03 *(default)* | 125d | — | +| [0103_[-]_TASKS_EMBEDDED_IN_PAGES_BACKED_BY_NODES_MENTIONS_DUE_DATES_NESTED_SUBTASKS_DATABASES_CANVASES_AND_CROSS_SURFACE_TASK_MODEL.md](0103_%5B-%5D_TASKS_EMBEDDED_IN_PAGES_BACKED_BY_NODES_MENTIONS_DUE_DATES_NESTED_SUBTASKS_DATABASES_CANVASES_AND_CROSS_SURFACE_TASK_MODEL.md) | 2026-06-04 *(default)* | 124d | — | +| [0104_[-]_EXPLORE_DRAMATICALLY_SIMPLIFYING_THE_UX_AROUND_A_CANVAS_FIRST_PRIMARY_APP_INSPIRED_BY_AFFINE_MINIMIZING_BUTTONS_AND_CHROME_WITH_ZOOM_IN_DOCUMENTS_AND_DATABASES.md](0104_%5B-%5D_EXPLORE_DRAMATICALLY_SIMPLIFYING_THE_UX_AROUND_A_CANVAS_FIRST_PRIMARY_APP_INSPIRED_BY_AFFINE_MINIMIZING_BUTTONS_AND_CHROME_WITH_ZOOM_IN_DOCUMENTS_AND_DATABASES.md) | 2026-06-04 *(default)* | 124d | — | +| [0105_[_]_WHAT_TO_WORK_ON_NEXT_AFTER_OPEN_SOURCE_LAUNCH.md](0105_%5B_%5D_WHAT_TO_WORK_ON_NEXT_AFTER_OPEN_SOURCE_LAUNCH.md) | 2026-06-05 *(default)* | 123d | — | +| [0106_[_]_CI_PERF_TESTING_OPTIONS.md](0106_%5B_%5D_CI_PERF_TESTING_OPTIONS.md) | 2026-06-05 *(default)* | 123d | — | +| [0106_[_]_JOIN_QUERIES_MULTI_TYPE_AGGREGATES_QUERY_PLANNING_API.md](0106_%5B_%5D_JOIN_QUERIES_MULTI_TYPE_AGGREGATES_QUERY_PLANNING_API.md) | 2026-06-05 *(default)* | 123d | — | +| [0107_[_]_STORYBOOK_PERFORMANCE_PANEL_AND_ELECTRON_IDE_WORKSHOP.md](0107_%5B_%5D_STORYBOOK_PERFORMANCE_PANEL_AND_ELECTRON_IDE_WORKSHOP.md) | 2026-06-06 *(default)* | 122d | — | +| [0108_[_]_CANVAS_V1_PAGES_DATABASES_AND_INFINITE_CANVAS_DEEP_DIVE.md](0108_%5B_%5D_CANVAS_V1_PAGES_DATABASES_AND_INFINITE_CANVAS_DEEP_DIVE.md) | 2026-06-07 *(default)* | 121d | — | +| [0108_[_]_EXPO_APP_PARITY_WITH_ELECTRON_AND_WEB.md](0108_%5B_%5D_EXPO_APP_PARITY_WITH_ELECTRON_AND_WEB.md) | 2026-06-07 *(default)* | 121d | — | +| [0108_[_]_TIMING_FOR_INTEGRATING_CHAT_AND_VIDEO_INTO_XNET_NOW_VS_LATER.md](0108_%5B_%5D_TIMING_FOR_INTEGRATING_CHAT_AND_VIDEO_INTO_XNET_NOW_VS_LATER.md) | 2026-06-07 *(default)* | 121d | — | +| [0108_[_]_USEQUERY_UPGRADE_TIMING_AND_INTEGRATION_SEQUENCING.md](0108_%5B_%5D_USEQUERY_UPGRADE_TIMING_AND_INTEGRATION_SEQUENCING.md) | 2026-06-07 *(default)* | 121d | — | +| [0109_[_]_REPOSITORY_AND_PROJECT_SUMMARY_FOR_NON_TECHNICAL_USERS.md](0109_%5B_%5D_REPOSITORY_AND_PROJECT_SUMMARY_FOR_NON_TECHNICAL_USERS.md) | 2026-06-08 *(default)* | 120d | — | +| [0110_[_]_XNET_AS_A_VIABLE_WIKIPEDIA_ALTERNATIVE.md](0110_%5B_%5D_XNET_AS_A_VIABLE_WIKIPEDIA_ALTERNATIVE.md) | 2026-07-04 *(default)* | 94d | — | +| [0111_[_]_UNIFIED_WORKBENCH_ARCHITECTURE_FOR_XNET.md](0111_%5B_%5D_UNIFIED_WORKBENCH_ARCHITECTURE_FOR_XNET.md) | 2026-07-04 *(default)* | 94d | — | +| [0112_[_]_UNIVERSAL_CLIPPER_AND_AI_KNOWLEDGE_GRAPH_INGESTION.md](0112_%5B_%5D_UNIVERSAL_CLIPPER_AND_AI_KNOWLEDGE_GRAPH_INGESTION.md) | 2026-07-04 *(default)* | 94d | — | +| [0113_[_]_OTHER_INTERNET_INFRASTRUCTURE_ROLES_FOR_XNET.md](0113_%5B_%5D_OTHER_INTERNET_INFRASTRUCTURE_ROLES_FOR_XNET.md) | 2026-07-04 *(default)* | 94d | — | +| [0114_[_]_DECENTRALIZED_ALTERNATIVES_FOR_NON_XNET_INTERNET_LAYERS.md](0114_%5B_%5D_DECENTRALIZED_ALTERNATIVES_FOR_NON_XNET_INTERNET_LAYERS.md) | 2026-07-04 *(default)* | 94d | — | +| [0115_[_]_ARCHITECTING_FULLY_DECENTRALIZED_GLOBAL_WEB_SEARCH.md](0115_%5B_%5D_ARCHITECTING_FULLY_DECENTRALIZED_GLOBAL_WEB_SEARCH.md) | 2026-07-06 *(default)* | 92d | — | +| [0116_[_]_ARCHITECTING_DECENTRALIZED_TWITTER_X_ON_XNET.md](0116_%5B_%5D_ARCHITECTING_DECENTRALIZED_TWITTER_X_ON_XNET.md) | 2026-07-06 *(default)* | 92d | — | +| [0117_[_]_ARCHITECTING_DECENTRALIZED_AI_ON_XNET.md](0117_%5B_%5D_ARCHITECTING_DECENTRALIZED_AI_ON_XNET.md) | 2026-07-06 *(default)* | 92d | — | +| [0118_[_]_ARCHITECTING_A_DECENTRALIZED_OSS_FORGE_ON_XNET.md](0118_%5B_%5D_ARCHITECTING_A_DECENTRALIZED_OSS_FORGE_ON_XNET.md) | 2026-07-06 *(default)* | 92d | — | +| [0119_[_]_XNET_AS_A_COMPELLING_WEB_AND_MOBILE_DEVELOPER_TOOL.md](0119_%5B_%5D_XNET_AS_A_COMPELLING_WEB_AND_MOBILE_DEVELOPER_TOOL.md) | 2026-07-06 *(default)* | 92d | — | +| [0120_[_]_XNET_PACKAGE_SECURITY_AND_RELIABILITY_EXPLORATION.md](0120_%5B_%5D_XNET_PACKAGE_SECURITY_AND_RELIABILITY_EXPLORATION.md) | 2026-07-06 *(default)* | 92d | — | ## Undated diff --git a/docs/reference/desktop-storage.md b/docs/reference/desktop-storage.md new file mode 100644 index 000000000..19e3fc2a3 --- /dev/null +++ b/docs/reference/desktop-storage.md @@ -0,0 +1,229 @@ +# Desktop storage and recovery + +This inventory describes the Electron paths inspected for exploration 0466. A +native recovery copy covers `xnet-data`, including an encrypted logical copy of +known desktop settings. Device-bound sign-in sessions, unlisted browser state, +and external files remain outside that claim. The Settings screen states this limit. Native copies remain local. The separate encrypted export below supports portable recovery of the covered files. + +## Data locations + +`src/main/profile.ts` resolves Electron's `userData` before taking the instance +lock. The packaged default keeps its existing directory. Source builds prefix +their profile with `dev-`; packaged builds reject that reserved prefix. The +workspace directory is `/xnet-data`, and recovery copies live beside +it in `/xnet-recovery`. + +| Location | Contents | Current recovery coverage | +| ----------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | +| `xnet-data/data.db` and its WAL | Nodes, properties, signed changes, Yjs state/history, sync state, indexes, and the data-process blob table | Required; copied and normalized into a standalone database | +| `xnet-data/xnet.db` and its WAL | Main-process blob service bytes, including editor attachments | Required; copied and normalized | +| `xnet-data/identity-seed.json` | Desktop signing seed, normally encrypted by Electron `safeStorage` | Required outside explicit test mode; same Mac key store needed | +| `xnet-data/seed-recovery.json` | Optional encrypted recovery mnemonic | Included when present; same Mac key store needed | +| `xnet-data/import-sources//` | Exact source export bytes retained before an import writes nodes | Included recursively, including unselected source categories | +| `xnet-data/import-jobs/` | Reviewed import selections, adapter versions, and acknowledged batch cursors; no signing keys | Included with retained source evidence; interrupted jobs reopen paused | +| Other files under `xnet-data`, including tunnel state | Native persisted configuration and future files | Included recursively; symlinks and special files fail the copy | +| `xnet-data/desktop-settings.json` | Encrypted logical copy of known settings, provider key, workspace layout, and unfinished capture draft | Included after the renderer save barrier; portable export rewraps it | +| Chromium `Local Storage`, `IndexedDB`, `Preferences`, and cookies | Original settings, device-bound sessions, and caches | Raw files excluded; only the explicit logical settings contract above travels | +| macOS Keychain | The OS secret used by `safeStorage`; platform sign-in credentials | Not copied; a file copy alone cannot recover these on another Mac | +| `/library-helpers` | Verified downloadable video helper; no workspace data or keys | Excluded; reinstall through Library after restore | +| `/dictation` | Downloaded local transcription models | Excluded; models can be downloaded again | +| `/agent-bridge-mcp.json` | Generated agent connection configuration | Excluded; regenerate it after restoring | +| System temporary recording directories | In-progress screen/audio recordings | Not a saved attachment yet; quit stops capture before checkpointing | +| Files selected from outside the workspace | Export inputs, capture sources, published output | Only a source explicitly retained under `xnet-data` is covered | +| `/xnet-recovery` | Verified generations, pinned pre-upgrade points, preserved originals, and restore journals | Kept outside the source to prevent recursive backups; still on the same disk | + +The renderer currently supplies its desktop signing identity directly to +`XNetProvider` and uses IPC-backed node and blob storage. Shared browser identity +and session code also exists in the repository, but is not the desktop signing +identity. Browser passkey credentials and non-extractable session wrapping keys +are not exported. AI vector indexes, downloaded models, and telemetry buffers are +rebuildable or disposable. Never replace a failed or missing identity with a +fresh one just to get the app open. A real daily-profile inventory is still needed +before claiming coverage for arbitrary plugins or future browser stores. + +### Existing local profile observation (2026-09-30) + +The existing `xnet-desktop` profile has storage version 9, 207 node records, +20 saved Yjs states, both native databases, and Chromium local/IndexedDB stores. +It has no `identity-seed.json`. These are metadata observations, not a claim that +all records belong to one recoverable identity. No migration or identity +replacement was attempted. The new startup guard refuses this combination. + +Preserve the whole legacy profile before investigating its identity history. +Do not enable the deterministic test identity to make it open, generate a new +seed over it, or treat this profile as a successful installed-upgrade fixture. +Legacy ownership recovery remains a separate prerequisite for using it with the +new daily build. Development profiles continue to use separate directories. + +## Atomic record saves + +Desktop `NodeStore` creates, updates, deletes, restores, and explicit multi-record +transactions use the native SQLite batch operation. The materialized records, +property indexes, signed changes, and logical clock commit together. IPC returns +success and broadcasts changes only after that commit. A rejected write rolls the +batch back and remains an error to its caller. History reads use the shared +storage decoder, including records written by imports; change IDs, signatures, +protocol versions, and batch positions survive reopening. + +```mermaid +sequenceDiagram + participant UI as Desktop NodeStore + participant IPC as Native data process + participant DB as SQLite FULL transaction + UI->>IPC: Signed changes and final record states + IPC->>DB: Apply one atomic batch + alt Commit succeeds + DB-->>IPC: Committed + IPC-->>UI: Change notification and acknowledgement + else Any statement fails + DB-->>IPC: Entire batch rolled back + IPC-->>UI: Save error + end +``` + +A disposable real Electron profile verified a deliberately rejected transaction, +retry, signed batch history after a direct exit that bypassed the quit barrier, +and a final acknowledged edit after normal quit and restart. A local synthetic +sample of 100 creates and 100 updates on 2026-09-30 measured median 4.3/5.3 ms and +95th-percentile 8.6/10.0 ms respectively, through the renderer and native IPC with +the configured `FULL` writer. This is a small warm sample, not a full-corpus +benchmark or a physical power-loss test. + +This does not yet cover every mutation path. Document and blob operations have +separate persistence paths; concurrent writers can still need reconciliation +between reading a record and submitting its batch. A renderer operation still +signing or waiting in an application debounce has not reached this commit +boundary. Do not interpret atomic batches as a global save barrier. + +## Native checkpoint contract + +`xnet-desktop-checkpoint/1` records each retained file's path, size, and SHA-256, +the creating app version, profile, identity mode, and capture time. New points +also record the workspace fingerprint and known storage version. A point is +visible only after file hashes, required files, and both databases pass checks. +An interrupted copy stays incomplete. A changed source, unreadable file, or +missing identity fails visibly. An older point with an unreadable manifest stays +on disk and appears as a warning in Settings. It is excluded from retention +deletion and does not block fresh copies or safe quit. Restore still verifies +every file, even when the manifest can be listed. + +```mermaid +flowchart LR + Edits[Flush document edits] --> Copy[Copy native workspace] + Copy --> Standalone[Normalize both SQLite stores] + Standalone --> Verify[Verify files and database integrity] + Verify --> Promote[Atomically publish recovery point] + Promote --> Retain[Apply retention to older points] + Verify -->|failure| Keep[Keep previous verified points] +``` + +The scheduler checks once a minute and creates a point when changed data is at +least fifteen minutes overdue. A successful manual or quit copy resets that +age. Retention keeps one point per fifteen-minute bucket for a day, per day for +a week, and per week for four weeks, plus the latest two. Pinned pre-upgrade +points and replaced original workspaces are not pruned automatically. Their disk +use must remain visible; this is not a total storage cap. + +Restore verifies the chosen point, stages a complete copy, and writes a durable +journal before renaming anything. Startup completes that journal before opening +normal stores. The replaced workspace is retained under `preserved/`. Automatic +sync, the local API, and the agent bridge remain paused until review and explicit +reconnection. + +## Supported storage upgrades + +The candidate migration path currently supports version 8 through the current +version 9. Its version-8 fixture is captured from `6650c1f39^`, before the pin +registry migration. Earlier, unversioned, future, or unreadable formats stop in +recovery. That is preservation, not a claim of migration support. + +An upgrade validates the identity first, pins a verified original, and applies +each registered SQL migration to a separate copy. It checks the resulting column +contract, existing table row counts, foreign keys, and database integrity before +promotion through the restore journal. Missing migration steps fail. Both the +pinned original and the replaced workspace remain available after promotion. +The next format change must bring its own historical fixture and migration +validation; a higher version number alone is not proof of compatibility. + +Native tests and source-app runs do not establish cross-Mac recovery, signed +installer continuity, or a power-loss guarantee for the underlying hardware. +Those remain separate acceptance checks in exploration 0466. + +## Encrypted export + +Settings → Data can export a verified native point into a password-encrypted +`.xnetbackup` folder. Keep the entire folder and password separately. Export +verifies all encrypted objects before publishing the folder. Restore verifies +all files in a new private directory and rewraps the actual signing seed with +the destination key store before replacing anything. The current workspace +remains preserved and the recovered workspace opens offline for review. + +The original Keychain secret is unnecessary for these encrypted exports. The +app does not save the recovery password. It does not claim off-device protection +merely because a folder was chosen: the user must retain a copy on another disk +or device. Version 2 adds the logical desktop settings and rewraps them with the +destination key store. Version-1 exports remain readable, with their original +coverage; they do not gain settings retroactively. Device-bound sessions remain +excluded. +Only the current storage version can be restored through this control today. +See ADR-40 and ADR-45 for format details and the limits of the cross-Mac evidence. + +## Library captures + +Library → Save a link creates a private Page with your note and optional +excerpt, linked to a source resource. The source and your writing remain separate. +The Library indexes the saved Page body and refreshes it when you return from the +editor. Matching source URLs reuse the existing resource while keeping each +explicitly requested note independent. + +Before native writes begin, `library.db` retains a versioned capture intent and +the original text. Startup retries incomplete intents with the same record IDs; +it never replaces a saved Page body during retry. These intents are currently +retained, including completed ones, so the original captured text remains in +local storage and recovery copies even after the Page changes. The Library database +is included in native checkpoints and encrypted exports when present. + +The unsubmitted form draft lives in Chromium local storage. The save barrier +now includes it in the encrypted settings copy before checkpoints, export, and +normal quit. Text entered after the last completed copy is not protected by that +copy. A successful save clears the draft only after +native acknowledgement. Failed attempts retain their original payload for retry. +The desktop shortcut reads the clipboard only when explicitly invoked. + +## Desktop settings recovery + +The versioned list in `src/shared/desktop-settings.ts` covers the workspace +layout and queued pins, theme and token overrides, selected hub, AI provider key +and preferences, meeting settings, consent choices, dismissed data suggestions, +and capture draft. Absent values are recorded too, so restoring an older point +can clear a setting introduced later. Unknown keys are not silently added to the +contract. Debug switches, test bypasses, temporary OAuth verifiers, and bridge +pairing tokens are excluded. Bridge pairing and sign-in may need to be repeated. + +The renderer snapshots these values only after document writes finish. Main +encrypts them with `safeStorage`, writes a private temporary file, fsyncs it, and +renames it into place. A locked key store or damaged prior settings copy fails +the operation and keeps the existing bytes. Unchanged settings do not rewrite +the file. The periodic checkpoint check captures settings before comparing the +workspace fingerprint, so a settings-only change can trigger an overdue copy. + +```mermaid +sequenceDiagram + participant Renderer + participant Main + participant Recovery + Renderer->>Renderer: Finish pending document writes + Renderer->>Main: Logical settings snapshot + Main->>Main: Encrypt, fsync, atomic rename + Main->>Recovery: Copy and verify native workspace + Recovery->>Main: Restore generation with unique ID + Main->>Renderer: Decrypted settings and restore ID + Renderer->>Renderer: Apply settings before importing the app +``` + +A boot module restores settings before theme, consent, and workspace stores +hydrate. Each restore has a unique receipt; ordinary restarts keep later edits. +An interrupted application leaves no completed receipt and retries on the next +launch. Losing the entire browser profile also triggers recovery from the native +settings copy. This is checkpoint recovery, not a claim that every keystroke in +an unfinished form has already reached a backup. diff --git a/docs/reference/personal-library.md b/docs/reference/personal-library.md new file mode 100644 index 000000000..db3acd866 --- /dev/null +++ b/docs/reference/personal-library.md @@ -0,0 +1,210 @@ +# Trying the desktop Library + +This is a development preview of exploration 0466. It can import supported +archives, fetch source details and images, search saved text offline, and capture +a link with an editable personal note. The full daily-use acceptance check is +still open. Keep the original exports and a separate copy of important writing. + +From the repository root, run `pnpm --filter xnet-desktop dev:electron`. The +launcher rebuilds changed workspace dependencies and uses a separate `dev-` +profile. Open **Library** from the top bar. Renderer edits use the development +reload loop; native-process changes need a restart. This is not the signed, +automatically updated installation proposed in the exploration. + +## Save and find + +Choose **Save a link**, enter a URL, and add a title, a thought, or an excerpt. +Saving works offline. Your writing becomes a private Page linked to the source; +metadata refreshes do not replace it. If the source is already in Library, the +form offers its existing notes and can save a separate new note. + +Open a result's details and choose **Open editable note** to keep writing. Return +to Library to refresh the note-body index. Search can find source descriptions, +URLs, personal notes, and retrieved caption segments. Imported items do not each +create a blank Page. + +The desktop registers **Command/Ctrl + Shift + L** for capture. It reads a copied +URL only when invoked. The Library reports if another app prevents registration. +Registration has been exercised; the switch from another app and return of focus +still needs the native shortcut acceptance check. + +A failed save keeps its original request and text for retry. You can explicitly edit it as a separate capture if you need a different note; the earlier attempt may already be saved. The unsubmitted +form draft is stored in Chromium local storage and included in the logical settings +snapshot at completed checkpoints. Once the native capture intent is written, +it joins the covered workspace files. Completed +intents currently retain the original capture text even after later Page edits. + +## Explore links in 3D + +Open **3D graph** in Library. The graph fills the window and includes every saved +web link, even when it has no known connections. Local text and conversations +stay in Resources; the inspector shows both counts. This view reads existing +data and does not change the database schema or write new relationships. + +Drag to orbit, scroll to zoom, and right-drag to pan. Hover over a dot to read its +saved title, thumbnail, description, and connections. Click to pin the preview. +**All saved metadata & provenance** expands the original imported fields and +saved enrichment. **Read in Library** opens the source's notes and transcript. +Missing images or descriptions are shown as missing, not fetched by the graph. + +Search for a title, URL, creator, tag, category, or playlist. Autocomplete ranks +matches in the current view as you type; use the arrow keys and Enter, or click +a suggestion, to move the camera and open its preview. An empty search suggests +popular groups. Escape dismisses suggestions first, then returns to Library. + +**Browse groups** lists categories, tags, playlists, and creators, with a search +field for each type. Counts show unique links from the selected source. Choose +one or more groups to filter the graph. **Any group** includes links in at least +one selected group; **All groups** includes only links in every selected group. +Remove a filter using its chip, or use **Clear filters** to return to all links. +Category labels come from saved source categories, subreddits, or repository +languages; they are not an AI classification of the whole Library. + +**Isolate neighborhood** shows a group's links, or the links sharing a connection +with the selected resource, within the active filters. **Leave neighborhood** +clears that focus and keeps the filters. The **Show connections** checkboxes +control relationship lines and hubs; hiding them keeps the selected membership +filters in effect. **Hide groups** and **Hide inspector** give the graph more +room. The graph counts always describe the current view. + +Connections use imported playlist and collection memberships, GitHub topics, +garden tags, source hashtags, categories, and matching creator names within a +platform. A matching name is not a verified identity. The graph does not infer +missing playlist memberships or generate AI categories. All graph work stays on +this Mac. The force layout runs in the background and can be paused; it stops +when settled and is released when the view closes. Reduced-motion settings start +it paused. **Reload graph** picks up new imports and enrichment. + +## Import and enrich + +Choose **Import archive**, select or drop a ZIP/JSON export, and review its categories. +The adapters cover Twitter/X, Instagram, YouTube, TikTok, Reddit, GitHub stars, +Claude, ChatGPT, Grok, and the personal garden. A garden file uses the existing version-1 `garden.json` +format. A retained copy of the complete source export is kept before node writes, +including categories you did not select. Keep this in mind when choosing files. + +Paused or interrupted import jobs can resume from retained source files. A batch +only advances the saved cursor after acknowledgement. Repeating a batch after a +crash uses deterministic IDs. The October 2 run matched all 120,985 expected +graph records and 37,518 nonempty text values from eight additional sources +against their original exports. This does not establish reconciliation across +every historical export or archive version. + +To preview the available archives without writing to a workspace, run: + +```bash +pnpm exec tsx scripts/inventory-personal-library.ts --all-supported \ + --garden-file /path/to/garden.json --output /tmp/library-preview.json +``` + +This includes social activity and AI conversations. It excludes direct messages, +billing, and account/security categories. The report lists selected and excluded +categories, raw counts, unique records, and parser warnings. Review it before +choosing categories in the app; it does not import them. + +Imported AI conversations and source records with local text can be searched and +read in Library. Conversation text stays local; the app does not fetch private +chat pages. Attachments remain in the retained export. Cited public links have +their own enrichment jobs. + +Open **Collections** to find an imported playlist or saved group by name. Both +the collection list and its entries have pages of forty items. Entries follow +the export's ordering key when present. Repeated saves remain separate entries; +opening either occurrence shows the same resource's source details and notes. +An entry without an indexed resource remains visible with an explanation. + +Collection cards show the item count reported by the source when available; +the opened collection counts the actual local membership records. These can +differ after an incomplete import. **Data & saved views** still opens the +existing data workspace for inspecting the broader graph. Dedicated creator +views and bounded resource neighborhoods remain unfinished. + +Choose **Start enrichment** to request source websites from this Mac. It is +paused by default. Coverage & gaps separates metadata, thumbnail, caption, and +index work, and shows missing fields and retry reasons. Saved thumbnails use local +blob storage, so expired source URLs do not remove the fetched image. + +YouTube titles, full descriptions, authors, thumbnail URLs, and caption lists now +come directly from the public video page. No helper installation is needed for +these cards. A public preview can supply a title and image when the full page +fails; the description and caption coverage then remain partial. + +The progress strip shows completed metadata, pending videos, saved images, and +current requests. Local indexing does not hold up network work. Several requests +can run at once, with separate pacing for metadata, images, and captions. A +private or removed video leaves a gap on that video. A provider rate limit pauses +that provider and survives retry or restart. Pausing cancels active requests; +starting again resumes the saved queue without resetting completed work. +Leave the desktop app running for a bulk pass. Different websites have independent +cooldowns, even when their links came from the same archive. Temporary provider +throttling stays retryable; completed work is retained across restarts. When a +separate image server is throttled, its thumbnail requests wait while source +pages can continue. Same-host and unidentified throttling still pause the source. +**Retry missing details** on a selected item moves its unfinished jobs ahead of +the bulk import queue. It still respects any provider rate limit. + +Caption discovery does not guarantee readable text. When YouTube returns an empty +or expired track, the app tries another format and a fresh track from the video +helper. It saves successful captions with their language and timestamps. An empty +response stays an explicit gap; it is never counted as a retrieved transcript. + +Instagram public embeds supply written captions, author names, and post images. +The saved post URL is used even when the export identifies the record with a +numeric Facebook ID. Photos and carousels can have a thumbnail without containing +a video. Display titles come from the written caption, so their coverage remains +partial. A public page preview is also marked partial when the full caption is +unavailable. Private, removed, or login-limited posts remain explicit gaps. + +Instagram's written caption is separate from speech in the video. The public +embed does not supply transcript tracks; local transcription is still pending. +Completed YouTube and Instagram metadata, saved images, and captions survive +provider upgrades. Unresolved caption jobs receive a fresh pass. + +For X/Twitter metadata and YouTube caption fallback on macOS, +**Coverage & gaps → Install video helper** downloads the tested +`yt-dlp 2026.07.04` executable from its official release (about 38 MB). The app +checks its pinned size, checksum, and version before installation. You can cancel +the download or repair a damaged helper. Installation does not start enrichment +or read browser cookies. The managed helper is preferred over compatible existing +local installations. It can be downloaded again after restoring a workspace. +The installer’s integrity and failure tests pass. In the live Electron check on +this Mac, the GitHub asset connection timed out; cancellation, cleanup, retry, +and restart worked. Successful download and installation through this control +still need a live check. Existing compatible local helpers can already supply +source metadata and captions. + +TikTok uses public post data for written captions, creator names, posters, and +available subtitle tracks. GitHub uses the public repository page, its embedded +repository details, and its rendered README. Retrieved details include topics, +website, stars, forks, and license information when the page supplies them. The +README and topics are searchable. Older page previews receive one fresh pass +automatically. Reddit uses public embed previews; +these do not promise full post bodies, images, or video transcripts. Ordinary web +pages contribute visible article text and embedded subtitle tracks. Cited links use the provider named by +their URL while retaining the archive they came from. + +Available source captions are supported; automatic local transcription and +generated video posters are not connected. Ordinary web descriptions may be preview text rather than the +full article, and login-limited sources remain explicit gaps. + +## Recovery and readiness + +Settings → Data provides verified native recovery copies and password-encrypted +`.xnetbackup` export/restore. Keep the whole encrypted folder and its password. +Native copies remain on this disk. Selecting a folder does not prove it has +reached another device. See [Desktop storage and recovery](./desktop-storage.md) +for exactly which files are covered. New copies include known desktop settings, +the AI provider key, and an unfinished capture draft. Restore applies these before +the app opens; device sessions and unlisted browser state are excluded. + +The current implementation has been exercised in isolated Electron profiles: +checkpoint restore, encrypted export/restore, interrupted-import resume after +removing its original source, saved thumbnails and late-caption search after an +offline restart, capture followed by ordinary title/body edits and restart, and +settings/draft recovery after deleting an entire test profile. +The installed signed-Mac upgrade, complete settings/key recovery, off-device +retention, complete real-corpus enrichment, guide publication, and founder trial +remain open in the exploration. The preview is not a durability guarantee for +irreplaceable data. + +The [partially implemented exploration](../explorations/0466_[-]_PERSONAL_LIBRARY_FOR_LEARNING_AND_SHARING.md) records the remaining work and command results. diff --git a/package.json b/package.json index 079f18d2e..3afba5fab 100644 --- a/package.json +++ b/package.json @@ -133,7 +133,11 @@ "y-webrtc@10.3.0": "patches/y-webrtc@10.3.0.patch" }, "overrides": { - "prosemirror-transform": "^1.12.0" + "prosemirror-transform": "^1.12.0", + "undici@>=6.0.0 <6.28.1": "6.28.1", + "@grpc/grpc-js@>=1.14.0 <1.14.5": "1.14.5", + "brace-expansion@>=1.0.0 <1.1.20": "1.1.20", + "brace-expansion@>=2.0.0 <2.1.6": "2.1.6" } } } diff --git a/packages/cli/src/__tests__/agent-commands.test.ts b/packages/cli/src/__tests__/agent-commands.test.ts index 5bec2e825..c6eda5095 100644 --- a/packages/cli/src/__tests__/agent-commands.test.ts +++ b/packages/cli/src/__tests__/agent-commands.test.ts @@ -314,6 +314,7 @@ describe('agent CLI commands', () => { await new Promise((resolve) => setTimeout(resolve, 10)) } handle.close() + await services.watcher.waitForIdle() expect(summaries.length).toBeGreaterThan(0) expect(summaries[0]).toContain('planned\tPages/q3-planning.md') diff --git a/packages/cli/src/commands/agent.ts b/packages/cli/src/commands/agent.ts index 93b884fb5..a85cfc5c0 100644 --- a/packages/cli/src/commands/agent.ts +++ b/packages/cli/src/commands/agent.ts @@ -13,8 +13,8 @@ * - skill: print the cross-harness SKILL.md */ -import type { EntrySearch } from '@xnetjs/brain' import type { AgentBackend } from '../utils/agent-remote.js' +import type { EntrySearch } from '@xnetjs/brain' import type { AiMutationPlan, AiSurfaceService, @@ -716,7 +716,7 @@ export function startDaemon(services: AgentCliServices, options: DaemonOptions): usePolling: options.poll, ...(options.pollIntervalMs !== undefined ? { pollIntervalMs: options.pollIntervalMs } : {}) }, - (scan) => void handleScan(scan) + handleScan ) return handle } @@ -957,9 +957,24 @@ export function registerAgentCommands( const services = await createServices({ ...options, forWrites: Boolean(options.apply) }) const handle = startDaemon(services, options) console.log(`watching ${resolve(options.dir)} (ctrl-c to stop)`) - process.on('SIGINT', () => { + process.once('SIGINT', () => { handle.close() - void services.dispose?.().finally(() => process.exit(0)) + void (async () => { + let exitCode = 0 + try { + await services.watcher.waitForIdle() + } catch (error) { + console.error('Workspace watcher did not shut down cleanly:', error) + exitCode = 1 + } + try { + await services.dispose?.() + } catch (error) { + console.error('Workspace services did not shut down cleanly:', error) + exitCode = 1 + } + process.exit(exitCode) + })() }) }) diff --git a/packages/comms/src/buzz/buzz.test.ts b/packages/comms/src/buzz/buzz.test.ts index 5f9db96c9..f9004e16a 100644 --- a/packages/comms/src/buzz/buzz.test.ts +++ b/packages/comms/src/buzz/buzz.test.ts @@ -82,7 +82,10 @@ describe('NIP-19 npub decoding (exploration 0416)', () => { it('rejects malformed, mis-prefixed, and mixed-case input', () => { expect(decodeNpub('not-an-npub')).toBeNull() expect(decodeNpub('')).toBeNull() - expect(decodeNpub(npub.slice(0, -1) + 'q')).toBeNull() // bad checksum + // The random valid checksum can already end in q; always change the symbol. + const malformed = npub.slice(0, -1) + (npub.endsWith('q') ? 'p' : 'q') + expect(malformed).not.toBe(npub) + expect(decodeNpub(malformed)).toBeNull() expect(decodeNpub(npub.toUpperCase().slice(0, 4) + npub.slice(4))).toBeNull() // mixed case expect(decodeBech32(npub.replace('npub', 'nsec'))).toBeNull() // checksum binds the hrp }) diff --git a/packages/data/etc/data.api.md b/packages/data/etc/data.api.md index 8fbf7091d..47870286e 100644 --- a/packages/data/etc/data.api.md +++ b/packages/data/etc/data.api.md @@ -897,6 +897,7 @@ export type BuiltInSchemaIRI = keyof typeof builtInSchemas; export const builtInSchemas: { readonly 'xnet://xnet.fyi/Page@1.0.0': () => Promise; + sourceResources: PropertyBuilder; icon: PropertyBuilder; cover: PropertyBuilder; folder: PropertyBuilder; @@ -2216,6 +2217,7 @@ export const builtInSchemas: { }>>; readonly 'xnet://xnet.fyi/Page': () => Promise; + sourceResources: PropertyBuilder; icon: PropertyBuilder; cover: PropertyBuilder; folder: PropertyBuilder; @@ -8192,6 +8194,7 @@ export class NodeStore { query(descriptor: NodeQueryDescriptor): Promise; // (undocumented) rebuildIndexesForSchemas(schemaIds: readonly SchemaIRI[]): Promise; + refreshPersistedNodes(nodeIds: readonly NodeId[]): Promise; restore(id: NodeId): Promise; searchText(query: string, limit: number, options?: NodeTextSearchOptions): Promise; setCheckedOutDraft(overlay: CheckedOutDraftOverlay | null): void; @@ -8460,6 +8463,7 @@ export type Page = InferNode<(typeof PageSchema)['_properties']>; // @public (undocumented) export const PageSchema: DefinedSchema<{ title: PropertyBuilder; + sourceResources: PropertyBuilder; icon: PropertyBuilder; cover: PropertyBuilder; folder: PropertyBuilder; @@ -11510,7 +11514,7 @@ export { YXmlText } // Warnings were encountered during analysis: // -// dist/types-DmwyWSm5.d.ts:444:9 - (ae-forgotten-export) The symbol "GrantStatus" needs to be exported by the entry point index.d.ts +// dist/types-CjDvN0zw.d.ts:444:9 - (ae-forgotten-export) The symbol "GrantStatus" needs to be exported by the entry point index.d.ts // (No @packageDocumentation comment for this package) diff --git a/packages/data/src/schema/schemas/page.test.ts b/packages/data/src/schema/schemas/page.test.ts new file mode 100644 index 000000000..98c8ff4c9 --- /dev/null +++ b/packages/data/src/schema/schemas/page.test.ts @@ -0,0 +1,17 @@ +import type { DID } from '../node' +import { expect, it } from 'vitest' +import { PageSchema } from './page' + +const author = 'did:key:fixture' as DID +it('accepts old Pages without source relations and private notes with multiple citations', () => { + const old = PageSchema.create({ title: 'Existing page' }, { createdBy: author }) + expect(PageSchema.validate(old).valid).toBe(true) + const note = PageSchema.create( + { title: 'My note', sourceResources: ['video-1', 'repository-2'], visibility: 'private' }, + { createdBy: author } + ) + expect(PageSchema.validate(note).valid).toBe(true) + expect(note.sourceResources).toEqual(['video-1', 'repository-2']) + expect(note.visibility).toBe('private') + expect(note.publishedAt).toBeUndefined() +}) diff --git a/packages/data/src/schema/schemas/page.ts b/packages/data/src/schema/schemas/page.ts index cf90d9f36..06b8a0c01 100644 --- a/packages/data/src/schema/schemas/page.ts +++ b/packages/data/src/schema/schemas/page.ts @@ -19,6 +19,9 @@ export const PageSchema = defineSchema({ /** Page title */ title: text({ required: true, maxLength: 500 }), + /** Resources cited by a personal note or guide. Visibility remains independent. */ + sourceResources: relation({ multiple: true }), + /** Emoji or icon URL */ icon: text({}), diff --git a/packages/data/src/store/shared-storage.test.ts b/packages/data/src/store/shared-storage.test.ts new file mode 100644 index 000000000..3ae299d5b --- /dev/null +++ b/packages/data/src/store/shared-storage.test.ts @@ -0,0 +1,69 @@ +import type { SchemaIRI } from '../schema/node' +import type { DID } from '@xnetjs/core' +import { generateSigningKeyPair } from '@xnetjs/crypto' +import { createDID } from '@xnetjs/identity' +import { expect, it, vi } from 'vitest' +import { MemoryNodeStorageAdapter } from './memory-adapter' +import { NodeStore } from './store' + +const schemaId = 'xnet://fixture/Page' as SchemaIRI +async function paired(legacy = false) { + const key = generateSigningKeyPair() + const storage = new MemoryNodeStorageAdapter() + if (legacy) Object.defineProperty(storage, 'applyNodeBatch', { value: undefined }) + const options = { + storage, + authorDID: createDID(key.publicKey) as DID, + signingKey: key.privateKey + } + const renderer = new NodeStore(options), + native = new NodeStore(options) + await renderer.initialize() + await native.initialize() + for (let index = 0; index < 10; index++) + await native.create({ schemaId, properties: { title: `Earlier ${index}` } }) + return { renderer, native, storage } +} + +it.each([false, true])( + 'orders later local edits after a native import (legacy=%s)', + async (legacy) => { + const { renderer, native, storage } = await paired(legacy) + const page = await native.create({ schemaId, properties: { title: 'Captured title' } }) + const importedClock = await storage.getLastLamportTime() + const result = await renderer.update(page.id, { properties: { title: 'My edited title' } }) + expect(result.properties.title).toBe('My edited title') + expect(result.timestamps.title.lamport).toBeGreaterThan(importedClock) + expect(await storage.getLastLamportTime()).toBeGreaterThan(importedClock) + } +) + +it('refreshes query and node subscribers without reapplying signed changes', async () => { + const { renderer, native, storage } = await paired() + const page = await native.create({ schemaId, properties: { title: 'Imported note' } }) + const all = vi.fn(), + one = vi.fn() + renderer.subscribe(all) + renderer.subscribeToNode(page.id, one) + const changes = await storage.getAllChanges() + await renderer.refreshPersistedNodes([page.id, page.id]) + expect(all).toHaveBeenCalledOnce() + expect(one).toHaveBeenCalledOnce() + expect(one.mock.calls[0][0]).toMatchObject({ node: { id: page.id }, isRemote: true }) + expect(await storage.getAllChanges()).toEqual(changes) + await expect(renderer.refreshPersistedNodes(['missing'])).rejects.toThrow('missing or unreadable') + expect(all).toHaveBeenCalledOnce() +}) + +it('uses the persisted clock for a batch from a still-open store', async () => { + const { renderer, native } = await paired() + const page = await native.create({ schemaId, properties: { title: 'Import' } }) + await renderer.transaction([ + { type: 'update', nodeId: page.id, options: { properties: { title: 'Batch edit' } } } + ]) + expect((await renderer.get(page.id))?.properties.title).toBe('Batch edit') + await native.importDeterministicNodes([ + { id: page.id, schemaId, properties: { title: 'Later import' } } + ]) + expect((await renderer.get(page.id))?.properties.title).toBe('Later import') +}) diff --git a/packages/data/src/store/store.ts b/packages/data/src/store/store.ts index 5d3e9b492..414f3619b 100644 --- a/packages/data/src/store/store.ts +++ b/packages/data/src/store/store.ts @@ -222,7 +222,38 @@ export class NodeStore { */ async initialize(): Promise { const lastTime = await this.storage.getLastLamportTime() - this.clock = { ...this.clock, time: lastTime } + if (!Number.isSafeInteger(lastTime) || lastTime < 0) + throw new Error('Stored Lamport time is unreadable; refusing to allocate a new change.') + this.clock = { ...this.clock, time: Math.max(this.clock.time, lastTime) } + } + + private async nextLamportTime(): Promise { + // Another local store (for example the desktop import utility) can commit + // while this instance stays open. A later edit must follow that clock. + await this.initialize() + const [clock, timestamp] = tick(this.clock) + this.clock = clock + return timestamp.time + } + + /** + * Refresh subscribers after another local owner commits to the SAME storage. + * Reads persisted changes and nodes; never applies or broadcasts them again. + * Call only after commit, with IDs readable by this store's identity. + * This is a notification boundary, not a lock for concurrent writers. + */ + async refreshPersistedNodes(nodeIds: readonly NodeId[]): Promise { + await this.initialize() + const events = await Promise.all( + Array.from(new Set(nodeIds)).map(async (id) => { + this.authEvaluator?.invalidate(id) + const [change, node] = await Promise.all([this.storage.getLastChange(id), this.getRaw(id)]) + if (!change || !node) + throw new Error(`Cannot refresh a missing or unreadable persisted node: ${id}`) + return { change, node } + }) + ) + for (const { change, node } of events) this.emit(change, node, null, true) } /** @@ -294,9 +325,7 @@ export class NodeStore { const now = Date.now() // Tick the clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() // Create the change const payload: NodePayload = { @@ -722,9 +751,7 @@ export class NodeStore { const now = Date.now() // Tick the clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() // Create the change with sparse properties const payload: NodePayload = { @@ -824,9 +851,7 @@ export class NodeStore { const now = Date.now() // Tick the clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() // Create the delete change const payload: NodePayload = { @@ -904,9 +929,7 @@ export class NodeStore { const now = Date.now() // Tick the clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() // Create the restore change const payload: NodePayload = { @@ -1234,9 +1257,7 @@ export class NodeStore { const previousClock = this.clock // Tick the clock once for the entire batch - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() try { const result = await this.runTransactionOperationsBatch({ @@ -1403,9 +1424,7 @@ export class NodeStore { const batchSize = resolvedOps.length const now = Date.now() const previousClock = this.clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() try { const applyStartedAt = Date.now() @@ -1511,9 +1530,7 @@ export class NodeStore { const previousClock = this.clock const indexMode = this.resolveDeterministicImportIndexMode(options) - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() try { let result: DeterministicNodeImportAppliedPlan @@ -2477,9 +2494,7 @@ export class NodeStore { const now = Date.now() const previousClock = this.clock - const [newClock, ts] = tick(this.clock) - this.clock = newClock - const lamport = ts.time + const lamport = await this.nextLamportTime() try { const parentHash = preflight.lastChangesByNodeId.get(input.nodeId)?.hash ?? null diff --git a/packages/plugins/src/__tests__/ai-workspace-exporter.test.ts b/packages/plugins/src/__tests__/ai-workspace-exporter.test.ts index bbcdbf286..9f485c6c0 100644 --- a/packages/plugins/src/__tests__/ai-workspace-exporter.test.ts +++ b/packages/plugins/src/__tests__/ai-workspace-exporter.test.ts @@ -6,7 +6,7 @@ import type { AIProvider } from '../ai/providers' import { mkdtemp, readFile, rm, unlink, writeFile } from 'fs/promises' import { tmpdir } from 'os' import { join } from 'path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { createAiAgentRuntime } from '../ai/runtime' import { createAiWorkspaceExporter, @@ -667,8 +667,94 @@ describe('AiWorkspaceExporter', () => { await new Promise((resolve) => setTimeout(resolve, 10)) } handle.close() + await watcher.waitForIdle() expect(scans.length).toBeGreaterThan(0) expect(scans[scans.length - 1]).toBe(1) }) + + it.each(['scan', 'callback'] as const)('drains an active %s after closing', async (stage) => { + const config = createRoadmapWorkspace() + await createAiWorkspaceExporter(config).exportWorkspace({ + rootDir, + scope: { nodeIds: ['page_1'] } + }) + const watcher = createAiWorkspaceWatcher(config) + let release = () => {} + let entered = false + const held = new Promise((resolve) => { + release = resolve + }) + const hold = async () => { + entered = true + await held + } + const originalScan = watcher.scanChangedFiles.bind(watcher) + const scan = vi.spyOn(watcher, 'scanChangedFiles').mockImplementation(async (options) => { + if (stage === 'scan') await hold() + return originalScan(options) + }) + const callback = vi.fn(async () => { + if (stage === 'callback') await hold() + }) + const handle = watcher.watchWorkspace( + { rootDir, usePolling: true, pollIntervalMs: 5 }, + callback + ) + try { + await vi.waitFor(() => expect(entered).toBe(true)) + handle.close() + let drained = false + const finished = watcher.waitForIdle().then(() => { + drained = true + }) + await Promise.resolve() + expect(drained).toBe(false) + release() + await finished + expect(scan).toHaveBeenCalledTimes(1) + expect(callback).toHaveBeenCalledTimes(1) + expect(drained).toBe(true) + expect( + JSON.parse(await readFile(join(rootDir, '.xnet/review/index.json'), 'utf8')) + ).toMatchObject({ rootDir, entries: [] }) + } finally { + handle.close() + release() + await watcher.waitForIdle() + scan.mockRestore() + } + }) + + it.each(['scan', 'callback'] as const)( + 'retains a background %s failure for shutdown', + async (stage) => { + const config = createRoadmapWorkspace() + await createAiWorkspaceExporter(config).exportWorkspace({ + rootDir, + scope: { nodeIds: ['page_1'] } + }) + const watcher = createAiWorkspaceWatcher(config) + const failure = new Error('fixture watch failure') + const scan = vi.spyOn(watcher, 'scanChangedFiles') + if (stage === 'scan') scan.mockRejectedValue(failure) + const logged = vi.spyOn(console, 'error').mockImplementation(() => {}) + const handle = watcher.watchWorkspace( + { rootDir, usePolling: true, pollIntervalMs: 5 }, + async () => { + if (stage === 'callback') throw failure + } + ) + try { + await vi.waitFor(() => expect(logged).toHaveBeenCalled()) + handle.close() + await expect(watcher.waitForIdle()).rejects.toMatchObject({ errors: [failure] }) + expect(scan).toHaveBeenCalledTimes(1) + } finally { + handle.close() + scan.mockRestore() + logged.mockRestore() + } + } + ) }) diff --git a/packages/plugins/src/services/ai-workspace-exporter.ts b/packages/plugins/src/services/ai-workspace-exporter.ts index 6bf1a296a..7e4626659 100644 --- a/packages/plugins/src/services/ai-workspace-exporter.ts +++ b/packages/plugins/src/services/ai-workspace-exporter.ts @@ -560,6 +560,8 @@ export class AiWorkspaceExporter { export class AiWorkspaceWatcher { private readonly aiSurface: AiSurfaceService private readonly clock: () => Date + private readonly activeWatches = new Set>() + private readonly watchFailures: unknown[] = [] constructor(private readonly config: AiWorkspaceExporterConfig) { this.aiSurface = @@ -572,6 +574,13 @@ export class AiWorkspaceWatcher { this.clock = config.clock ?? (() => new Date()) } + /** Close watch handles first, then wait for their scans and async callbacks before disposal. */ + async waitForIdle(): Promise { + while (this.activeWatches.size > 0) await Promise.all([...this.activeWatches]) + if (this.watchFailures.length > 0) + throw new AggregateError(this.watchFailures, 'AI workspace watching failed before shutdown') + } + async scanChangedFiles( options: AiWorkspaceWatcherScanOptions ): Promise { @@ -685,30 +694,53 @@ export class AiWorkspaceWatcher { let pollTimer: ReturnType | null = null let watcher: FSWatcher | null = null let closed = false + let scanning = false + let rescan = false - const scheduleScan = (): void => { + const close = (): void => { + closed = true if (timer) clearTimeout(timer) - timer = setTimeout(() => { - void this.scanChangedFiles(options).then(onScan) - }, 250) + if (pollTimer) clearInterval(pollTimer) + watcher?.close() } - // Polling ticks scan directly (with an overlap guard) instead of going - // through the debounce, which a fast poll interval would starve forever. - let scanning = false - const pollScan = (): void => { - if (scanning) return + const scan = (): void => { + if (closed) return + if (scanning) { + rescan = true + return + } scanning = true - void this.scanChangedFiles(options) + const work = this.scanChangedFiles(options) .then(onScan) + .catch((error: unknown) => { + this.watchFailures.push(error) + close() + console.error('[AiWorkspaceWatcher] Watch stopped after a failed scan:', error) + }) .finally(() => { + this.activeWatches.delete(work) scanning = false + if (rescan && !closed) { + rescan = false + scheduleScan() + } }) + this.activeWatches.add(work) + } + + const scheduleScan = (): void => { + if (closed) return + if (timer) clearTimeout(timer) + timer = setTimeout(scan, 250) } const startPolling = (): void => { if (closed || pollTimer) return - pollTimer = setInterval(pollScan, options.pollIntervalMs ?? 2000) + // Polling bypasses debounce so a short interval cannot starve the scan. + pollTimer = setInterval(() => { + if (!scanning) scan() + }, options.pollIntervalMs ?? 2000) } // fs.watch recursive is reliable on macOS/Windows but historically flaky @@ -744,12 +776,7 @@ export class AiWorkspaceWatcher { } return { - close: () => { - closed = true - if (timer) clearTimeout(timer) - if (pollTimer) clearInterval(pollTimer) - watcher?.close() - }, + close, isPolling: () => pollTimer !== null } } diff --git a/packages/react/etc/react.api.md b/packages/react/etc/react.api.md index 70360fff0..cf98ff940 100644 --- a/packages/react/etc/react.api.md +++ b/packages/react/etc/react.api.md @@ -3476,7 +3476,7 @@ export interface YDocRegistryLike { // Warnings were encountered during analysis: // -// dist/core.d.ts:347:5 - (ae-forgotten-export) The symbol "ErrorBoundaryFallbackProps" needs to be exported by the entry point index.d.ts +// dist/core.d.ts:380:5 - (ae-forgotten-export) The symbol "ErrorBoundaryFallbackProps" needs to be exported by the entry point index.d.ts // dist/experimental-S9T6neyx.d.ts:98:5 - (ae-forgotten-export) The symbol "TaskStatus" needs to be exported by the entry point index.d.ts // dist/experimental-S9T6neyx.d.ts:103:5 - (ae-forgotten-export) The symbol "TaskNode" needs to be exported by the entry point index.d.ts // dist/index.d.ts:80:5 - (ae-forgotten-export) The symbol "QueryStatus" needs to be exported by the entry point index.d.ts diff --git a/packages/react/src/hooks/document-writes.test.ts b/packages/react/src/hooks/document-writes.test.ts new file mode 100644 index 000000000..e1308f25b --- /dev/null +++ b/packages/react/src/hooks/document-writes.test.ts @@ -0,0 +1,103 @@ +import { describe, expect, it } from 'vitest' +import { createDocumentWriteBarrier } from './document-writes' + +describe('document write barrier', () => { + it('waits for debounced editors and pending storage acknowledgements', async () => { + const barrier = createDocumentWriteBarrier() + const events: string[] = [] + let release!: () => void + const disk = new Promise((resolve) => { + release = resolve + }) + barrier.register(() => + barrier.write('note', async () => { + await disk + events.push('saved') + }) + ) + const flush = barrier.flush().then(() => events.push('flushed')) + await Promise.resolve() + expect(events).toEqual([]) + release() + await flush + expect(events).toEqual(['saved', 'flushed']) + }) + + it('prevents an earlier slow save from overwriting a newer edit', async () => { + const barrier = createDocumentWriteBarrier() + const written: string[] = [] + let release!: () => void + const disk = new Promise((resolve) => { + release = resolve + }) + const old = barrier.write('note', async () => { + await disk + written.push('old') + }) + const fresh = barrier.write('note', async () => { + written.push('new') + }) + await Promise.resolve() + expect(written).toEqual([]) + release() + await Promise.all([old, fresh]) + expect(written).toEqual(['old', 'new']) + }) + + it('keeps an unmounted failed write retryable and never reports a failed flush as saved', async () => { + const barrier = createDocumentWriteBarrier() + let full = true + let persisted = false + await expect( + barrier.write('unmounted', async () => { + if (full) throw new Error('disk full') + persisted = true + }) + ).rejects.toThrow('disk full') + await expect(barrier.flush()).rejects.toThrow('Document writes failed') + full = false + await barrier.flush() + expect(persisted).toBe(true) + }) + + it('retries retained bytes before a reopened document reads storage', async () => { + const barrier = createDocumentWriteBarrier() + let blocked = true + let stored = 'old' + await expect( + barrier.write('note', async () => { + if (blocked) throw new Error('disk full') + stored = 'latest' + }) + ).rejects.toThrow() + await expect(barrier.flushKey('note')).rejects.toThrow('disk full') + blocked = false + await barrier.flushKey('note') + expect(stored).toBe('latest') + }) + + it('does not replay a failed old snapshot over a successful newer save', async () => { + const barrier = createDocumentWriteBarrier() + const written: string[] = [] + await expect( + barrier.write('note', async () => { + throw new Error('offline disk') + }) + ).rejects.toThrow() + await barrier.write('note', async () => { + written.push('latest') + }) + await barrier.flush() + expect(written).toEqual(['latest']) + }) + + it('rejects encoding failures and unregisters closed editors', async () => { + const barrier = createDocumentWriteBarrier() + const unregister = barrier.register(async () => { + throw new Error('cannot encode') + }) + await expect(barrier.flush()).rejects.toThrow('Could not flush every open document') + unregister() + await expect(barrier.flush()).resolves.toBeUndefined() + }) +}) diff --git a/packages/react/src/hooks/document-writes.ts b/packages/react/src/hooks/document-writes.ts new file mode 100644 index 000000000..890798c7b --- /dev/null +++ b/packages/react/src/hooks/document-writes.ts @@ -0,0 +1,86 @@ +type Write = () => Promise + +/** Serializes each document and retains failed writes for an explicit retry. */ +export function createDocumentWriteBarrier() { + const tails = new Map>() + const failures = new Map() + const flushers = new Set() + + const write = (key: string, operation: Write): Promise => { + const previous = tails.get(key) + const next = (previous ? previous.catch(() => undefined) : Promise.resolve()).then(operation) + tails.set(key, next) + // Attach both handlers immediately, including for unmount writes whose caller is gone. + void next.then( + () => { + failures.delete(key) + if (tails.get(key) === next) tails.delete(key) + }, + (error: unknown) => { + failures.set(key, { write: operation, error }) + if (tails.get(key) === next) tails.delete(key) + } + ) + return next + } + + return { + write, + async flushKey(key: string): Promise { + while (tails.has(key)) await tails.get(key)?.catch(() => undefined) + const failed = failures.get(key) + if (failed) await write(key, failed.write) + }, + register(flush: Write): () => void { + flushers.add(flush) + return () => flushers.delete(flush) + }, + async flush(): Promise { + while (tails.size) await Promise.allSettled([...tails.values()]) + // Unmounted documents have no active hook to retry their retained bytes. + await Promise.allSettled([...failures].map(([key, failed]) => write(key, failed.write))) + const active = await Promise.allSettled(Array.from(flushers, (flush) => flush())) + while (tails.size) await Promise.allSettled([...tails.values()]) + if (failures.size) + throw new AggregateError( + [...failures.values()].map((item) => item.error), + 'Document writes failed. Your unsaved content is retained.' + ) + const rejected = active.filter( + (result): result is PromiseRejectedResult => result.status === 'rejected' + ) + if (rejected.length) + throw new AggregateError( + rejected.map((result) => result.reason), + 'Could not flush every open document. Retry before closing.' + ) + } + } +} + +const documents = createDocumentWriteBarrier() +const storeIds = new WeakMap() +let nextStoreId = 0 + +function keyFor(store: object, id: string): string { + if (!storeIds.has(store)) storeIds.set(store, ++nextStoreId) + return `${storeIds.get(store)}:${id}` +} + +export function persistDocument( + store: { setDocumentContent(id: string, content: Uint8Array): Promise }, + id: string, + content: Uint8Array +): Promise { + const retained = content.slice() + return documents.write(`document:${keyFor(store, id)}`, () => + store.setDocumentContent(id, retained) + ) +} + +export const registerDocumentFlush = documents.register +export const flushDocumentWrites = documents.flush + +export function retryDocumentWrite(store: object, id: string): Promise { + return documents.flushKey(`document:${keyFor(store, id)}`) +} diff --git a/packages/react/src/hooks/useNode.ts b/packages/react/src/hooks/useNode.ts index a59549d33..866654903 100644 --- a/packages/react/src/hooks/useNode.ts +++ b/packages/react/src/hooks/useNode.ts @@ -30,16 +30,16 @@ * update({ typo: 'x' }) // Type error! * ``` */ -import type { SyncManager } from '@xnetjs/runtime' import type { DefinedSchema, PropertyBuilder, InferCreateProps } from '@xnetjs/data' +import type { SyncManager } from '@xnetjs/runtime' +import { METABRIDGE_ORIGIN, METABRIDGE_SEED_ORIGIN, WebSocketSyncProvider } from '@xnetjs/runtime' import { useState, useEffect, useCallback, useRef, useMemo } from 'react' import { Awareness } from 'y-protocols/awareness' import * as Y from 'yjs' import { useDataBridge, useXNetInternal } from '../context' import { useInstrumentation } from '../instrumentation' -import { METABRIDGE_ORIGIN, METABRIDGE_SEED_ORIGIN } from '@xnetjs/runtime' -import { WebSocketSyncProvider } from '@xnetjs/runtime' import { flattenNode, type FlatNode } from '../utils/flattenNode' +import { persistDocument, registerDocumentFlush, retryDocumentWrite } from './document-writes' import { useNodeStore } from './useNodeStore' import { useSyncManager } from './useSyncManager' @@ -214,14 +214,6 @@ function hasSyncManagerSetter( // Hook Implementation // ============================================================================= -/** - * Module-level map tracking in-flight save promises by document ID. - * When a useNode instance unmounts, it stores its flush promise here. - * The next useNode instance loading the same doc awaits this before reading, - * ensuring content survives navigation. - */ -const pendingFlushes = new Map>() - /** High-resolution clock with a Date.now fallback for non-DOM runtimes. */ function nowMs(): number { return typeof performance !== 'undefined' ? performance.now() : Date.now() @@ -303,6 +295,7 @@ export function useNode

>( const providerRef = useRef(null) const saveTimeoutRef = useRef | null>(null) const docRef = useRef(null) + const editRevision = useRef(0) const creatingRef = useRef(false) const storeRef = useRef(store) storeRef.current = store @@ -360,10 +353,7 @@ export function useNode

>( try { // Await any in-flight flush from a previous unmount to ensure // we read the latest persisted content (race condition on navigation) - const pendingFlush = pendingFlushes.get(id) - if (pendingFlush) { - await pendingFlush - } + await retryDocumentWrite(store, id) // Load node properties let node = await store.get(id) @@ -527,16 +517,21 @@ export function useNode

>( const save = useCallback(async () => { if (!store || !id || !docRef.current) return - // Clear the timeout ref since we're about to save + if (saveTimeoutRef.current) clearTimeout(saveTimeoutRef.current) saveTimeoutRef.current = null + const revision = editRevision.current try { const content = Y.encodeStateAsUpdate(docRef.current) - await store.setDocumentContent(id, content) - setIsDirty(false) + await persistDocument(store, id, content) + if (revision === editRevision.current) { + setIsDirty(false) + setError(null) + } setLastSavedAt(Date.now()) } catch (err) { setError(err instanceof Error ? err : new Error(String(err))) + throw err } }, [store, id]) @@ -544,8 +539,11 @@ export function useNode

>( const saveRef = useRef(save) saveRef.current = save + useEffect(() => registerDocumentFlush(() => saveRef.current()), [store, id]) + // Debounced save - use ref to avoid dependency changes const scheduleSave = useCallback(() => { + editRevision.current += 1 setIsDirty(true) if (saveTimeoutRef.current) { @@ -553,7 +551,9 @@ export function useNode

>( } saveTimeoutRef.current = setTimeout(() => { - saveRef.current() + void saveRef.current().catch(() => { + /* The hook exposes the error; the barrier retains the write. */ + }) }, persistDebounce) }, [persistDebounce]) @@ -1046,16 +1046,10 @@ export function useNode

>( // This ensures content survives navigation regardless of sync mode. if (docRef.current && store && id) { const content = Y.encodeStateAsUpdate(docRef.current) - const flushPromise = store - .setDocumentContent(id, content) - .catch(() => { - // Silent fail on unmount - }) - .finally(() => { - pendingFlushes.delete(id) - }) - // Store the flush promise so the next load() can await it - pendingFlushes.set(id, flushPromise) + const flushPromise = persistDocument(store, id, content) + void flushPromise.catch((error: unknown) => + console.error('[useNode] Unsaved document retained:', error) + ) } // Release doc back to DataBridge (or SyncManager) on unmount @@ -1089,8 +1083,8 @@ export function useNode

>( try { const content = Y.encodeStateAsUpdate(docRef.current) // Fire and forget - we can't await in beforeunload - store.setDocumentContent(id, content).catch(() => { - // Silent fail - page is unloading anyway + void persistDocument(store, id, content).catch((error: unknown) => { + console.error('[useNode] Document flush failed:', error) }) } catch { // Silent fail on encoding error diff --git a/packages/react/src/internal.ts b/packages/react/src/internal.ts index 50a2462b4..7d9070e6f 100644 --- a/packages/react/src/internal.ts +++ b/packages/react/src/internal.ts @@ -21,3 +21,4 @@ export { export { useDataBridge, useXNet, useXNetInternal } from './context' export type { XNetContextValue, XNetInternalContextValue } from './context' export type { XNetRuntimeStatus } from './runtime' +export { flushDocumentWrites } from './hooks/document-writes' diff --git a/packages/social/src/__tests__/garden.test.ts b/packages/social/src/__tests__/garden.test.ts new file mode 100644 index 000000000..c82401cea --- /dev/null +++ b/packages/social/src/__tests__/garden.test.ts @@ -0,0 +1,110 @@ +import { mkdtemp, writeFile, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { expect, it } from 'vitest' +import { createSocialNodeId } from '../import/ids' +import { resourceIdentityForUrl } from '../import/resource-url' +import { openSocialImportSource } from '../import/source-reader' +import { gardenAdapter, mapGarden } from '../importers/garden' +import { builtInSocialImportAdapters } from '../importers/registry' + +const context = { archiveId: 'source-a', importedAt: '2026-09-29T00:00:00Z' } +const source = { path: 'garden.json', byteSize: 100 } +const entry = { + url: 'https://youtu.be/abcdefghijk?t=15', + title: 'A useful talk', + note: 'My own words about why this matters.', + tags: ['learning'], + category: 'Videos', + source: 'https://bsky.app/profile/example.test/post/one', + createdAt: '2026-09-01T00:00:00Z' +} +const snapshot = (entries: unknown[] = [entry]) => ({ + version: 1, + profile: { did: 'did:plc:fixture', handle: 'example.test' }, + entries +}) +const map = (input: unknown) => mapGarden({ context, source, input }) + +it('reuses exported resource identities while keeping authored commentary separate and private', () => { + const rows = map(snapshot()) + const content = rows.filter((row) => row.kind === 'content') + expect(content).toHaveLength(2) + expect(content[0].deterministicId).toBe( + createSocialNodeId('content', ['youtube', 'video', 'abcdefghijk']) + ) + expect(content[0].properties.searchText).toBeUndefined() + expect(content[1].properties.parentContent).toBe(content[0].deterministicId) + expect(content[1].properties.searchText).toBe(entry.note) + expect(content[1].properties.canonicalUrl).toBe(entry.source) + expect( + rows + .filter((row) => row.kind !== 'source-record') + .every((row) => row.properties.visibility === 'private') + ).toBe(true) + expect(rows.filter((row) => row.kind === 'collection-item')).toHaveLength(3) +}) + +it('retains repeated-resource observations from distinct source posts and stable identities across snapshots', () => { + const rows = map(snapshot([entry, { ...entry, source: entry.source + 'two' }])) + const content = rows.filter((row) => row.kind === 'content') + expect( + new Set( + content.filter((row) => !row.properties.parentContent).map((row) => row.deterministicId) + ).size + ).toBe(1) + expect( + new Set(content.filter((row) => row.properties.parentContent).map((row) => row.deterministicId)) + .size + ).toBe(2) + const overlap = mapGarden({ + context: { ...context, archiveId: 'source-b' }, + source, + input: snapshot() + }) + expect( + overlap.filter((row) => row.kind !== 'source-record').map((row) => row.deterministicId) + ).toEqual( + map(snapshot()) + .filter((row) => row.kind !== 'source-record') + .map((row) => row.deterministicId) + ) +}) + +it('preserves generic query/fragment meaning and recognizes provider aliases', () => { + expect(resourceIdentityForUrl('https://example.test/?q=one#section').url).toBe( + 'https://example.test/?q=one#section' + ) + expect(resourceIdentityForUrl('https://instagram.com/reel/abcd/?igsh=tracking').id).toBe( + createSocialNodeId('content', ['instagram', 'post', 'abcd']) + ) + expect(resourceIdentityForUrl('https://twitter.com/example/status/123').id).toBe( + createSocialNodeId('content', ['x', 'tweet', '123']) + ) + expect(() => resourceIdentityForUrl('file:///private/notes')).toThrow() + expect(() => resourceIdentityForUrl('https://user:secret@example.test/')).toThrow() +}) + +it('fails on unsupported versions, duplicate observations, invalid dates, or oversized notes', () => { + expect(() => map({ ...snapshot(), version: 2 })).toThrow() + expect(() => map(snapshot([entry, entry]))).toThrow('repeats') + expect(() => map(snapshot([{ ...entry, createdAt: 'not-a-date' }]))).toThrow('timestamp') + expect(() => map(snapshot([{ ...entry, note: 'x'.repeat(20001) }]))).toThrow( + 'no text was truncated' + ) +}) + +it('detects a standalone garden JSON through the actual registry and source reader', async () => { + const root = await mkdtemp(join(tmpdir(), 'xnet-garden-')) + try { + const path = join(root, 'renamed-export.json') + await writeFile(path, JSON.stringify(snapshot())) + const opened = await openSocialImportSource(path) + expect(opened.manifest.entries[0].path).toBe('garden.json') + expect(builtInSocialImportAdapters).toContain(gardenAdapter) + expect(await gardenAdapter.detect(opened.manifest)).toBeGreaterThan(0.9) + expect(await opened.readJsonEntry('garden.json')).toEqual(snapshot()) + } finally { + await rm(root, { recursive: true, force: true }) + } +}) diff --git a/packages/social/src/__tests__/github-stars.test.ts b/packages/social/src/__tests__/github-stars.test.ts new file mode 100644 index 000000000..7b0eafd21 --- /dev/null +++ b/packages/social/src/__tests__/github-stars.test.ts @@ -0,0 +1,144 @@ +import type { StagedSocialRecord } from '../import/types' +import { mkdtemp, writeFile, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { openSocialImportSource } from '../import/source-reader' +import { githubAdapter, mapGitHubStars } from '../importers/github' +import { SocialActorSchema } from '../schemas/actor' +import { SocialCollectionItemSchema, SocialCollectionSchema } from '../schemas/collection' +import { SocialContentSchema } from '../schemas/content' +import { SocialInteractionSchema } from '../schemas/interaction' + +const context = { + archiveId: 'archive:github', + observedBy: 'did:key:fixture', + importedAt: '2026-09-01T00:00:00Z' +} +const source = { path: 'github-stars.json', byteSize: 100 } +const repo = { + id: 42, + full_name: 'example/useful-tool', + html_url: 'https://github.com/example/useful-tool', + description: 'A complete searchable description.', + private: false, + topics: ['knowledge'], + language: 'TypeScript', + owner: { + id: 7, + login: 'example', + html_url: 'https://github.com/example', + avatar_url: 'https://avatars.githubusercontent.com/u/7' + } +} +const star = { starred_at: '2025-05-01T12:00:00Z', repo } +const snapshot = (stars: unknown[]) => ({ + format: 'xnet-github-stars/1', + account: 'fixture-user', + coverage: { paginationCompleted: true, atomic: false }, + stars +}) +const kind = (records: StagedSocialRecord[], value: StagedSocialRecord['kind']) => + records.filter((record) => record.kind === value) +const map = (input: unknown) => mapGitHubStars({ context, source, input }) + +describe('GitHub stars', () => { + it('retains native repository and star identity, metadata, and private defaults', () => { + const records = map(snapshot([star])) + const content = kind(records, 'content')[0] + expect(content.properties.platformContentId).toBe('42') + expect(content.properties.searchText).toContain(repo.description) + const event = kind(records, 'interaction')[0] + expect(event.properties.interactionKind).toBe('star') + expect(event.properties.publishedAt).toBe(star.starred_at) + expect(event.properties.importedAt).toBe(context.importedAt) + expect( + records + .filter((record) => record.kind !== 'source-record') + .every((record) => record.properties.visibility === 'private') + ).toBe(true) + const schemas = { + actor: SocialActorSchema, + content: SocialContentSchema, + interaction: SocialInteractionSchema, + collection: SocialCollectionSchema, + 'collection-item': SocialCollectionItemSchema + } + for (const record of records) { + if (!(record.kind in schemas)) continue + const schema = schemas[record.kind as keyof typeof schemas] + for (const name of Object.keys(record.properties)) + expect( + schema.schema.properties.some((property) => property.name === name), + `${record.kind}.${name}` + ).toBe(true) + } + }) + + it('survives repository renames and overlapping snapshots without duplicating a star', () => { + const before = map(snapshot([star])) + const after = map( + snapshot([ + { + ...star, + repo: { + ...repo, + full_name: 'new-owner/renamed', + html_url: 'https://github.com/new-owner/renamed' + } + } + ]) + ) + for (const type of ['content', 'interaction', 'collection-item'] as const) + expect(kind(after, type)[0].deterministicId).toBe(kind(before, type)[0].deterministicId) + const restarred = map(snapshot([{ ...star, starred_at: '2026-01-01T00:00:00Z' }])) + expect(kind(restarred, 'interaction')[0].deterministicId).not.toBe( + kind(before, 'interaction')[0].deterministicId + ) + }) + + it('does not invent timestamps or complete coverage for a plain API array', () => { + const records = map([repo]) + expect(kind(records, 'interaction')[0].properties.publishedAt).toBeUndefined() + expect(kind(records, 'collection-item')[0].properties.addedAt).toBeUndefined() + expect(kind(records, 'interaction')[0].warnings).toContain( + 'Native star timestamp is unavailable.' + ) + expect(kind(records, 'collection')[0].warnings[0]).toContain('may be incomplete') + }) + + it('keeps private and unknown repository privacy private', () => { + for (const privateValue of [true, undefined]) { + const records = map([{ ...repo, private: privateValue }]) + expect(kind(records, 'content')[0].privacyClass).toBe('private') + } + }) + + it('rejects malformed rows and unknown formats explicitly', () => { + expect(() => map([{ ...repo, id: undefined }])).toThrow('native ID') + expect(() => map([{ repo, starred_at: 'yesterday' }])).toThrow('timestamp') + expect(() => map({ format: 'unknown', stars: [] })).toThrow('Unsupported') + }) + + it('opens a standalone snapshot through the same archive contract', async () => { + const directory = await mkdtemp(join(tmpdir(), 'xnet-github-fixture-')) + try { + const path = join(directory, 'my-snapshot.json') + await writeFile(path, JSON.stringify(snapshot([star]))) + const opened = await openSocialImportSource(path) + expect(opened.manifest.filename).toBe('my-snapshot.json') + expect(opened.manifest.archiveHash).toMatch(/^[a-f0-9]{64}$/) + expect(githubAdapter.detect(opened.manifest)).toBeGreaterThan(0) + const staged = [] + for await (const row of githubAdapter.stage( + { ...context, ...opened }, + { buckets: ['github.stars'], includeSensitive: true } + )) + staged.push(row) + expect(kind(staged, 'content')).toHaveLength(1) + await expect(opened.readJsonEntry('../another-file')).rejects.toThrow('not found') + } finally { + await rm(directory, { recursive: true }) + } + }) +}) diff --git a/packages/social/src/__tests__/importers.test.ts b/packages/social/src/__tests__/importers.test.ts index 3f85f0f87..4c42ff3b5 100644 --- a/packages/social/src/__tests__/importers.test.ts +++ b/packages/social/src/__tests__/importers.test.ts @@ -139,6 +139,8 @@ describe('social import adapters', () => { availableEntries.map((entry) => entry.id) ) expect(builtInSocialImportAdapters.map((adapter) => adapter.id)).toEqual([ + 'garden', + 'github', 'instagram', 'grok', 'youtube', diff --git a/packages/social/src/__tests__/seed-import-fidelity.test.ts b/packages/social/src/__tests__/seed-import-fidelity.test.ts new file mode 100644 index 000000000..d6e2052f8 --- /dev/null +++ b/packages/social/src/__tests__/seed-import-fidelity.test.ts @@ -0,0 +1,227 @@ +import type { SocialImportContext } from '../import/core' +import type { StagedSocialRecord } from '../import/types' +import { describe, expect, it } from 'vitest' +import { mapInstagramLikedPosts, mapInstagramSavedPosts } from '../importers/instagram' +import { redditAdapter } from '../importers/reddit' +import { mapYouTubePlaylists } from '../importers/youtube' +import { SocialContentSchema } from '../schemas/content' + +const context = { + archiveId: 'archive:fixture', + importRunId: 'run:fixture', + observedBy: 'did:key:fixture', + importedAt: '2026-09-01T00:00:00Z' +} +const source = (path: string) => ({ path, byteSize: 100 }) +const ofKind = (records: StagedSocialRecord[], kind: StagedSocialRecord['kind']) => + records.filter((record) => record.kind === kind) +const post = { + fbid: 'native-1', + timestamp: 1750000000, + label_values: [ + { label: 'URL', href: 'https://www.instagram.com/p/sample1/' }, + { label: 'Title', value: 'Example movement lesson' }, + { label: 'Caption', value: 'Full caption for finding this source later.' }, + { title: 'Creators', dict: [{ dict: [{ label: 'Username', value: 'example_creator' }] }] } + ] +} + +describe('observed seed export shapes (synthetic content)', () => { + it('shares one Instagram resource across saves, likes, and named collections', () => { + const saved = mapInstagramSavedPosts({ + context, + source: source('saved_posts.json'), + selfActorId: 'self', + input: [post] + }) + const liked = mapInstagramLikedPosts({ + context, + source: source('liked_posts.json'), + selfActorId: 'self', + input: [post] + }) + const named = mapInstagramSavedPosts({ + context, + source: source('saved_collections.json'), + selfActorId: 'self', + input: [ + { + fbid: 'collection-1', + label_values: [ + { label: 'Name', value: 'Practice' }, + { dict: [{ title: 'Item', dict: post.label_values }] } + ] + } + ] + }) + const resources = [saved, liked, named].map((records) => ofKind(records, 'content')[0]) + expect(new Set(resources.map((record) => record.deterministicId)).size).toBe(1) + expect(resources[0].properties.searchText).toContain('Full caption') + expect(resources[0].properties.actorHandle).toBe('example_creator') + for (const resource of resources) + for (const key of Object.keys(resource.properties)) + expect( + SocialContentSchema.schema.properties.some((property) => property.name === key) + ).toBe(true) + expect(ofKind(saved, 'interaction')[0].properties.interactionKind).toBe('save') + expect(ofKind(liked, 'interaction')[0].properties.interactionKind).toBe('like') + expect(ofKind(named, 'interaction')).toHaveLength(0) + expect(ofKind(named, 'collection')[0].properties.title).toBe('Practice') + expect(ofKind(named, 'collection')[0].privacyClass).toBe('private') + expect(ofKind(named, 'collection-item')).toHaveLength(1) + }) + + it('preserves repeated Instagram collection memberships and reimports deterministically', () => { + const input = { + context, + source: source('saved_collections.json'), + selfActorId: 'self', + input: [ + { + fbid: 'collection-1', + label_values: [ + { label: 'Name', value: 'Practice' }, + { dict: [{ dict: post.label_values }, { dict: post.label_values }] } + ] + } + ] + } + const records = mapInstagramSavedPosts(input) + expect( + new Set(ofKind(records, 'collection-item').map((record) => record.deterministicId)).size + ).toBe(2) + expect(mapInstagramSavedPosts(input)).toEqual(records) + expect( + ofKind(records, 'collection-item').every((record) => record.properties.addedAt === undefined) + ).toBe(true) + }) + + it('reads the liked-comments wrapper and refuses unknown wrappers', () => { + const records = mapInstagramLikedPosts({ + context, + source: source('liked_comments.json'), + selfActorId: 'self', + input: { + likes_comment_likes: [ + { + title: 'A comment', + string_list_data: [ + { + href: 'https://www.instagram.com/p/sample1/c/42/', + value: 'Liked', + timestamp: 1750000000 + } + ] + } + ] + } + }) + expect(ofKind(records, 'content')[0].properties.contentKind).toBe('comment') + expect(ofKind(records, 'interaction')).toHaveLength(1) + expect(() => + mapInstagramLikedPosts({ + context, + source: source('liked_posts.json'), + selfActorId: 'self', + input: { unsupported: [] } + }) + ).toThrow('Unsupported Instagram') + }) + + it('joins sanitized YouTube filenames without dropping repeated memberships', () => { + const input = { + context, + selfActorId: 'self', + catalogSource: source('playlists.csv'), + catalogRows: [ + { 'Playlist ID': 'PLfixture', 'Playlist Title (Original)': 'Learning / Practice' } + ], + videoFiles: [ + { + source: source('Learning _ Practice-videos.csv'), + rows: [ + { 'Video ID': 'video1', 'Playlist Video Creation Timestamp': '2025-01-01T00:00:00Z' }, + { 'Video ID': 'video1', 'Playlist Video Creation Timestamp': '2025-01-01T00:00:00Z' } + ] + } + ] + } + const records = mapYouTubePlaylists(input) + expect(ofKind(records, 'collection')).toHaveLength(1) + expect(new Set(ofKind(records, 'content').map((record) => record.deterministicId)).size).toBe(1) + expect( + new Set(ofKind(records, 'collection-item').map((record) => record.deterministicId)).size + ).toBe(2) + expect(mapYouTubePlaylists(input)).toEqual(records) + }) + + it('reports missing membership files and ambiguous catalog matches', () => { + const records = mapYouTubePlaylists({ + context, + selfActorId: 'self', + catalogSource: source('playlists.csv'), + catalogRows: [ + { 'Playlist ID': 'PLone', 'Playlist Title (Original)': 'Same / Title' }, + { 'Playlist ID': 'PLtwo', 'Playlist Title (Original)': 'Same _ Title' }, + { 'Playlist ID': 'PLempty', 'Playlist Title (Original)': 'No file' } + ], + videoFiles: [{ source: source('Same _ Title-videos.csv'), rows: [{ 'Video ID': 'video1' }] }] + }) + expect(ofKind(records, 'collection')).toHaveLength(4) + expect( + records.some((record) => record.warnings.some((warning) => warning.includes('Ambiguous'))) + ).toBe(true) + expect( + records.some((record) => + record.warnings.some((warning) => warning.includes('coverage is unknown')) + ) + ).toBe(true) + }) +}) + +it.each([false, true])( + 'merges Reddit content once with references first: %s', + async (referencesFirst) => { + const body = 'Full authored comment with a searchable final passage' + const permalink = '/r/example/comments/post/_/comment' + const files: Record = { + 'comments.csv': `id,permalink,body\ncomment,${permalink},${body}`, + 'comment_votes.csv': `id,permalink,direction\ncomment,${permalink},up`, + 'saved_comments.csv': `id,permalink\ncomment,${permalink}` + } + const context: SocialImportContext = { + archiveId: 'archive', + importRunId: 'run', + observedBy: 'did:key:fixture', + importedAt: '2026-10-02T00:00:00Z', + manifest: { + filename: 'reddit.zip', + byteSize: 1000, + entries: (referencesFirst ? Object.keys(files).reverse() : Object.keys(files)).map( + (path) => ({ path, byteSize: 100, compressedByteSize: 100 }) + ) + }, + readJsonEntry: async () => { + throw new Error('CSV only') + }, + readTextEntry: async (path) => files[path] + } + const records: StagedSocialRecord[] = [] + for await (const record of redditAdapter.stage(context, { + buckets: ['reddit.authored-content', 'reddit.votes', 'reddit.saved-hidden'], + includeSensitive: true + })) + records.push(record) + const comments = records.filter( + (record) => record.kind === 'content' && record.properties.contentKind === 'comment' + ) + expect(comments).toHaveLength(1) + expect(new Set(comments.map((record) => record.deterministicId)).size).toBe(1) + for (const comment of comments) { + expect(comment.properties.searchText).toBe(body) + expect(comment.properties.canonicalUrl).toBe('https://www.reddit.com' + permalink) + expect(comment.properties.confidence).toBe(0.95) + } + expect(records.filter((record) => record.kind === 'interaction')).toHaveLength(3) + } +) diff --git a/packages/social/src/import/core.ts b/packages/social/src/import/core.ts index 0efc65d8b..5aa5752fb 100644 --- a/packages/social/src/import/core.ts +++ b/packages/social/src/import/core.ts @@ -134,3 +134,4 @@ export type { StagedSourceRecord, StagingSummary } from './types' +export { resourceIdentityForUrl } from './resource-url' diff --git a/packages/social/src/import/node.ts b/packages/social/src/import/node.ts index 5965cea74..53d106471 100644 --- a/packages/social/src/import/node.ts +++ b/packages/social/src/import/node.ts @@ -12,3 +12,5 @@ export { type ZipArchiveManifestOptions, type ZipCentralDirectoryEntry } from './archive-reader' + +export { openSocialImportSource } from './source-reader' diff --git a/packages/social/src/import/resource-url.ts b/packages/social/src/import/resource-url.ts new file mode 100644 index 000000000..e61f2f28c --- /dev/null +++ b/packages/social/src/import/resource-url.ts @@ -0,0 +1,62 @@ +import type { SocialPlatform } from '../schemas/constants' +import { createSocialNodeId } from './ids' + +/** Preserve generic query/fragment meaning; collapse only known provider resource aliases. */ +export function resourceIdentityForUrl(raw: string): { + id: string + url: string + platform: SocialPlatform + nativeId: string + kind: 'video' | 'post' | 'link' +} { + const parsed = new URL(raw.trim()) + if (!['https:', 'http:'].includes(parsed.protocol) || parsed.username || parsed.password) + throw new Error('Save an HTTP or HTTPS URL without embedded credentials.') + const host = parsed.hostname.toLowerCase().replace(/^www\./, '') + const youtube = + host === 'youtu.be' + ? parsed.pathname.split('/')[1] + : ['youtube.com', 'm.youtube.com'].includes(host) + ? parsed.searchParams.get('v') || + parsed.pathname.match(/^\/(?:shorts|live|embed)\/([\w-]+)/)?.[1] + : null + if (youtube && /^[\w-]{11}$/.test(youtube)) + return { + id: createSocialNodeId('content', ['youtube', 'video', youtube]), + platform: 'youtube', + nativeId: youtube, + kind: 'video', + url: `https://www.youtube.com/watch?v=${youtube}` + } + const instagram = + host === 'instagram.com' + ? parsed.pathname.match(/^\/(?:p|reel|tv)\/([\w-]+)(?:\/|$)/)?.[1] + : null + if (instagram) + return { + id: createSocialNodeId('content', ['instagram', 'post', instagram]), + platform: 'instagram', + nativeId: instagram, + kind: 'post', + url: `https://www.instagram.com/p/${instagram}/` + } + const tweet = ['x.com', 'twitter.com', 'mobile.twitter.com'].includes(host) + ? parsed.pathname.match(/\/(?:i\/web|[^/]+)\/status\/(\d+)(?:\/|$)/)?.[1] + : null + if (tweet) + return { + id: createSocialNodeId('content', ['x', 'tweet', tweet]), + platform: 'x', + nativeId: tweet, + kind: 'post', + url: `https://x.com/i/status/${tweet}` + } + const url = parsed.href + return { + id: createSocialNodeId('content', ['generic', 'url', url]), + platform: 'generic', + nativeId: url, + kind: 'link', + url + } +} diff --git a/packages/social/src/import/source-reader.ts b/packages/social/src/import/source-reader.ts new file mode 100644 index 000000000..2f9943e4b --- /dev/null +++ b/packages/social/src/import/source-reader.ts @@ -0,0 +1,74 @@ +/** A ZIP export or a supported JSON snapshot, exposed through the same entry readers. */ +import type { ArchiveManifest, JsonArchiveEntryReader, TextArchiveEntryReader } from './types' +import { createHash } from 'node:crypto' +import { readFile, stat } from 'node:fs/promises' +import { basename, extname } from 'node:path' +import { + createZipJsonEntryReader, + createZipTextEntryReader, + readZipArchiveManifest +} from './archive-reader' + +export async function openSocialImportSource(path: string): Promise<{ + manifest: ArchiveManifest + readJsonEntry: JsonArchiveEntryReader + readTextEntry: TextArchiveEntryReader +}> { + if (extname(path).toLowerCase() === '.zip') { + return { + manifest: await readZipArchiveManifest(path, { hashEntries: false }), + readJsonEntry: await createZipJsonEntryReader(path), + readTextEntry: await createZipTextEntryReader(path) + } + } + if (extname(path).toLowerCase() !== '.json') + throw new Error('Select a ZIP export, GitHub stars snapshot, or garden JSON file') + const before = await stat(path) + if (!before.isFile() || before.size > 128 * 1024 * 1024) + throw new Error('JSON snapshots must be regular files smaller than 128 MiB') + const bytes = await readFile(path) + const after = await stat(path) + if ( + before.size !== bytes.byteLength || + before.ino !== after.ino || + before.mtimeMs !== after.mtimeMs || + before.ctimeMs !== after.ctimeMs + ) + throw new Error('The snapshot changed while reading it; retry after the export finishes') + const json: unknown = JSON.parse(bytes.toString('utf8').replace(/^\uFEFF/, '')) + const object = + json && typeof json === 'object' && !Array.isArray(json) + ? (json as Record) + : undefined + const isGarden = + object?.version === 1 && + typeof object.profile === 'object' && + object.profile !== null && + Array.isArray(object.entries) + if (!Array.isArray(json) && object?.format !== 'xnet-github-stars/1' && !isGarden) + throw new Error('Unsupported standalone JSON snapshot') + // A standalone file is a one-entry virtual archive. Preserve its actual filename on the manifest. + const entryPath = isGarden ? 'garden.json' : 'github-stars.json' + const hash = createHash('sha256').update(bytes).digest('hex') + const manifest: ArchiveManifest = { + archivePath: path, + filename: basename(path), + byteSize: bytes.byteLength, + archiveHash: hash, + entries: [{ path: entryPath, byteSize: bytes.byteLength, sha256: hash }] + } + const requireEntry = (requested: string) => { + if (requested !== entryPath) throw new Error(`Snapshot entry not found: ${requested}`) + } + return { + manifest, + readJsonEntry: async (requested: string): Promise => { + requireEntry(requested) + return json as T + }, + readTextEntry: async (requested: string): Promise => { + requireEntry(requested) + return bytes.toString('utf8').replace(/^\uFEFF/, '') + } + } +} diff --git a/packages/social/src/importers/garden.ts b/packages/social/src/importers/garden.ts new file mode 100644 index 000000000..a9502ee68 --- /dev/null +++ b/packages/social/src/importers/garden.ts @@ -0,0 +1,197 @@ +/** Authored garden commentary remains separate from the resource it describes. */ +import type { + ArchiveEntryRef, + SocialImportAdapter, + SocialImportContext, + StagedSocialRecord +} from '../import/types' +import { createSocialNodeId, createSourceRecord, createStagedNode } from '../import/core' +import { resourceIdentityForUrl } from '../import/resource-url' + +const record = (value: unknown): value is Record => + value !== null && typeof value === 'object' && !Array.isArray(value) +const text = (value: unknown) => (typeof value === 'string' ? value : '') +const pathMatches = (path: string) => /(?:^|\/)garden\.json$/i.test(path) +export const gardenAdapter: SocialImportAdapter = { + id: 'garden', + version: '0.1.0', + platform: 'generic', + detect: (manifest) => (manifest.entries.some((entry) => pathMatches(entry.path)) ? 0.98 : 0), + probe: ({ manifest }) => ({ + adapterId: 'garden', + adapterVersion: '0.1.0', + platform: 'generic', + confidence: 0.98, + buckets: [ + { + id: 'garden.entries', + label: 'Garden resources and commentary', + description: 'Links, authored notes, tags, categories, and source-post provenance.', + entryPaths: manifest.entries + .filter((entry) => pathMatches(entry.path)) + .map((entry) => entry.path), + privacyClass: 'private', + defaultSelected: false + } + ], + warnings: [ + 'Imported garden entries stay private in this workspace, including entries already published elsewhere.' + ] + }), + async *stage(context, selection = {}) { + if (!selection.buckets?.includes('garden.entries') || !selection.includeSensitive) return + for (const source of context.manifest.entries.filter((entry) => pathMatches(entry.path))) + yield* mapGarden({ context, source, input: await context.readJsonEntry(source.path) }) + } +} + +export function mapGarden(input: { + context: Pick + source: ArchiveEntryRef + input: unknown +}): StagedSocialRecord[] { + const snapshot = input.input + if ( + !record(snapshot) || + snapshot.version !== 1 || + !record(snapshot.profile) || + !Array.isArray(snapshot.entries) + ) + throw new Error('Expected a version-1 garden snapshot with profile and entries.') + const owner = text(snapshot.profile.did) || text(snapshot.profile.handle) + if (!owner) throw new Error('Garden profile is missing its stable author identity.') + const base = { + platform: 'generic' as const, + bucketId: 'garden.entries', + source: input.source, + privacyClass: 'private' as const + } + const result: StagedSocialRecord[] = [] + const categories = new Set() + const seen = new Set() + for (const [index, value] of snapshot.entries.entries()) { + if (!record(value) || !text(value.url)) + throw new Error(`Garden entry ${index + 1} has no source URL.`) + const identity = resourceIdentityForUrl(text(value.url)) + const title = text(value.title) || identity.url + const note = text(value.note) + if (title.length > 1000 || note.length > 20000) + throw new Error( + `Garden entry ${index + 1} exceeds the supported title or note size; no text was truncated.` + ) + const postUrl = text(value.source) + if (postUrl) resourceIdentityForUrl(postUrl) + const sourceKey = postUrl || identity.url + if (seen.has(sourceKey)) + throw new Error(`Garden snapshot repeats an entry identity at row ${index + 1}.`) + seen.add(sourceKey) + const evidence = createSourceRecord({ + ...base, + archiveId: input.context.archiveId, + importRunId: input.context.importRunId, + sourceRecordKind: 'content', + sourceRecordId: sourceKey, + payload: value + }) + const authoredAt = text(value.createdAt) + if (authoredAt && !Number.isFinite(Date.parse(authoredAt))) + throw new Error(`Garden entry ${index + 1} has an invalid authored timestamp.`) + const sourceProps = { sourceRecordId: evidence.deterministicId, ...base } + result.push( + evidence, + createStagedNode({ + ...sourceProps, + platform: identity.platform, + kind: 'content', + deterministicId: identity.id, + properties: { + contentKind: identity.kind, + platformContentId: identity.nativeId, + canonicalUrl: identity.url, + title, + observedAt: input.context.importedAt + } + }) + ) + // This is the author's source post, not a fetched description or a new blank Page. + result.push( + createStagedNode({ + ...sourceProps, + kind: 'content', + deterministicId: createSocialNodeId('content', ['garden-commentary', owner, sourceKey]), + properties: { + contentKind: 'post', + platformContentKind: 'garden-commentary', + platformContentId: sourceKey, + parentContent: identity.id, + title, + textPreview: note.slice(0, 5000), + searchText: note, + ...(postUrl ? { canonicalUrl: postUrl } : {}), + ...(authoredAt ? { publishedAt: authoredAt } : {}), + actorHandle: text(snapshot.profile.handle), + observedAt: input.context.importedAt, + metadataJson: JSON.stringify({ + tags: value.tags, + thumbnail: value.thumbnail, + links: value.links, + mentions: value.mentions, + added: value.added, + source: value.source, + authored: true + }) + } + }) + ) + const names = ['Garden', text(value.category)].filter(Boolean) + if ( + value.tags !== undefined && + (!Array.isArray(value.tags) || value.tags.some((tag) => typeof tag !== 'string')) + ) + throw new Error(`Garden entry ${index + 1} has invalid tags.`) + for (const name of [ + ...names, + ...(Array.isArray(value.tags) ? value.tags.map((tag) => `#${tag}`) : []) + ]) { + const collectionId = createSocialNodeId('collection', ['garden', owner, name]) + if (!categories.has(name)) { + categories.add(name) + result.push( + createStagedNode({ + ...sourceProps, + kind: 'collection', + deterministicId: collectionId, + properties: { + title: name, + collectionKind: 'saved', + platformCollectionId: `garden:${name}`, + observedAt: input.context.importedAt + } + }) + ) + } + result.push( + createStagedNode({ + ...sourceProps, + kind: 'collection-item', + deterministicId: createSocialNodeId('collection-item', [ + 'garden', + owner, + name, + sourceKey + ]), + properties: { + collection: collectionId, + item: identity.id, + sortKey: String(index).padStart(8, '0'), + ...(authoredAt ? { addedAt: authoredAt } : {}), + metadataJson: JSON.stringify({ + commentary: createSocialNodeId('content', ['garden-commentary', owner, sourceKey]) + }) + } + }) + ) + } + } + return result +} diff --git a/packages/social/src/importers/github.ts b/packages/social/src/importers/github.ts new file mode 100644 index 000000000..04206eb96 --- /dev/null +++ b/packages/social/src/importers/github.ts @@ -0,0 +1,272 @@ +/** GitHub stars snapshots. Native repository IDs survive renames and transfers. */ +import type { + ArchiveEntryRef, + SocialImportAdapter, + SocialImportContext, + StagedSocialRecord +} from '../import/types' +import { createSocialNodeId, createSourceRecord, createStagedNode } from '../import/core' + +export const GITHUB_ADAPTER_ID = 'github' +export const GITHUB_ADAPTER_VERSION = '0.1.0' +const snapshotPath = /(?:^|\/)(?:github[-_]stars(?:[-_].*)?|starred-repositories|stars)\.json$/i +const record = (value: unknown): value is Record => + Boolean(value && typeof value === 'object' && !Array.isArray(value)) +const text = (value: unknown) => (typeof value === 'string' ? value : undefined) +const nativeId = (value: unknown): string | undefined => + typeof value === 'number' && Number.isSafeInteger(value) && value > 0 + ? String(value) + : typeof value === 'string' && /^[1-9]\d*$/.test(value) + ? value + : undefined + +export const githubAdapter: SocialImportAdapter = { + id: GITHUB_ADAPTER_ID, + version: GITHUB_ADAPTER_VERSION, + platform: 'github', + detect: (manifest) => + manifest.entries.some((entry) => snapshotPath.test(entry.path)) ? 0.95 : 0, + probe: ({ manifest }) => ({ + adapterId: GITHUB_ADAPTER_ID, + adapterVersion: GITHUB_ADAPTER_VERSION, + platform: 'github', + confidence: manifest.entries.some((entry) => snapshotPath.test(entry.path)) ? 0.95 : 0, + buckets: [ + { + id: 'github.stars', + label: 'Starred repositories', + description: 'Repositories, native star events, owners, and snapshot coverage.', + entryPaths: manifest.entries + .filter((entry) => snapshotPath.test(entry.path)) + .map((entry) => entry.path), + privacyClass: 'private', + defaultSelected: false + } + ], + warnings: [ + 'A stars snapshot records currently visible stars; it is not a history of removed stars.' + ] + }), + async *stage(context, selection = {}) { + if (!selection.buckets?.includes('github.stars') || !selection.includeSensitive) return + for (const source of context.manifest.entries.filter((entry) => + snapshotPath.test(entry.path) + )) { + yield* mapGitHubStars({ context, source, input: await context.readJsonEntry(source.path) }) + } + } +} + +export function mapGitHubStars(input: { + context: Pick + source: ArchiveEntryRef + input: unknown +}): StagedSocialRecord[] { + const snapshot = record(input.input) ? input.input : undefined + if (snapshot && snapshot.format !== 'xnet-github-stars/1') + throw new Error('Unsupported GitHub stars snapshot format') + const rows = snapshot?.stars ?? input.input + if (!Array.isArray(rows)) throw new Error('Expected a GitHub stars array') + const account = + text(snapshot?.account)?.trim() || + input.context.observedBy || + `archive:${input.context.archiveId}` + const self = createSocialNodeId('actor', ['github', 'self', account]) + const collection = createSocialNodeId('collection', ['github', 'stars', self]) + const base = { platform: 'github' as const, bucketId: 'github.stars', source: input.source } + const evidence = (id: string, payload: unknown, kind: 'collection' | 'interaction') => + createSourceRecord({ + ...base, + archiveId: input.context.archiveId, + importRunId: input.context.importRunId, + sourceRecordKind: kind, + sourceRecordId: id, + payload, + privacyClass: 'private' + }) + const catalog = evidence( + 'stars-snapshot', + { account, coverage: snapshot?.coverage, rows: rows.length }, + 'collection' + ) + const warnings = snapshot + ? [] + : ['Snapshot account and pagination coverage are unknown; this array may be incomplete.'] + const nodes: StagedSocialRecord[] = [ + catalog, + createStagedNode({ + ...base, + kind: 'actor', + deterministicId: self, + sourceRecordId: catalog.deterministicId, + privacyClass: 'private', + properties: { + actorKind: 'account', + platformActorId: account, + handle: account, + displayName: account, + isSelf: true, + observedBy: input.context.observedBy, + observedAt: input.context.importedAt + } + }), + createStagedNode({ + ...base, + kind: 'collection', + deterministicId: collection, + sourceRecordId: catalog.deterministicId, + privacyClass: 'private', + warnings, + properties: { + collectionKind: 'saved', + platformCollectionId: `stars:${account}`, + title: 'Imported GitHub stars', + ownerActor: self, + itemCount: rows.length, + observedAt: input.context.importedAt, + metadataJson: JSON.stringify({ + coverage: snapshot?.coverage ?? { paginationCompleted: null }, + startedAt: snapshot?.startedAt, + finishedAt: snapshot?.finishedAt + }) + } + }) + ] + return [ + ...nodes, + ...rows.flatMap((row: unknown, index): StagedSocialRecord[] => { + if (!record(row)) throw new Error(`Invalid GitHub star at row ${index + 1}`) + const repo = record(row.repo) ? row.repo : row + const id = nativeId(repo.id) + const name = text(repo.full_name) + const url = text(repo.html_url) + if (!id || !name || !url) + throw new Error( + `GitHub repository at row ${index + 1} is missing its native ID, full name, or URL` + ) + const starredAt = text(row.starred_at) + if (row.starred_at !== undefined && (!starredAt || !Number.isFinite(Date.parse(starredAt)))) + throw new Error(`Invalid native star timestamp at row ${index + 1}`) + const owner = record(repo.owner) ? repo.owner : undefined + const handle = text(owner?.login) + const ownerId = nativeId(owner?.id) + const actor = + ownerId || handle + ? createSocialNodeId('actor', ['github', 'owner', ownerId ?? handle]) + : undefined + const content = createSocialNodeId('content', ['github', 'repository', id]) + const eventId = createSocialNodeId('interaction', [ + 'github', + self, + 'star', + id, + starredAt ?? 'time-unavailable' + ]) + const sourceRecord = evidence( + `star:${id}:${starredAt ?? 'time-unavailable'}:${index}`, + row, + 'interaction' + ) + const privacyClass = repo.private === false ? ('public' as const) : ('private' as const) + const description = text(repo.description) + const topics = Array.isArray(repo.topics) + ? repo.topics.filter((topic): topic is string => typeof topic === 'string') + : [] + return [ + sourceRecord, + ...(actor + ? [ + createStagedNode({ + ...base, + kind: 'actor', + deterministicId: actor, + sourceRecordId: sourceRecord.deterministicId, + privacyClass, + properties: { + actorKind: 'account', + platformActorId: ownerId ?? handle, + handle, + displayName: handle, + profileUrl: text(owner?.html_url), + observedAt: input.context.importedAt, + metadataJson: JSON.stringify({ avatarUrl: owner?.avatar_url }) + } + }) + ] + : []), + createStagedNode({ + ...base, + kind: 'content', + deterministicId: content, + sourceRecordId: sourceRecord.deterministicId, + privacyClass, + properties: { + contentKind: 'link', + platformContentKind: 'github_repository', + platformContentId: id, + canonicalUrl: url, + platformUrl: url, + authorActor: actor, + actorHandle: handle, + title: name, + textPreview: description, + searchText: [name, description, ...topics, text(repo.language)] + .filter(Boolean) + .join('\n'), + importedAt: input.context.importedAt, + observedAt: input.context.importedAt, + metadataJson: JSON.stringify({ + nodeId: repo.node_id, + descriptionAvailable: description !== undefined, + topics, + language: repo.language, + homepage: repo.homepage, + license: repo.license, + archived: repo.archived, + disabled: repo.disabled, + private: repo.private, + pushedAt: repo.pushed_at, + stars: repo.stargazers_count, + forks: repo.forks_count, + ownerAvatarUrl: owner?.avatar_url + }) + } + }), + createStagedNode({ + ...base, + kind: 'interaction', + deterministicId: eventId, + sourceRecordId: sourceRecord.deterministicId, + privacyClass: 'private', + warnings: starredAt ? [] : ['Native star timestamp is unavailable.'], + properties: { + interactionKind: 'star', + platformInteractionKind: 'star', + actor: self, + target: content, + targetSchema: 'SocialContent', + targetTitle: name, + publishedAt: starredAt, + observedAt: input.context.importedAt, + importedAt: input.context.importedAt, + metadataJson: JSON.stringify({ nativeStarTimestampAvailable: Boolean(starredAt) }) + } + }), + createStagedNode({ + ...base, + kind: 'collection-item', + deterministicId: createSocialNodeId('collection-item', [collection, eventId]), + sourceRecordId: sourceRecord.deterministicId, + privacyClass: 'private', + properties: { + collection, + item: content, + itemSchema: 'SocialContent', + sortKey: String(index).padStart(8, '0'), + addedAt: starredAt + } + }) + ] + }) + ] +} diff --git a/packages/social/src/importers/index.ts b/packages/social/src/importers/index.ts index 7e476e9f6..780851f01 100644 --- a/packages/social/src/importers/index.ts +++ b/packages/social/src/importers/index.ts @@ -102,3 +102,7 @@ export { type SocialImporterAvailability, type SocialImporterRegistryEntry } from './registry' + +export { GITHUB_ADAPTER_ID, GITHUB_ADAPTER_VERSION, githubAdapter, mapGitHubStars } from './github' + +export { gardenAdapter, mapGarden } from './garden' diff --git a/packages/social/src/importers/instagram.ts b/packages/social/src/importers/instagram.ts index 257a4aa2a..84258e275 100644 --- a/packages/social/src/importers/instagram.ts +++ b/packages/social/src/importers/instagram.ts @@ -22,7 +22,7 @@ import { } from '../import/core' export const INSTAGRAM_ADAPTER_ID = 'instagram' -export const INSTAGRAM_ADAPTER_VERSION = '0.1.0' +export const INSTAGRAM_ADAPTER_VERSION = '0.2.0' type InstagramStringListData = { href?: string @@ -35,10 +35,20 @@ type InstagramRelationshipRecord = { string_list_data?: InstagramStringListData[] } +type InstagramLabel = { + label?: string + value?: string + href?: string + title?: string + timestamp_value?: number + dict?: InstagramLabel[] +} + type InstagramLabeledRecord = { + raw?: unknown timestamp?: number fbid?: string - label_values?: Array<{ label?: string; value?: string; href?: string }> + label_values?: InstagramLabel[] media?: unknown[] } @@ -426,22 +436,58 @@ export function mapInstagramSavedPosts(input: { selfActorId: string input: unknown }): StagedSocialRecord[] { + const rows = asLabeledArray(input.input) + if (input.source.path.includes('saved_collections')) { + return rows.flatMap((row) => { + const labels = labelMap(row) + const members = (row.label_values ?? []) + .flatMap((label) => label.dict ?? []) + .filter((member) => member.dict?.some((label) => label.label?.toLowerCase() === 'url')) + .map((member) => ({ label_values: member.dict, raw: member })) + return mapSavedCollection(input, { + key: row.fbid ?? labels.name ?? 'unnamed', + title: labels.name ?? 'Instagram collection', + rows: members, + payload: row, + named: true + }) + }) + } + return mapSavedCollection(input, { + key: input.source.path, + title: savedCollectionTitle(input.source.path), + rows, + payload: { rowCount: rows.length }, + named: false + }) +} + +function mapSavedCollection( + input: Parameters[0], + collection: { + key: string + title: string + rows: InstagramLabeledRecord[] + payload: unknown + named: boolean + } +): StagedSocialRecord[] { const collectionId = createSocialNodeId('collection', [ 'instagram', input.selfActorId, - input.source.path + collection.key ]) const collectionSource = createSourceRecord({ ...sourceBase( input, 'instagram.saves', - `collection:${input.source.path}`, - {}, + `collection:${collection.key}`, + collection.payload, 'collection', - 'public' + 'private' ) }) - + const occurrences = new Map() return [ collectionSource, createStagedNode({ @@ -451,35 +497,43 @@ export function mapInstagramSavedPosts(input: { bucketId: 'instagram.saves', source: input.source, sourceRecordId: collectionSource.deterministicId, - privacyClass: 'public', + privacyClass: 'private', properties: { collectionKind: 'saved', - platformCollectionId: input.source.path, - title: savedCollectionTitle(input.source.path), + platformCollectionId: collection.key, + title: collection.title, ownerActor: input.selfActorId, - itemCount: asLabeledArray(input.input).length, + itemCount: collection.rows.length, observedAt: input.context.importedAt } }), - ...asLabeledArray(input.input).flatMap((row, index) => { + ...collection.rows.flatMap((row, index) => { const mapped = mapInstagramLabeledContentInteraction({ ...input, row, index, bucketId: 'instagram.saves', interactionKind: 'save', - platformInteractionKind: input.source.path.includes('music') ? 'saved_music' : 'saved_posts' + sourceDiscriminator: collectionId, + platformInteractionKind: collection.named + ? 'collection_membership' + : input.source.path.includes('music') + ? 'saved_music' + : 'saved_posts' }) const content = mapped.find((record) => record.kind === 'content') if (!content) return mapped + const occurrence = occurrences.get(content.deterministicId) ?? 0 + occurrences.set(content.deterministicId, occurrence + 1) return [ - ...mapped, + ...mapped.filter((record) => !collection.named || record.kind !== 'interaction'), createStagedNode({ kind: 'collection-item', deterministicId: createSocialNodeId('collection-item', [ 'instagram', collectionId, - content.deterministicId + content.deterministicId, + occurrence ]), platform: 'instagram', bucketId: 'instagram.saves', @@ -487,7 +541,7 @@ export function mapInstagramSavedPosts(input: { sourceRecordId: mapped.find((record) => record.kind === 'source-record')?.deterministicId ?? collectionSource.deterministicId, - privacyClass: 'public', + privacyClass: 'private', properties: { collection: collectionId, item: content.deterministicId, @@ -769,21 +823,36 @@ function mapInstagramLabeledContentInteraction(input: { bucketId: string interactionKind: 'like' | 'save' platformInteractionKind: string + sourceDiscriminator?: string }): StagedSocialRecord[] { const labels = labelMap(input.row) const url = labels.href ?? labels.url - const contentKey = input.row.fbid ?? url ?? `${input.source.path}:${input.index}` - const sourceRecordId = `${input.platformInteractionKind}:${contentKey}:${input.index}` + const shortcode = url?.match( + /^https?:\/\/(?:www\.)?instagram\.com\/(?:p|reel|tv)\/([^/?#]+)/i + )?.[1] + const contentKey = input.platformInteractionKind.includes('comment') + ? (url ?? input.row.fbid ?? `${input.source.path}:${input.index}`) + : (shortcode ?? input.row.fbid ?? url ?? `${input.source.path}:${input.index}`) + const sourceRecordId = `${input.platformInteractionKind}:${input.sourceDiscriminator ?? ''}:${contentKey}:${input.index}` const sourceRecord = createSourceRecord({ - ...sourceBase(input, input.bucketId, sourceRecordId, input.row, 'interaction', 'public') + ...sourceBase( + input, + input.bucketId, + sourceRecordId, + input.row.raw ?? input.row, + input.platformInteractionKind === 'collection_membership' ? 'collection-item' : 'interaction', + 'private' + ) }) const contentId = createSocialNodeId('content', [ 'instagram', - input.platformInteractionKind, + input.platformInteractionKind.includes('comment') ? 'comment' : 'post', contentKey ]) const observedAt = secondsToIso(input.row.timestamp) const title = labels.title ?? labels.name ?? input.row.fbid + const nestedLabels = flattenInstagramLabels(input.row.label_values ?? []) + const creator = nestedLabels.find((label) => label.label?.toLowerCase() === 'username')?.value return [ sourceRecord, @@ -802,8 +871,12 @@ function mapInstagramLabeledContentInteraction(input: { canonicalUrl: url ? normalizeUrl(url) : undefined, platformUrl: url ? normalizeUrl(url) : undefined, title, - textPreview: trimPreview(labels.description ?? title ?? ''), - searchText: Object.values(labels).filter(Boolean).join('\n'), + textPreview: trimPreview(labels.caption ?? labels.description ?? title ?? ''), + searchText: nestedLabels + .map((label) => label.value ?? label.href) + .filter(Boolean) + .join('\n'), + actorHandle: creator, observedAt, importedAt: input.context.importedAt, confidence: input.row.fbid ? 0.9 : 0.7, @@ -823,7 +896,7 @@ function mapInstagramLabeledContentInteraction(input: { bucketId: input.bucketId, source: input.source, sourceRecordId: sourceRecord.deterministicId, - privacyClass: 'public', + privacyClass: 'private', properties: { interactionKind: input.interactionKind, platformInteractionKind: input.platformInteractionKind, @@ -921,7 +994,19 @@ function asRelationshipArray(input: unknown, key?: string): InstagramRelationshi } function asLabeledArray(input: unknown): InstagramLabeledRecord[] { - return asArray(input) + if (Array.isArray(input)) return input as InstagramLabeledRecord[] + if (isRecord(input) && Array.isArray(input.likes_comment_likes)) { + return (input.likes_comment_likes as InstagramRelationshipRecord[]).map((row) => ({ + raw: row, + timestamp: row.string_list_data?.[0]?.timestamp, + label_values: [ + { label: 'URL', href: row.string_list_data?.[0]?.href }, + { label: 'Title', value: row.title }, + { label: 'Description', value: row.string_list_data?.[0]?.value } + ] + })) + } + throw new Error('Unsupported Instagram likes/saves shape; no records were imported.') } function asArray(input: unknown): T[] { @@ -932,6 +1017,10 @@ function isRecord(value: unknown): value is Record { return Boolean(value && typeof value === 'object' && !Array.isArray(value)) } +function flattenInstagramLabels(labels: InstagramLabel[]): InstagramLabel[] { + return labels.flatMap((label) => [label, ...flattenInstagramLabels(label.dict ?? [])]) +} + function labelMap(row: InstagramLabeledRecord): Record { return Object.fromEntries( (row.label_values ?? []).flatMap((label) => { diff --git a/packages/social/src/importers/reddit.ts b/packages/social/src/importers/reddit.ts index 7efd4a1d8..4b8b18d4b 100644 --- a/packages/social/src/importers/reddit.ts +++ b/packages/social/src/importers/reddit.ts @@ -22,7 +22,7 @@ import { } from '../import/core' export const REDDIT_ADAPTER_ID = 'reddit' -export const REDDIT_ADAPTER_VERSION = '0.1.0' +export const REDDIT_ADAPTER_VERSION = '0.1.2' export type RedditCsvRow = Record @@ -135,6 +135,45 @@ export const redditAdapter: SocialImportAdapter = { export async function* stageRedditArchive( context: SocialImportContext, selection: ImportSelection = {} +): AsyncIterable { + const content = new Map() + for await (const record of readRedditArchive(context, selection)) { + if (record.kind !== 'content') { + yield record + continue + } + const previous = content.get(record.deterministicId) + if (!previous) { + content.set(record.deterministicId, record) + continue + } + // Votes and saves often contain only an ID. Keep the fuller observation from + // this same export, so a later reference cannot erase an authored body. + const strength = (value: StagedSocialRecord) => Number(value.properties.confidence ?? 0) + const bodyLength = (value: StagedSocialRecord) => + typeof value.properties.searchText === 'string' ? value.properties.searchText.length : 0 + const preferPrevious = + strength(previous) > strength(record) || + (strength(previous) === strength(record) && bodyLength(previous) >= bodyLength(record)) + const defined = (value: StagedSocialRecord) => + Object.fromEntries( + Object.entries(value.properties).filter( + ([, value]) => value !== undefined && value !== null && value !== '' + ) + ) + const richer = preferPrevious ? previous : record + const poorer = preferPrevious ? record : previous + const merged = { ...record, properties: { ...defined(poorer), ...defined(richer) } } + content.set(record.deterministicId, merged) + } + // Repeated fields within one native write batch share a timestamp. Emit the + // final merged content once so a thin first reference cannot win that tie. + yield* content.values() +} + +async function* readRedditArchive( + context: SocialImportContext, + selection: ImportSelection = {} ): AsyncIterable { const readTextEntry = requireTextEntry(context) const selectedBuckets = resolveSelectedBuckets(createRedditBuckets(context.manifest), selection) diff --git a/packages/social/src/importers/registry.ts b/packages/social/src/importers/registry.ts index c6b9a2aed..026264fa2 100644 --- a/packages/social/src/importers/registry.ts +++ b/packages/social/src/importers/registry.ts @@ -5,6 +5,8 @@ import type { SocialImportAdapter } from '../import/types' import type { SocialPlatform, SocialPrivacyClass } from '../schemas/constants' import { claudeAdapter } from './claude' +import { gardenAdapter } from './garden' +import { githubAdapter } from './github' import { grokAdapter } from './grok' import { instagramAdapter } from './instagram' import { openaiAdapter } from './openai' @@ -60,6 +62,22 @@ const plannedImporter = (input: { }) export const builtInSocialImporterRegistry = [ + availableImporter({ + adapter: gardenAdapter, + label: 'Personal garden', + description: 'Authored notes and source links from a version-1 garden JSON snapshot.', + archiveFormats: ['Garden JSON'], + recordTypes: ['resources', 'commentary', 'categories', 'tags'], + privacyClasses: ['private'] + }), + availableImporter({ + adapter: githubAdapter, + label: 'GitHub stars', + description: 'Saved GitHub API stars snapshots with native repository IDs and star timestamps.', + archiveFormats: ['JSON snapshot', 'ZIP with JSON snapshot'], + recordTypes: ['repositories', 'stars', 'owners', 'collections'], + privacyClasses: ['public', 'private'] + }), availableImporter({ adapter: instagramAdapter, label: 'Instagram', diff --git a/packages/social/src/importers/x.ts b/packages/social/src/importers/x.ts index 5f2027437..b8779a05a 100644 --- a/packages/social/src/importers/x.ts +++ b/packages/social/src/importers/x.ts @@ -21,7 +21,7 @@ import { } from '../import/core' export const X_ADAPTER_ID = 'x' -export const X_ADAPTER_VERSION = '0.1.0' +export const X_ADAPTER_VERSION = '0.2.0' type XBucketPattern = { id: string @@ -264,7 +264,12 @@ export const xAdapter: SocialImportAdapter = { platform: 'x', confidence: hasXArchiveSignals(manifest) ? 0.95 : 0, buckets: createXBuckets(manifest), - warnings: [] + warnings: [ + manifest.entries.some((entry) => /bookmarks?\.js$/.test(entry.path)) + ? 'Bookmark data is present, but this adapter does not yet map that category.' + : 'No bookmark file is present in this archive.', + 'The likes export does not include native like timestamps.' + ] }), stage: stageXArchive } diff --git a/packages/social/src/importers/youtube.ts b/packages/social/src/importers/youtube.ts index 31fa66d46..de0a00307 100644 --- a/packages/social/src/importers/youtube.ts +++ b/packages/social/src/importers/youtube.ts @@ -21,7 +21,7 @@ import { } from '../import/core' export const YOUTUBE_ADAPTER_ID = 'youtube' -export const YOUTUBE_ADAPTER_VERSION = '0.1.0' +export const YOUTUBE_ADAPTER_VERSION = '0.2.0' export type YouTubeCsvRow = Record @@ -407,31 +407,63 @@ export function mapYouTubePlaylists(input: { videoFiles?: readonly YouTubePlaylistVideoFile[] }): StagedSocialRecord[] { const catalogRows = input.catalogRows ?? [] - const rowsByTitle = new Map( - catalogRows.map((row) => [normalizePlaylistTitle(row['Playlist Title (Original)']), row]) - ) + const matchingRows = (title: string) => + catalogRows.filter( + (row) => + normalizePlaylistTitle(row['Playlist Title (Original)']) === normalizePlaylistTitle(title) + ) const emittedCollections = new Set() const collectionRecords = catalogRows.flatMap((row, index) => { if (!input.catalogSource) return [] const collection = createPlaylistCollection(input, input.catalogSource, row, index) emittedCollections.add(collection.collectionId) - return collection.records + const hasFile = (input.videoFiles ?? []).some( + (file) => + normalizePlaylistTitle(playlistTitleFromVideoPath(file.source.path)) === + normalizePlaylistTitle(row['Playlist Title (Original)']) + ) + return collection.records.map((record) => ({ + ...record, + warnings: hasFile + ? record.warnings + : [ + ...record.warnings, + 'No membership file was exported for this playlist; membership coverage is unknown.' + ] + })) }) const itemRecords = (input.videoFiles ?? []).flatMap((file) => { const playlistTitle = playlistTitleFromVideoPath(file.source.path) - const catalogRow = rowsByTitle.get(normalizePlaylistTitle(playlistTitle)) + const candidates = matchingRows(playlistTitle) + // A filename cannot disambiguate two catalog titles that normalize alike. + // Keep the file's own collection and report the ambiguity instead of guessing. + const catalogRow = candidates.length === 1 ? candidates[0] : undefined const collection = createPlaylistCollection(input, file.source, catalogRow, 0, playlistTitle) const collectionIntro = emittedCollections.has(collection.collectionId) ? [] : collection.records emittedCollections.add(collection.collectionId) - + const occurrences = new Map() + const warnings = + candidates.length === 1 + ? [] + : [ + candidates.length + ? 'Ambiguous playlist catalog match; file preserved separately.' + : 'No playlist catalog match; file preserved separately.' + ] return [ - ...collectionIntro, - ...file.rows.flatMap((row, index) => - createPlaylistItemRecords({ + ...collectionIntro.map((record) => ({ + ...record, + warnings: [...record.warnings, ...warnings] + })), + ...file.rows.flatMap((row, index) => { + const identity = JSON.stringify([row['Video ID'], row['Playlist Video Creation Timestamp']]) + const occurrence = occurrences.get(identity) ?? 0 + occurrences.set(identity, occurrence + 1) + return createPlaylistItemRecords({ context: input.context, source: file.source, selfActorId: input.selfActorId, @@ -439,9 +471,10 @@ export function mapYouTubePlaylists(input: { collectionTitle: collection.title, row, index, + occurrence, privacyClass: collection.privacyClass }) - ) + }) ] }) @@ -810,6 +843,7 @@ function createPlaylistItemRecords(input: { collectionTitle: string row: YouTubeCsvRow index: number + occurrence: number privacyClass: SocialPrivacyClass }): StagedSocialRecord[] { const videoId = input.row['Video ID'] || `${input.collectionId}:${input.index}` @@ -842,7 +876,9 @@ function createPlaylistItemRecords(input: { deterministicId: createSocialNodeId('collection-item', [ 'youtube', input.collectionId, - contentId + contentId, + addedAt, + input.occurrence ]), platform: 'youtube', bucketId: 'youtube.playlists', @@ -1024,7 +1060,10 @@ function playlistTitleFromVideoPath(path: string): string { } function normalizePlaylistTitle(value?: string): string { - return (value ?? '').trim().toLowerCase() + return (value ?? '') + .trim() + .toLowerCase() + .replace(/[\\/:*?"<>|]/g, '_') } function youtubeVisibilityToPrivacy(value?: string): SocialPrivacyClass { diff --git a/packages/social/src/schemas/constants.ts b/packages/social/src/schemas/constants.ts index cdc758977..1d62c9d16 100644 --- a/packages/social/src/schemas/constants.ts +++ b/packages/social/src/schemas/constants.ts @@ -5,6 +5,7 @@ export const SOCIAL_NAMESPACE = 'xnet://xnet.fyi/social/' as const export const socialPlatforms = [ + { id: 'github', name: 'GitHub' }, { id: 'instagram', name: 'Instagram' }, { id: 'grok', name: 'Grok' }, { id: 'x', name: 'X' }, @@ -120,6 +121,7 @@ export const contentKinds = [ export type SocialContentKind = (typeof contentKinds)[number]['id'] export const interactionKinds = [ + { id: 'star', name: 'Star' }, { id: 'follow', name: 'Follow' }, { id: 'like', name: 'Like' }, { id: 'save', name: 'Save' }, diff --git a/packages/sqlite/src/adapters/electron.test.ts b/packages/sqlite/src/adapters/electron.test.ts index a49eb2a3c..23732de4e 100644 --- a/packages/sqlite/src/adapters/electron.test.ts +++ b/packages/sqlite/src/adapters/electron.test.ts @@ -78,6 +78,16 @@ describeNativeSQLite('ElectronSQLiteAdapter', () => { }) describe('Lifecycle', () => { + it('does not report an inspection failure as an unversioned database', async () => { + await adapter.close() + await expect(adapter.getSchemaVersion()).rejects.toThrow() + }) + + it('rejects malformed version tracking instead of initializing over it', async () => { + await adapter.exec('DROP TABLE _schema_version; CREATE TABLE _schema_version (wrong TEXT)') + await expect(adapter.getSchemaVersion()).rejects.toThrow() + }) + it('creates database file', () => { expect(existsSync(dbPath)).toBe(true) }) @@ -87,6 +97,11 @@ describeNativeSQLite('ElectronSQLiteAdapter', () => { expect(result?.journal_mode).toBe('wal') }) + it('syncs the WAL before acknowledging a committed write', async () => { + const result = await adapter.queryOne<{ synchronous: number }>('PRAGMA synchronous') + expect(result?.synchronous).toBe(2) + }) + it('enables foreign keys', async () => { const result = await adapter.queryOne<{ foreign_keys: number }>('PRAGMA foreign_keys') expect(result?.foreign_keys).toBe(1) diff --git a/packages/sqlite/src/adapters/electron.ts b/packages/sqlite/src/adapters/electron.ts index c8cd27bd4..2c5f6d046 100644 --- a/packages/sqlite/src/adapters/electron.ts +++ b/packages/sqlite/src/adapters/electron.ts @@ -110,8 +110,8 @@ export class ElectronSQLiteAdapter implements SQLiteAdapter { this.db.pragma('busy_timeout = 5000') } - // Performance optimizations - this.db.pragma('synchronous = NORMAL') + // A resolved desktop write must include the WAL sync, not just a memory acknowledgement. + this.db.pragma('synchronous = FULL') this.db.pragma('cache_size = -64000') // 64MB cache this.db.pragma('temp_store = MEMORY') // Query-planner statistics hygiene (exploration 0264): bound ANALYZE @@ -443,15 +443,18 @@ export class ElectronSQLiteAdapter implements SQLiteAdapter { } async getSchemaVersion(): Promise { - try { - const row = await this.queryOne<{ version: number }>( - 'SELECT version FROM _schema_version ORDER BY version DESC LIMIT 1' - ) - return row?.version ?? 0 - } catch { - // Table doesn't exist yet - return 0 + const table = await this.queryOne<{ name: string }>( + "SELECT name FROM sqlite_master WHERE type = 'table' AND name = '_schema_version'" + ) + if (!table) return 0 + const row = await this.queryOne<{ version: number }>( + 'SELECT version FROM _schema_version ORDER BY version DESC LIMIT 1' + ) + if (!row) return 0 + if (!Number.isSafeInteger(row.version) || row.version < 1) { + throw new Error('Invalid SQLite schema version') } + return row.version } async setSchemaVersion(version: number): Promise { diff --git a/packages/workbench/src/LibraryCaptureHost.tsx b/packages/workbench/src/LibraryCaptureHost.tsx new file mode 100644 index 000000000..8350a9272 --- /dev/null +++ b/packages/workbench/src/LibraryCaptureHost.tsx @@ -0,0 +1,295 @@ +import { getCommandRegistry } from '@xnetjs/plugins' +import { useEffect, useRef, useState } from 'react' +import { workbenchHost } from './host' +import { useNavigateTo } from './platform' + +const blank = () => ({ requestId: crypto.randomUUID(), url: '', title: '', note: '', excerpt: '' }) +const DRAFT_KEY = 'xnet.library.capture-draft.v1' +const inputClass = 'w-full rounded-md border border-border bg-background px-3 py-2 text-sm' + +/** Shared form; the host supplies a durable local save implementation. */ +export function LibraryCaptureHost() { + const navigate = useNavigateTo() + const [open, setOpen] = useState(false) + const [draft, setDraft] = useState(blank) + const [busy, setBusy] = useState(false) + const [locked, setLocked] = useState(false) + const [error, setError] = useState(null) + const [existing, setExisting] = useState<{ + id: string + title: string + notes: { pageId: string; title: string }[] + } | null>(null) + const previousFocus = useRef(null) + const linkInput = useRef(null) + const form = useRef(null) + const host = workbenchHost().library + const show = (url?: string) => { + previousFocus.current = + document.activeElement instanceof HTMLElement ? document.activeElement : null + setError(null) + try { + const saved = localStorage.getItem(DRAFT_KEY) + if (saved) { + const parsed: unknown = JSON.parse(saved) + if ( + !parsed || + typeof parsed !== 'object' || + !['requestId', 'url', 'title', 'note', 'excerpt'].every( + (key) => typeof (parsed as Record)[key] === 'string' + ) + ) + throw new Error( + 'The saved capture draft could not be read. It has been kept in browser storage.' + ) + const value = parsed as ReturnType & { attempted?: boolean } + setDraft({ + requestId: value.requestId, + url: value.url, + title: value.title, + note: value.note, + excerpt: value.excerpt + }) + setLocked(value.attempted === true) + } else { + setDraft({ ...blank(), url: url ?? '' }) + setLocked(false) + } + } catch (error) { + setError(String(error)) + } + setOpen(true) + } + useEffect(() => { + if (!host) return + const command = getCommandRegistry().register({ + id: 'library.capture', + title: 'Save a link to Library', + run: () => show() + }) + const receive = (event: Event) => show((event as CustomEvent<{ url?: string }>).detail?.url) + window.addEventListener('xnet:open-library-capture', receive) + return () => { + command.dispose() + window.removeEventListener('xnet:open-library-capture', receive) + } + }, [host]) + useEffect(() => { + if (open) linkInput.current?.focus() + }, [open]) + useEffect(() => { + if (!open || !host || !/^https?:\/\//i.test(draft.url.trim())) { + setExisting(null) + return + } + let active = true + const timer = setTimeout(() => { + void host.lookup(draft.url).then( + (value) => { + if (active) setExisting(value) + }, + (error) => { + if (active) setError(String(error)) + } + ) + }, 250) + return () => { + active = false + clearTimeout(timer) + } + }, [open, draft.url, host]) + const update = (field: 'url' | 'title' | 'note' | 'excerpt', value: string) => { + if (locked) return + const next = { ...draft, [field]: value } + setDraft(next) + try { + localStorage.setItem(DRAFT_KEY, JSON.stringify(next)) + } catch (error) { + setError(`This draft could not be kept across a restart: ${String(error)}`) + } + } + const close = () => { + if (busy) return + setOpen(false) + previousFocus.current?.focus() + host?.closed?.() + } + const submit = async () => { + if (!host || busy) return + setBusy(true) + setError(null) + try { + const url = new URL(draft.url.trim()) + if (!['http:', 'https:'].includes(url.protocol) || url.username || url.password) + throw new Error('Use an HTTP or HTTPS URL without embedded credentials.') + // Freeze the retry payload until the native writer acknowledges it. + localStorage.setItem(DRAFT_KEY, JSON.stringify({ ...draft, attempted: true })) + setLocked(true) + await host.capture(draft) + localStorage.removeItem(DRAFT_KEY) + setDraft(blank()) + setLocked(false) + setOpen(false) + previousFocus.current?.focus() + host.closed?.() + } catch (error) { + setError(error instanceof Error ? error.message : String(error)) + } finally { + setBusy(false) + } + } + if (!open || !host) return null + return ( +

{ + if (event.key === 'Escape') { + event.stopPropagation() + close() + } + }} + > +
{ + if (event.key !== 'Tab') return + const inputs = Array.from( + form.current?.querySelectorAll( + 'input:not(:disabled), textarea:not(:disabled), button:not(:disabled)' + ) ?? [] + ) + const first = inputs[0], + last = inputs.at(-1) + if (event.shiftKey && document.activeElement === first) { + event.preventDefault() + last?.focus() + } else if (!event.shiftKey && document.activeElement === last) { + event.preventDefault() + first?.focus() + } + }} + onSubmit={(event) => { + event.preventDefault() + void submit() + }} + > +

Save a link

+

+ Keep the source and your own note. Saving works offline. +

+ + +