From 4b1da5b1ab71c70762c537071607534f57275f4b Mon Sep 17 00:00:00 2001 From: austin Date: Mon, 3 Aug 2026 11:39:03 -0500 Subject: [PATCH 1/6] security: harden V2 extraction and local persistence --- .../src/runtime/background-controller.test.ts | 57 +++++++ .../src/runtime/background-controller.ts | 77 +++++++++- apps/extension/src/runtime/redaction.test.ts | 11 ++ apps/extension/src/runtime/redaction.ts | 34 ++++- .../src/storage/analysis-artifacts.ts | 64 +++++--- .../src/storage/analysis-history.test.ts | 84 ++++++++++ .../extension/src/storage/analysis-history.ts | 99 ++++++++++++ .../src/storage/credential-vault.test.ts | 14 ++ .../extension/src/storage/credential-vault.ts | 20 ++- .../src/storage/evidence-cache.test.ts | 46 ++++++ apps/extension/src/storage/evidence-cache.ts | 71 +++++++-- apps/extension/src/storage/job-journal.ts | 34 +++-- apps/extension/src/storage/job-store.ts | 72 ++++++++- apps/extension/src/storage/stores.test.ts | 70 ++++++++- packages/extraction/src/index.test.ts | 39 ++++- packages/extraction/src/index.ts | 144 +++++++++++------- 16 files changed, 820 insertions(+), 116 deletions(-) create mode 100644 apps/extension/src/storage/analysis-history.test.ts create mode 100644 apps/extension/src/storage/analysis-history.ts create mode 100644 apps/extension/src/storage/evidence-cache.test.ts diff --git a/apps/extension/src/runtime/background-controller.test.ts b/apps/extension/src/runtime/background-controller.test.ts index 24300ce..3d777bb 100644 --- a/apps/extension/src/runtime/background-controller.test.ts +++ b/apps/extension/src/runtime/background-controller.test.ts @@ -375,6 +375,63 @@ describe("BackgroundController runtime protocol", () => { ); }); + it("rejects stale and terminal telemetry by run token", async () => { + const context = installChrome(); + const controller = new BackgroundController(); + const jobs = await putJob(context, job("analyzing")); + const entry = { + timestamp: "2026-07-29T12:00:00.000Z", + level: "info" as const, + scope: "test", + event: "test.event", + message: "safe diagnostic", + payload: null, + }; + + await expect( + dispatch( + controller, + { + type: "internal.analysis.log", + requestId: "stale-log", + jobId: "job-1", + runToken: "old-run", + entry, + }, + sender("offscreen.html"), + ), + ).resolves.toMatchObject({ ok: true, data: { ignored: true } }); + await expect(jobs.getLogs("job-1")).resolves.toEqual([]); + + await expect( + dispatch( + controller, + { + type: "internal.analysis.log", + requestId: "current-log", + jobId: "job-1", + runToken: "run-1", + entry, + }, + sender("offscreen.html"), + ), + ).resolves.toMatchObject({ ok: true, data: { accepted: true } }); + await jobs.update("job-1", (current) => ({ ...current, status: "complete" })); + await expect( + dispatch( + controller, + { + type: "internal.analysis.log", + requestId: "terminal-log", + jobId: "job-1", + runToken: "run-1", + entry, + }, + sender("offscreen.html"), + ), + ).resolves.toMatchObject({ ok: true, data: { ignored: true } }); + }); + it("does not dispatch an already-terminal job during service-worker resume", async () => { const context = installChrome(); const controller = new BackgroundController(); diff --git a/apps/extension/src/runtime/background-controller.ts b/apps/extension/src/runtime/background-controller.ts index 5ab722e..3e6a1a0 100644 --- a/apps/extension/src/runtime/background-controller.ts +++ b/apps/extension/src/runtime/background-controller.ts @@ -7,6 +7,8 @@ import { CredentialVault } from "../storage/credential-vault"; import { IndexedDbCryptoKeyStore } from "../storage/indexed-db-key-store"; import { JobStore } from "../storage/job-store"; import { EvidenceCache } from "../storage/evidence-cache"; +import { EncryptedAnalysisHistoryStore } from "../storage/analysis-history"; +import { IndexedDbAnalysisArtifactStore } from "../storage/analysis-artifacts"; import { PreferencesStore } from "../storage/preferences-store"; import { AnalysisJobSchema, @@ -23,6 +25,7 @@ import { type ExtensionResponse, type InternalRequest, type RuntimeState, + type SearchProviderKind, } from "./messages"; import { describeError, redactText, redactUrl, serializeRedacted } from "./redaction"; import { @@ -221,8 +224,12 @@ export class BackgroundController { chrome.runtime.id, ); private readonly preferences = new PreferencesStore(this.local); - private readonly jobs = new JobStore(this.local); private readonly researchCache = new EvidenceCache(); + private readonly artifactStore = new IndexedDbAnalysisArtifactStore(); + private readonly history = new EncryptedAnalysisHistoryStore(this.vault); + private readonly jobs = new JobStore(this.local, undefined, this.history, () => + this.clearTransientResearch(), + ); private readonly auth = new ChatGptSessionManager(this.session, this.local, this.vault); private initialization: Promise | null = null; private readonly ports = new Set(); @@ -276,6 +283,34 @@ export class BackgroundController { }; } + private async cacheScopeFor(provider: SearchProviderKind): Promise { + if (provider === "chatgpt") { + const accountId = (await this.auth.getState()).account?.accountId; + return `chatgpt:${accountId ?? "anonymous"}`; + } + if (provider === "free") return "free:public"; + try { + const credential = await this.vault.readExa(); + if (!credential) return "exa:unconfigured"; + const digest = await crypto.subtle.digest( + "SHA-256", + new TextEncoder().encode(credential.apiKey), + ); + const bytes = new Uint8Array(digest); + return `exa:${Array.from(bytes, (value) => value.toString(16).padStart(2, "0")).join("")}`; + } catch { + return "exa:unconfigured"; + } + } + + private async clearTransientResearch(): Promise { + await Promise.all([this.researchCache.clearAll(), this.artifactStore.clearAll()]); + } + + private async clearAccountData(): Promise { + await Promise.all([this.clearTransientResearch(), this.jobs.clearHistory()]); + } + /** Re-dispatch a persisted non-terminal run after a worker/offscreen restart. */ async resumeActiveJob(): Promise { if (this.resumeInFlight) return this.resumeInFlight; @@ -295,6 +330,21 @@ export class BackgroundController { return; } try { + const cacheScope = resume.cacheScope ?? (await this.cacheScopeFor(resume.searchProvider)); + if (resume.cacheScope && cacheScope !== resume.cacheScope) { + const failed = await this.jobs.update(job.id, (current) => ({ + ...current, + status: "failed", + runToken: null, + revision: current.revision + 1, + updatedAt: now(), + error: + "The previous analysis belonged to a different provider session. Start a new analysis.", + })); + await this.clearTransientResearch(); + if (failed) await this.publish({ type: "analysis.jobChanged", job: failed }); + return; + } await ensureCompatibleOffscreenDocument(); await chrome.runtime.sendMessage( OffscreenCommandSchema.parse({ @@ -305,6 +355,7 @@ export class BackgroundController { initialSequence: job.lastEventSequence, request: resume.request, searchProvider: resume.searchProvider, + cacheScope, }), ); await this.publish({ type: "analysis.jobChanged", job }); @@ -383,6 +434,7 @@ export class BackgroundController { throw new Error("Add an Exa API key in Perspectica settings before analyzing."); } const { tabId, tabUrl, article } = await this.extractActiveArticle(); + const cacheScope = await this.cacheScopeFor(preferences.searchProvider); // Disconnect can race extraction. Re-check ownership immediately before // creating a resumable job so a stale token cannot start a new run. await this.auth.getFreshTokens(); @@ -425,6 +477,7 @@ export class BackgroundController { }, }, searchProvider: preferences.searchProvider, + cacheScope, }; const command = OffscreenCommandSchema.parse({ type: "offscreen.analysis.start", @@ -434,6 +487,7 @@ export class BackgroundController { initialSequence: 0, request: resume.request, searchProvider: preferences.searchProvider, + cacheScope, }); await this.saveAndPushJob(job, resume); try { @@ -524,6 +578,10 @@ export class BackgroundController { if (!resume || resume.runToken !== current.runToken) { throw new Error("The bounded retry context is unavailable. Start a new analysis."); } + const cacheScope = resume.cacheScope ?? (await this.cacheScopeFor(resume.searchProvider)); + if (resume.cacheScope && cacheScope !== resume.cacheScope) { + throw new Error("The bounded retry context belongs to a different provider session."); + } await this.auth.getFreshTokens(); const retrying = await this.jobs.update(jobId, (job) => { if (job.status !== "partial" || job.runToken !== current.runToken) return undefined; @@ -548,6 +606,7 @@ export class BackgroundController { initialSequence: retrying.lastEventSequence, request: resume.request, searchProvider: resume.searchProvider, + cacheScope, sections, }), ); @@ -586,12 +645,20 @@ export class BackgroundController { case "auth.getPending": return ok(request.requestId, await this.auth.pendingAuthorization()); case "auth.poll": { + const previousAccountId = (await this.auth.getState()).account?.accountId ?? null; const result = await this.auth.poll(); - if (result.status === "authenticated") await this.pushAuth(); + if (result.status === "authenticated") { + const nextAccountId = result.state.account?.accountId ?? null; + if (previousAccountId && nextAccountId && previousAccountId !== nextAccountId) { + await this.clearTransientResearch(); + } + await this.pushAuth(); + } return ok(request.requestId, result); } case "auth.disconnect": { await this.cancelActiveAnalysisForDisconnect(); + await this.clearAccountData(); const state = await this.auth.disconnect(); await this.publish({ type: "auth.changed", auth: state }); return ok(request.requestId, state); @@ -614,6 +681,7 @@ export class BackgroundController { return ok(request.requestId, { available: true }); case "providers.clearExaKey": await this.vault.remove("exa"); + await this.clearTransientResearch(); return ok(request.requestId, { removed: true }); case "providers.test": { if (request.provider === "exa") { @@ -685,7 +753,7 @@ export class BackgroundController { return ok(request.requestId, { removed: true, jobId: job.id }); } case "research.cache.clear": - await this.researchCache.clear(); + await this.clearTransientResearch(); return ok(request.requestId, { removed: true }); case "analysis.cancel": return ok(request.requestId, await this.cancelAnalysis(request.jobId)); @@ -735,7 +803,8 @@ export class BackgroundController { } case "internal.analysis.log": { const job = await this.jobs.get(request.jobId); - if (!job) return ok(request.requestId, { ignored: true }); + if (!job || terminal(job.status) || !job.runToken || job.runToken !== request.runToken) + return ok(request.requestId, { ignored: true }); await this.jobs.appendLog(request.jobId, request.entry); return ok(request.requestId, { accepted: true }); } diff --git a/apps/extension/src/runtime/redaction.test.ts b/apps/extension/src/runtime/redaction.test.ts index 1d2dd25..72b479c 100644 --- a/apps/extension/src/runtime/redaction.test.ts +++ b/apps/extension/src/runtime/redaction.test.ts @@ -47,6 +47,17 @@ describe("runtime redaction", () => { expect(result).toContain("…"); }); + it("strips all URL query and hash state from diagnostic serialization", () => { + const result = serializeRedacted({ + url: "https://example.com/story?edition=us#paragraph-2", + message: "https://example.com/story?edition=us#paragraph-2", + }); + + expect(result).toContain("https://example.com/story"); + expect(result).not.toContain("edition=us"); + expect(result).not.toContain("paragraph-2"); + }); + it("describes errors without exposing credential-shaped values", () => { expect(describeError(new Error("refresh_token=hidden"))).toBe("refresh_token=[redacted]"); expect(describeError(undefined, "fallback")).toBe("fallback"); diff --git a/apps/extension/src/runtime/redaction.ts b/apps/extension/src/runtime/redaction.ts index 3de99d6..e7702d9 100644 --- a/apps/extension/src/runtime/redaction.ts +++ b/apps/extension/src/runtime/redaction.ts @@ -21,6 +21,14 @@ function redactUrlToken(value: string): string { return `${normalized ?? "https://[redacted-url]"}${trailing}`; } +function redactDiagnosticText(value: string): string { + return redactText(value).replace(URL_PATTERN, (url) => { + const trailing = url.match(/[),.;!?]+$/)?.[0] ?? ""; + const candidate = trailing ? url.slice(0, -trailing.length) : url; + return `${redactUrl(candidate)}${trailing}`; + }); +} + /** Redacts secrets, credentials, and sensitive URL material from free text. */ export function redactText(value: string): string { return value @@ -36,7 +44,17 @@ export function redactText(value: string): string { */ export function redactUrl(value: string): string { const normalized = normalizeCanonicalUrl(value); - return normalized ?? "[redacted URL]"; + if (!normalized) return "[redacted URL]"; + try { + const url = new URL(normalized); + // Diagnostic URLs never need query state. Even a currently-benign query + // can become a credential or signed request after a provider redirect. + url.search = ""; + url.hash = ""; + return url.toString(); + } catch { + return "[redacted URL]"; + } } function isSensitiveField(key: string): boolean { @@ -62,13 +80,13 @@ export function serializeRedacted( if (item instanceof Error) { return { name: item.name, - message: redactText(item.message), - stack: item.stack ? redactText(item.stack) : undefined, + message: redactDiagnosticText(item.message), + stack: item.stack ? redactDiagnosticText(item.stack) : undefined, cause: item.cause, }; } if (typeof item === "string") { - const redacted = redactText(item); + const redacted = redactDiagnosticText(item); return redacted.length > maxStringLength ? `${redacted.slice(0, maxStringLength - 1)}…` : redacted; @@ -83,11 +101,13 @@ export function serializeRedacted( ); } catch (error) { serialized = JSON.stringify({ - serializationError: redactText(error instanceof Error ? error.message : String(error)), - value: redactText(String(value)), + serializationError: redactDiagnosticText( + error instanceof Error ? error.message : String(error), + ), + value: redactDiagnosticText(String(value)), }); } - return redactText(serialized).slice(0, maxOutputLength); + return redactDiagnosticText(serialized).slice(0, maxOutputLength); } export function describeError(error: unknown, fallback = "Unknown error"): string { diff --git a/apps/extension/src/storage/analysis-artifacts.ts b/apps/extension/src/storage/analysis-artifacts.ts index edfcb0d..ea776e3 100644 --- a/apps/extension/src/storage/analysis-artifacts.ts +++ b/apps/extension/src/storage/analysis-artifacts.ts @@ -1,7 +1,11 @@ +import { ArticleDocumentSchema } from "@perspectica/contracts"; +import { ArticleIndexSchema, type ArticleIndex } from "@perspectica/contracts/article"; +import { + SourceLedgerSnapshotSchema, + type SourceLedgerSnapshot, +} from "@perspectica/contracts/evidence"; +import { AnalysisPlanSchema, type AnalysisPlan } from "@perspectica/contracts/report"; import type { ArticleDocument } from "@perspectica/contracts"; -import type { ArticleIndex } from "@perspectica/contracts/article"; -import type { SourceLedgerSnapshot } from "@perspectica/contracts/evidence"; -import type { AnalysisPlan } from "@perspectica/contracts/report"; import type { AnalysisBudget, AnalysisArtifacts } from "@perspectica/intelligence"; import type { PipelineTelemetry } from "@perspectica/intelligence"; @@ -13,6 +17,7 @@ export interface AnalysisArtifactStore { set(jobId: string, runToken: string, artifacts: AnalysisArtifacts): Promise; get(jobId: string, runToken: string): Promise; clear(jobId: string, runToken?: string): Promise; + clearAll(): Promise; } interface ArtifactRecord { @@ -30,13 +35,14 @@ interface MemoryRecord { const DATABASE_NAME = "perspectica-analysis-artifacts-v1"; const DATABASE_VERSION = 1; const STORE_NAME = "artifacts"; +const MAX_ARTIFACT_BYTES = 25 * 1024 * 1024; function key(jobId: string, runToken: string): string { return `${jobId}:${runToken}`; } function persistable(artifacts: AnalysisArtifacts): PersistedAnalysisArtifacts { - return { + const value = { analysisId: artifacts.analysisId, article: structuredClone(artifacts.article), index: structuredClone(artifacts.index), @@ -45,17 +51,21 @@ function persistable(artifacts: AnalysisArtifacts): PersistedAnalysisArtifacts { telemetry: structuredClone(artifacts.telemetry), ledger: structuredClone(artifacts.ledger.snapshot()), }; + if (JSON.stringify(value).length > MAX_ARTIFACT_BYTES) { + throw new Error("The analysis artifact is too large to retain safely."); + } + return value; } function restoreable(value: PersistedAnalysisArtifacts): PersistedAnalysisArtifacts { return { analysisId: value.analysisId, - article: value.article as ArticleDocument, - index: value.index as ArticleIndex, - plan: value.plan as AnalysisPlan, + article: ArticleDocumentSchema.parse(value.article), + index: ArticleIndexSchema.parse(value.index), + plan: AnalysisPlanSchema.parse(value.plan), budget: value.budget as AnalysisBudget, telemetry: value.telemetry as PipelineTelemetry, - ledger: value.ledger as SourceLedgerSnapshot, + ledger: SourceLedgerSnapshotSchema.parse(value.ledger), }; } @@ -69,15 +79,19 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { this.dbPromise = Promise.resolve(null); return this.dbPromise; } - this.dbPromise = new Promise((resolve) => { + this.dbPromise = new Promise((resolve, reject) => { const request = indexedDB.open(DATABASE_NAME, DATABASE_VERSION); - request.onerror = () => resolve(null); + request.onerror = () => + reject(request.error ?? new Error("Could not open analysis artifacts.")); request.onupgradeneeded = () => { if (!request.result.objectStoreNames.contains(STORE_NAME)) { request.result.createObjectStore(STORE_NAME, { keyPath: ["jobId", "runToken"] }); } }; - request.onsuccess = () => resolve(request.result); + request.onsuccess = () => { + request.result.onversionchange = () => request.result.close(); + resolve(request.result); + }; }); return this.dbPromise; } @@ -91,7 +105,7 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { this.memory.set(key(jobId, runToken), { expiresAt, artifacts: value }); const db = await this.open(); if (!db) return; - await new Promise((resolve) => { + await new Promise((resolve, reject) => { const transaction = db.transaction(STORE_NAME, "readwrite"); transaction.objectStore(STORE_NAME).put({ jobId, @@ -108,7 +122,8 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { cursor.continue(); }; transaction.oncomplete = () => resolve(); - transaction.onerror = () => resolve(); + transaction.onerror = () => + reject(transaction.error ?? new Error("Could not persist analysis artifacts.")); }); } @@ -121,12 +136,13 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { } const db = await this.open(); if (!db) return null; - const record = await new Promise((resolve) => { + const record = await new Promise((resolve, reject) => { const request = db .transaction(STORE_NAME, "readonly") .objectStore(STORE_NAME) .get([jobId, runToken]); - request.onerror = () => resolve(null); + request.onerror = () => + reject(request.error ?? new Error("Could not read analysis artifacts.")); request.onsuccess = () => resolve((request.result as ArtifactRecord | undefined) ?? null); }); if (!record?.artifacts || record.expiresAt <= Date.now()) { @@ -147,7 +163,7 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { if (cacheKey.startsWith(`${jobId}:`)) this.memory.delete(cacheKey); const db = await this.open(); if (!db) return; - await new Promise((resolve) => { + await new Promise((resolve, reject) => { const transaction = db.transaction(STORE_NAME, "readwrite"); if (runToken) { transaction.objectStore(STORE_NAME).delete([jobId, runToken]); @@ -162,7 +178,21 @@ export class IndexedDbAnalysisArtifactStore implements AnalysisArtifactStore { }; } transaction.oncomplete = () => resolve(); - transaction.onerror = () => resolve(); + transaction.onerror = () => + reject(transaction.error ?? new Error("Could not clear analysis artifacts.")); + }); + } + + async clearAll(): Promise { + this.memory.clear(); + const db = await this.open(); + if (!db) return; + await new Promise((resolve, reject) => { + const transaction = db.transaction(STORE_NAME, "readwrite"); + transaction.objectStore(STORE_NAME).clear(); + transaction.oncomplete = () => resolve(); + transaction.onerror = () => + reject(transaction.error ?? new Error("Could not clear analysis artifacts.")); }); } } diff --git a/apps/extension/src/storage/analysis-history.test.ts b/apps/extension/src/storage/analysis-history.test.ts new file mode 100644 index 0000000..32e0fc1 --- /dev/null +++ b/apps/extension/src/storage/analysis-history.test.ts @@ -0,0 +1,84 @@ +import { describe, expect, it } from "vitest"; +import type { AnalysisEnvelope } from "@perspectica/contracts/events"; +import type { AnalysisJob } from "../runtime/messages"; +import { EncryptedAnalysisHistoryStore, type AnalysisHistoryVault } from "./analysis-history"; + +class MemoryHistoryVault implements AnalysisHistoryVault { + value: unknown[] | undefined; + + async readAnalysisHistory(): Promise { + return structuredClone(this.value) as T[] | undefined; + } + + async writeAnalysisHistory(value: T[]): Promise { + this.value = structuredClone(value) as unknown[]; + } +} + +function job(id: string, updatedAt = new Date().toISOString()): AnalysisJob { + return { + id, + tabId: 1, + tabUrl: "https://example.com/article", + articleFingerprint: `fingerprint-${id}`, + analysisConfigFingerprint: "analysis-config-v2", + status: "complete", + createdAt: updatedAt, + updatedAt, + error: null, + events: [], + runToken: `run-${id}`, + revision: 1, + lastEventSequence: 1, + }; +} + +const event = { + protocol: 2, + jobId: "job-1", + runToken: "run-job-1", + sequence: 1, + revision: 1, + event: { + type: "metadata.ready", + analysisId: "analysis-1", + emittedAt: "2026-08-03T00:00:00.000Z", + data: { + title: "Example", + author: null, + publication: null, + publishedAt: null, + contentType: "news", + }, + }, +} as AnalysisEnvelope; + +describe("EncryptedAnalysisHistoryStore", () => { + it("retains at most ten recent terminal runs", async () => { + const vault = new MemoryHistoryVault(); + const store = new EncryptedAnalysisHistoryStore(vault); + for (let index = 0; index < 12; index += 1) { + await store.retain(job(`job-${index}`), [event]); + } + + await expect(store.get()).resolves.toHaveLength(10); + }); + + it("drops expired history and can clear the encrypted archive", async () => { + const vault = new MemoryHistoryVault(); + const store = new EncryptedAnalysisHistoryStore(vault); + await store.retain(job("current"), [event]); + vault.value = [ + { + job: job("expired", new Date(Date.now() - 8 * 24 * 60 * 60 * 1_000).toISOString()), + events: [event], + savedAt: new Date(Date.now() - 8 * 24 * 60 * 60 * 1_000).toISOString(), + }, + ...(vault.value ?? []), + ]; + + await expect(store.get()).resolves.toHaveLength(1); + await store.clear(); + await expect(store.get()).resolves.toEqual([]); + }); +}); diff --git a/apps/extension/src/storage/analysis-history.ts b/apps/extension/src/storage/analysis-history.ts new file mode 100644 index 0000000..7bb22ce --- /dev/null +++ b/apps/extension/src/storage/analysis-history.ts @@ -0,0 +1,99 @@ +import type { AnalysisEnvelope } from "@perspectica/contracts/events"; +import type { AnalysisJob } from "../runtime/messages"; + +const MAX_HISTORY_JOBS = 10; +const MAX_HISTORY_AGE_MS = 7 * 24 * 60 * 60 * 1_000; +const MAX_HISTORY_BYTES = 25 * 1024 * 1024; + +export interface EncryptedAnalysisHistoryEntry { + job: AnalysisJob; + events: AnalysisEnvelope[]; + savedAt: string; +} + +export interface AnalysisHistoryVault { + readAnalysisHistory(): Promise; + writeAnalysisHistory(value: T[]): Promise; + removeAnalysisHistory?(): Promise; +} + +function isEntry(value: unknown): value is EncryptedAnalysisHistoryEntry { + if (!value || typeof value !== "object") return false; + const item = value as Partial; + return Boolean( + item.job && + typeof item.job === "object" && + typeof item.job.id === "string" && + Array.isArray(item.events) && + typeof item.savedAt === "string", + ); +} + +function timestamp(entry: EncryptedAnalysisHistoryEntry): number { + const saved = Date.parse(entry.savedAt); + if (Number.isFinite(saved)) return saved; + const updated = Date.parse(entry.job.updatedAt); + return Number.isFinite(updated) ? updated : 0; +} + +/** + * Encrypted terminal-report retention layered over the V2 journal. The + * journal remains authoritative for the active job; this store is only a + * bounded convenience archive and never participates in event sequencing. + */ +export class EncryptedAnalysisHistoryStore { + constructor(private readonly vault: AnalysisHistoryVault) {} + + private prune( + entries: readonly EncryptedAnalysisHistoryEntry[], + ): EncryptedAnalysisHistoryEntry[] { + const cutoff = Date.now() - MAX_HISTORY_AGE_MS; + const deduplicated = new Map(); + for (const entry of entries) { + if (!isEntry(entry) || timestamp(entry) < cutoff) continue; + deduplicated.set(entry.job.id, { + job: structuredClone(entry.job), + events: structuredClone(entry.events), + savedAt: entry.savedAt, + }); + } + const retained = [...deduplicated.values()].sort( + (left, right) => timestamp(right) - timestamp(left), + ); + retained.splice(MAX_HISTORY_JOBS); + while (retained.length > 0 && JSON.stringify(retained).length > MAX_HISTORY_BYTES) { + retained.pop(); + } + return retained; + } + + async get(): Promise { + const value = await this.vault.readAnalysisHistory(); + const retained = this.prune(value ?? []); + if (value && JSON.stringify(value) !== JSON.stringify(retained)) { + await this.vault.writeAnalysisHistory(retained); + } + return retained; + } + + async retain(job: AnalysisJob, events: readonly AnalysisEnvelope[]): Promise { + const current = await this.get(); + const next = this.prune([ + { + job: structuredClone(job), + events: events.map((event) => structuredClone(event)), + savedAt: new Date().toISOString(), + }, + ...current.filter((entry) => entry.job.id !== job.id), + ]); + await this.vault.writeAnalysisHistory(next); + } + + async clear(): Promise { + if (this.vault.removeAnalysisHistory) { + await this.vault.removeAnalysisHistory(); + } else { + await this.vault.writeAnalysisHistory([]); + } + } +} diff --git a/apps/extension/src/storage/credential-vault.test.ts b/apps/extension/src/storage/credential-vault.test.ts index 5f5c70c..a42002e 100644 --- a/apps/extension/src/storage/credential-vault.test.ts +++ b/apps/extension/src/storage/credential-vault.test.ts @@ -130,4 +130,18 @@ describe("CredentialVault", () => { expect(await vault.has("chatgpt")).toBe(false); expect(keys.values.size).toBe(0); }); + + it("encrypts bounded analysis history separately from provider credentials", async () => { + const storage = new MemoryStorage(); + const keys = new MemoryKeyStore(); + const vault = new CredentialVault(storage, keys, "extension-a"); + const history = [{ jobId: "job-1", articleText: "private article text" }]; + + await vault.writeAnalysisHistory(history); + + await expect(vault.readAnalysisHistory()).resolves.toEqual(history); + expect(JSON.stringify([...storage.values.values()])).not.toContain("private article text"); + await vault.remove("analysis"); + await expect(vault.readAnalysisHistory()).resolves.toBeUndefined(); + }); }); diff --git a/apps/extension/src/storage/credential-vault.ts b/apps/extension/src/storage/credential-vault.ts index b48b551..fe54b36 100644 --- a/apps/extension/src/storage/credential-vault.ts +++ b/apps/extension/src/storage/credential-vault.ts @@ -34,7 +34,9 @@ const ExaCredentialSchema = z.object({ }); export type ExaCredential = z.infer; -export type VaultPurpose = "chatgpt" | "exa"; +const AnalysisHistorySchema = z.array(z.unknown()).max(10); + +export type VaultPurpose = "chatgpt" | "exa" | "analysis"; export class MissingVaultKeyError extends Error { constructor() { @@ -179,7 +181,21 @@ export class CredentialVault { } async clear(): Promise { - await Promise.all([this.remove("chatgpt"), this.remove("exa")]); + await Promise.all([this.remove("chatgpt"), this.remove("exa"), this.remove("analysis")]); await this.keyStore.remove(KEY_ID); } + + /** Encrypted local report retention, kept generic to avoid a runtime cycle. */ + async readAnalysisHistory(): Promise { + const value = await this.read("analysis", AnalysisHistorySchema); + return value as T[] | undefined; + } + + async writeAnalysisHistory(value: T[]): Promise { + await this.write("analysis", AnalysisHistorySchema.parse(value)); + } + + async removeAnalysisHistory(): Promise { + await this.remove("analysis"); + } } diff --git a/apps/extension/src/storage/evidence-cache.test.ts b/apps/extension/src/storage/evidence-cache.test.ts new file mode 100644 index 0000000..5c77ea9 --- /dev/null +++ b/apps/extension/src/storage/evidence-cache.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, it } from "vitest"; +import { EvidenceCache } from "./evidence-cache"; + +describe("EvidenceCache", () => { + it("isolates identical provider keys by scope", async () => { + const accountA = new EvidenceCache("chatgpt:account-a"); + const accountB = new EvidenceCache("chatgpt:account-b"); + + await accountA.set("same-query", { source: "A" }, 60_000); + + await expect(accountA.get<{ source: string }>("same-query")).resolves.toEqual({ source: "A" }); + await expect(accountB.get("same-query")).resolves.toBeNull(); + }); + + it("clears one scope without removing another scope", async () => { + const accountA = new EvidenceCache("exa:key-a"); + const accountB = new EvidenceCache("exa:key-b"); + await accountA.set("query", "A", 60_000); + await accountB.set("query", "B", 60_000); + + await accountA.clear(); + + await expect(accountA.get("query")).resolves.toBeNull(); + await expect(accountB.get("query")).resolves.toBe("B"); + }); + + it("clears all scopes for account disconnect or extension reset", async () => { + const accountA = new EvidenceCache("chatgpt:account-a"); + const accountB = new EvidenceCache("exa:key-b"); + await accountA.set("query", "A", 60_000); + await accountB.set("query", "B", 60_000); + + await accountA.clearAll(); + + await expect(accountA.get("query")).resolves.toBeNull(); + await expect(accountB.get("query")).resolves.toBeNull(); + }); + + it("expires entries", async () => { + const cache = new EvidenceCache("scope"); + await cache.set("query", "expired", 1); + await new Promise((resolve) => setTimeout(resolve, 5)); + + await expect(cache.get("query")).resolves.toBeNull(); + }); +}); diff --git a/apps/extension/src/storage/evidence-cache.ts b/apps/extension/src/storage/evidence-cache.ts index 296c5bb..5fad307 100644 --- a/apps/extension/src/storage/evidence-cache.ts +++ b/apps/extension/src/storage/evidence-cache.ts @@ -13,11 +13,23 @@ interface CacheRecord { const DATABASE_NAME = "perspectica-evidence-cache-v1"; const DATABASE_VERSION = 1; const STORE_NAME = "results"; +const memoryCaches = new Map>>(); export class EvidenceCache implements EvidenceResultCache { - private readonly memory = new Map>(); + private readonly memory: Map>; + private readonly normalizedScope: string; private dbPromise: Promise | null = null; + constructor(scope = "global") { + this.normalizedScope = scope.trim().slice(0, 256) || "global"; + this.memory = memoryCaches.get(this.normalizedScope) ?? new Map(); + memoryCaches.set(this.normalizedScope, this.memory); + } + + private scopedKey(key: string): string { + return `${this.normalizedScope}:${key}`; + } + private open(): Promise { if (this.dbPromise) return this.dbPromise; if (typeof indexedDB === "undefined") { @@ -39,35 +51,38 @@ export class EvidenceCache implements EvidenceResultCache { async get(key: string): Promise { const now = Date.now(); - if (typeof indexedDB === "undefined") { - const memory = this.memory.get(key); - if (memory) { - if (memory.expiresAt > now) return structuredClone(memory.value) as T; - this.memory.delete(key); - } + const storageKey = this.scopedKey(key); + const memory = this.memory.get(storageKey); + if (memory) { + if (memory.expiresAt > now) return structuredClone(memory.value) as T; + this.memory.delete(storageKey); } const db = await this.open(); if (!db) return null; const record = await new Promise | null>((resolve) => { - const request = db.transaction(STORE_NAME, "readonly").objectStore(STORE_NAME).get(key); + const request = db + .transaction(STORE_NAME, "readonly") + .objectStore(STORE_NAME) + .get(storageKey); request.onerror = () => resolve(null); request.onsuccess = () => resolve((request.result as CacheRecord | undefined) ?? null); }); if (!record || record.expiresAt <= now) { - if (record) await this.delete(key); + if (record) await this.delete(storageKey); return null; } - this.memory.set(key, structuredClone(record)); + this.memory.set(storageKey, structuredClone(record)); return structuredClone(record.value); } async set(key: string, value: T, ttlMs: number): Promise { + const storageKey = this.scopedKey(key); const record: CacheRecord = { - key, + key: storageKey, expiresAt: Date.now() + Math.max(1, ttlMs), value: structuredClone(value), }; - this.memory.set(key, structuredClone(record)); + this.memory.set(storageKey, structuredClone(record)); const db = await this.open(); if (!db) return; await new Promise((resolve) => { @@ -90,14 +105,40 @@ export class EvidenceCache implements EvidenceResultCache { } async clear(): Promise { - this.memory.clear(); + await this.clearScope(this.normalizedScope); + } + + async clearScope(scope: string): Promise { + const prefix = `${scope.trim().slice(0, 256) || "global"}:`; + for (const key of this.memory.keys()) { + if (key.startsWith(prefix)) this.memory.delete(key); + } const db = await this.open(); if (!db) return; - await new Promise((resolve) => { + await new Promise((resolve, reject) => { + const tx = db.transaction(STORE_NAME, "readwrite"); + const request = tx.objectStore(STORE_NAME).openCursor(); + request.onsuccess = () => { + const cursor = request.result; + if (!cursor) return; + const record = cursor.value as CacheRecord; + if (record.key.startsWith(prefix)) cursor.delete(); + cursor.continue(); + }; + tx.oncomplete = () => resolve(); + tx.onerror = () => reject(tx.error ?? new Error("Could not clear evidence cache.")); + }); + } + + async clearAll(): Promise { + for (const memory of memoryCaches.values()) memory.clear(); + const db = await this.open(); + if (!db) return; + await new Promise((resolve, reject) => { const tx = db.transaction(STORE_NAME, "readwrite"); tx.objectStore(STORE_NAME).clear(); tx.oncomplete = () => resolve(); - tx.onerror = () => resolve(); + tx.onerror = () => reject(tx.error ?? new Error("Could not clear evidence cache.")); }); } } diff --git a/apps/extension/src/storage/job-journal.ts b/apps/extension/src/storage/job-journal.ts index 14124f8..0cc375d 100644 --- a/apps/extension/src/storage/job-journal.ts +++ b/apps/extension/src/storage/job-journal.ts @@ -22,9 +22,10 @@ export class AnalysisJournal { this.dbPromise = Promise.resolve(null); return this.dbPromise; } - this.dbPromise = new Promise((resolve) => { + this.dbPromise = new Promise((resolve, reject) => { const request = indexedDB.open(DATABASE_NAME, DATABASE_VERSION); - request.onerror = () => resolve(null); + request.onerror = () => + reject(request.error ?? new Error("Could not open analysis journal.")); request.onupgradeneeded = () => { const db = request.result; if (!db.objectStoreNames.contains(EVENT_STORE)) { @@ -36,7 +37,10 @@ export class AnalysisJournal { logs.createIndex("job", "jobId", { unique: false }); } }; - request.onsuccess = () => resolve(request.result); + request.onsuccess = () => { + request.result.onversionchange = () => request.result.close(); + resolve(request.result); + }; }); return this.dbPromise; } @@ -62,12 +66,12 @@ export class AnalysisJournal { if (cached) return structuredClone(cached); const db = await this.open(); if (!db) return undefined; - return new Promise((resolve) => { + return new Promise((resolve, reject) => { const request = db .transaction(EVENT_STORE, "readonly") .objectStore(EVENT_STORE) .get([jobId, sequence]); - request.onerror = () => resolve(undefined); + request.onerror = () => reject(request.error ?? new Error("Could not read analysis event.")); request.onsuccess = () => { const value = request.result as AnalysisEnvelope | undefined; if (value) this.eventMemory.set(memoryKey(jobId, sequence), structuredClone(value)); @@ -89,12 +93,12 @@ export class AnalysisJournal { .slice(0, limit) .map((event) => structuredClone(event)); } - return new Promise((resolve) => { + return new Promise((resolve, reject) => { const values: AnalysisEnvelope[] = []; const tx = db.transaction(EVENT_STORE, "readonly"); const range = IDBKeyRange.bound([jobId, afterSequence + 1], [jobId, Number.MAX_SAFE_INTEGER]); const request = tx.objectStore(EVENT_STORE).openCursor(range); - request.onerror = () => resolve([]); + request.onerror = () => reject(request.error ?? new Error("Could not read analysis events.")); request.onsuccess = () => { const cursor = request.result; if (!cursor || values.length >= limit) { @@ -112,7 +116,7 @@ export class AnalysisJournal { if (key.startsWith(`${jobId}:`)) this.eventMemory.delete(key); const db = await this.open(); if (!db) return; - await new Promise((resolve) => { + await new Promise((resolve, reject) => { const tx = db.transaction(EVENT_STORE, "readwrite"); const request = tx.objectStore(EVENT_STORE).index("job").openCursor(IDBKeyRange.only(jobId)); request.onsuccess = () => { @@ -122,7 +126,7 @@ export class AnalysisJournal { cursor.continue(); }; tx.oncomplete = () => resolve(); - tx.onerror = () => resolve(); + tx.onerror = () => reject(tx.error ?? new Error("Could not clear analysis events.")); }); } @@ -133,11 +137,11 @@ export class AnalysisJournal { this.logMemory.set(jobId, next); const db = await this.open(); if (db) { - await new Promise((resolve) => { + await new Promise((resolve, reject) => { const tx = db.transaction(LOG_STORE, "readwrite"); tx.objectStore(LOG_STORE).put({ jobId, ...entry }); tx.oncomplete = () => resolve(); - tx.onerror = () => resolve(); + tx.onerror = () => reject(tx.error ?? new Error("Could not append analysis log.")); }); } return entry; @@ -148,11 +152,11 @@ export class AnalysisJournal { if (cached) return structuredClone(cached); const db = await this.open(); if (!db) return []; - const result = await new Promise((resolve) => { + const result = await new Promise((resolve, reject) => { const values: AnalysisLogEntry[] = []; const tx = db.transaction(LOG_STORE, "readonly"); const request = tx.objectStore(LOG_STORE).index("job").openCursor(IDBKeyRange.only(jobId)); - request.onerror = () => resolve([]); + request.onerror = () => reject(request.error ?? new Error("Could not read analysis logs.")); request.onsuccess = () => { const cursor = request.result; if (!cursor) { @@ -174,7 +178,7 @@ export class AnalysisJournal { this.logMemory.delete(jobId); const db = await this.open(); if (!db) return; - await new Promise((resolve) => { + await new Promise((resolve, reject) => { const tx = db.transaction(LOG_STORE, "readwrite"); const request = tx.objectStore(LOG_STORE).index("job").openCursor(IDBKeyRange.only(jobId)); request.onsuccess = () => { @@ -184,7 +188,7 @@ export class AnalysisJournal { cursor.continue(); }; tx.oncomplete = () => resolve(); - tx.onerror = () => resolve(); + tx.onerror = () => reject(tx.error ?? new Error("Could not clear analysis logs.")); }); } } diff --git a/apps/extension/src/storage/job-store.ts b/apps/extension/src/storage/job-store.ts index b9fb118..29f3745 100644 --- a/apps/extension/src/storage/job-store.ts +++ b/apps/extension/src/storage/job-store.ts @@ -10,6 +10,10 @@ import { } from "../runtime/messages"; import type { AnalysisEnvelope, PipelineEvent } from "@perspectica/contracts/events"; import type { JsonStorageArea } from "./areas"; +import type { + EncryptedAnalysisHistoryEntry, + EncryptedAnalysisHistoryStore, +} from "./analysis-history"; import { AnalysisJournal } from "./job-journal"; const ACTIVE_JOB_KEY = "perspectica.jobs.active.v1"; @@ -33,11 +37,38 @@ export class JobStore { constructor( private readonly storage: JsonStorageArea, private readonly journal = new AnalysisJournal(), + private readonly history?: Pick, + private readonly onLegacyInvalidated?: () => Promise, ) {} async get(id: string): Promise { - const parsed = AnalysisJobSchema.safeParse(await this.storage.get(`${JOB_PREFIX}${id}`)); - return parsed.success ? parsed.data : undefined; + const raw = await this.storage.get(`${JOB_PREFIX}${id}`); + const parsed = AnalysisJobSchema.safeParse(raw); + if (!parsed.success) return undefined; + const resume = await this.storage.get(`${RESUME_PREFIX}${id}`); + if (!isLegacyJobRecord(raw, resume)) return parsed.data; + + // V1 event bodies are a different wire protocol. Never replay them as V2 + // pipeline events; invalidate the interrupted report instead of showing a + // blank terminal report or dispatching an incompatible resume payload. + const invalidated = AnalysisJobSchema.parse({ + ...parsed.data, + events: [], + status: "failed", + runToken: null, + revision: parsed.data.revision + 1, + lastEventSequence: 0, + updatedAt: new Date().toISOString(), + error: "This analysis was interrupted by an extension update. Start a new analysis.", + }); + await Promise.all([ + this.storage.set(`${JOB_PREFIX}${id}`, invalidated), + this.storage.remove(`${RESUME_PREFIX}${id}`), + this.journal.clearEvents(id), + this.journal.clearLogs(id), + ]); + await this.onLegacyInvalidated?.(); + return invalidated; } private enqueueMutation(operation: () => Promise): Promise { @@ -143,6 +174,14 @@ export class JobStore { await this.storage.remove(`${LOG_PREFIX}${id}`); } + async getHistory(): Promise { + return this.history ? this.history.get() : []; + } + + async clearHistory(): Promise { + await this.history?.clear(); + } + /** * Persist the journal row before advancing the job cursor. If storage fails * after the row is durable, a retry of the same request finds the row and @@ -215,6 +254,15 @@ export class JobStore { error: null, }); await this.setUnsafe(updated); + if (terminal && this.history) { + try { + const events = await this.journal.eventsSince(current.id, 0, 512); + await this.history.retain(updated, events); + } catch { + // Encrypted retention is a convenience archive. Journal durability + // and the active report must not depend on vault/quota availability. + } + } return updated; } @@ -230,3 +278,23 @@ export class JobStore { function terminalStatus(status: AnalysisJob["status"]): boolean { return ["complete", "partial", "failed", "cancelled"].includes(status); } + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null; +} + +function isLegacyJobRecord(value: unknown, resume: unknown): boolean { + if (!isRecord(value)) return false; + if ("analysisId" in value) return true; + if (Array.isArray(value.events) && value.events.length > 0) return true; + if (value.status !== "queued" && value.status !== "extracting" && value.status !== "analyzing") { + return false; + } + if (typeof value.runToken !== "string") return true; + if (value.searchProvider === "free") return true; + if (isRecord(resume) && isRecord(resume.request)) { + const preferences = resume.request.preferences; + if (isRecord(preferences) && !("mode" in preferences)) return true; + } + return false; +} diff --git a/apps/extension/src/storage/stores.test.ts b/apps/extension/src/storage/stores.test.ts index 7993353..ae14a7a 100644 --- a/apps/extension/src/storage/stores.test.ts +++ b/apps/extension/src/storage/stores.test.ts @@ -59,7 +59,7 @@ describe("extension stores", () => { model: "gpt-5.6-luna", reasoningEffort: "medium", mode: "balanced", - searchProvider: "exa", + searchProvider: "free", rememberChatGpt: true, }); @@ -265,4 +265,72 @@ describe("extension stores", () => { ).resolves.toMatchObject({ accepted: false, gap: true }); await expect(store.getEventsSince("job-1", 0)).resolves.toEqual([]); }); + + it("invalidates a legacy terminal job instead of rendering a blank V2 report", async () => { + const storage = new MemoryStorage(); + storage.values.set("perspectica.jobs.active.v1", "legacy-job"); + storage.values.set("perspectica.jobs.v1.legacy-job", { + ...job("legacy-job"), + analysisId: "legacy-analysis", + status: "complete", + events: [{ type: "analysis.completed", data: {} }], + lastEventSequence: 4, + }); + const store = new JobStore(storage); + + await expect(store.getActive()).resolves.toMatchObject({ + id: "legacy-job", + status: "failed", + runToken: null, + lastEventSequence: 0, + events: [], + error: expect.stringContaining("extension update"), + }); + await expect(store.getResume("legacy-job")).resolves.toBeUndefined(); + }); + + it("invalidates a legacy resumable job before dispatch can reuse its cursor", async () => { + const storage = new MemoryStorage(); + storage.values.set("perspectica.jobs.v1.legacy-job", { + ...job("legacy-job"), + status: "analyzing", + runToken: "legacy-run", + events: [], + }); + storage.values.set("perspectica.jobs.resume.v1.legacy-job", { + runToken: "legacy-run", + request: { + article: { + title: "Legacy", + author: null, + publication: null, + publishedAt: null, + canonicalUrl: "https://example.com/article", + contentType: "news", + paragraphs: [ + { id: "p-1", kind: "paragraph", text: "Legacy article text", index: 0, speaker: null }, + ], + links: [], + fingerprint: "fingerprint", + language: "en", + extraction: { + extractorVersion: "dom-v5", + extractedAt: "2026-07-29T12:00:00.000Z", + wordCount: 3, + }, + }, + client: { extensionVersion: "0.1.0" }, + preferences: { model: "gpt-5.6-luna", reasoningEffort: "medium" }, + }, + searchProvider: "free", + }); + const store = new JobStore(storage); + + await expect(store.get("legacy-job")).resolves.toMatchObject({ + status: "failed", + runToken: null, + lastEventSequence: 0, + }); + await expect(store.getResume("legacy-job")).resolves.toBeUndefined(); + }); }); diff --git a/packages/extraction/src/index.test.ts b/packages/extraction/src/index.test.ts index fcb8bd3..7bf866d 100644 --- a/packages/extraction/src/index.test.ts +++ b/packages/extraction/src/index.test.ts @@ -73,7 +73,7 @@ describe("extractArticleDocument", () => { const article = extractArticleDocument(document, "https://www.bbc.com/news/articles/example"); expect(article.author).toBe("Olivia Ireland"); - expect(article.extraction.extractorVersion).toBe("dom-v5"); + expect(article.extraction.extractorVersion).toBe("dom-v6"); }); it("removes repeated byline prefixes and a duplicated publication suffix", () => { @@ -117,6 +117,43 @@ describe("extractArticleDocument", () => { expect(article.author).toBe("Riley Reporter"); }); + it("keeps the real tab origin when hostile canonical metadata points elsewhere", () => { + document.head.innerHTML = ` + + + + + `; + document.body.innerHTML = `

This local report has enough readable text to pass article extraction safely.

`; + + const article = extractArticleDocument(document, "https://news.example/story?utm_source=mail"); + expect(article.canonicalUrl).toBe("https://news.example/story"); + }); + + it("ignores hidden text and oversized JSON-LD payloads", () => { + document.head.innerHTML = ` + + + + `; + document.body.innerHTML = ` +
+ +

This visible report has enough readable content to become the only extracted paragraph in this test.

+
+ `; + + const article = extractArticleDocument(document, "https://example.com/visible"); + + expect(article.author).toBeNull(); + expect(article.paragraphs).toHaveLength(1); + expect(article.paragraphs[0]?.text).not.toContain("hidden instruction"); + expect(article.extraction.extractorVersion).toBe("dom-v6"); + }); + it("rejects generic pages instead of treating the entire body as an article", () => { document.head.innerHTML = `Example dashboard`; document.body.innerHTML = ` diff --git a/packages/extraction/src/index.ts b/packages/extraction/src/index.ts index f3e797c..8cfa518 100644 --- a/packages/extraction/src/index.ts +++ b/packages/extraction/src/index.ts @@ -16,7 +16,7 @@ import { buildArticleIndex } from "./article-index"; export { buildArticleIndex } from "./article-index"; export type { ArticleIndex } from "@perspectica/contracts/article"; -export const EXTRACTOR_VERSION = "dom-v5"; +export const EXTRACTOR_VERSION = "dom-v6"; export class ArticleExtractionError extends Error { readonly code = "NOT_ARTICLE"; @@ -28,12 +28,65 @@ export class ArticleExtractionError extends Error { } const excludedAncestorSelector = - "nav, footer, aside, form, dialog, [aria-hidden='true'], [hidden], .advertisement, .ad, .promo, .newsletter, .comments, [class*='newsletter'], [class*='promo'], [data-component*='newsletter'], [data-component*='promo'], [data-testid*='newsletter']"; + "nav, footer, aside, form, dialog, [aria-hidden='true'], [hidden], [inert], template, noscript, .advertisement, .ad, .promo, .newsletter, .comments, [class*='newsletter'], [class*='promo'], [data-component*='newsletter'], [data-component*='promo'], [data-testid*='newsletter']"; + +const MAX_JSON_LD_SCRIPTS = 24; +const MAX_JSON_LD_CHARS = 128_000; +const MAX_JSON_LD_DEPTH = 24; +const MAX_JSON_LD_NODES = 2_000; function cleanText(value: string | null | undefined): string { return (value ?? "").replace(/\s+/g, " ").trim(); } +/** Page HTML is untrusted input, so ignore content readers cannot see. */ +function isVisibleContent(element: Element): boolean { + if (element.closest(excludedAncestorSelector)) return false; + const ownStyle = element.getAttribute("style")?.toLocaleLowerCase("en-US") ?? ""; + if (/display\s*:\s*none|visibility\s*:\s*hidden/.test(ownStyle)) return false; + try { + const style = window.getComputedStyle?.(element); + return style?.display !== "none" && style?.visibility !== "hidden"; + } catch { + return true; + } +} + +function parseBoundedJsonLd(value: string): unknown | null { + if (!value || value.length > MAX_JSON_LD_CHARS) return null; + try { + return JSON.parse(value); + } catch { + return null; + } +} + +function visitBoundedJson( + value: unknown, + visitor: (record: Record) => boolean | void, +): boolean { + const stack: Array<{ value: unknown; depth: number }> = [{ value, depth: 0 }]; + let nodes = 0; + while (stack.length > 0 && nodes < MAX_JSON_LD_NODES) { + const current = stack.pop(); + if (!current || current.depth > MAX_JSON_LD_DEPTH) continue; + nodes += 1; + if (Array.isArray(current.value)) { + for (const child of current.value.slice(0, MAX_JSON_LD_NODES - nodes)) { + stack.push({ value: child, depth: current.depth + 1 }); + } + continue; + } + if (!current.value || typeof current.value !== "object") continue; + const record = current.value as Record; + if (visitor(record) === true) return true; + if (record["@graph"] !== undefined) { + stack.push({ value: record["@graph"], depth: current.depth + 1 }); + } + } + return false; +} + function firstMeta(document: Document, selectors: string[]): string | null { for (const selector of selectors) { const value = document.querySelector(selector)?.content; @@ -76,17 +129,15 @@ function cleanAuthorCandidate( function jsonLdAuthors(document: Document, publication: string | null): string[] { const authors: string[] = []; - - const visit = (value: unknown): void => { - if (Array.isArray(value)) { - value.forEach(visit); - return; - } - if (!value || typeof value !== "object") return; - - const record = value as Record; - const type = Array.isArray(record["@type"]) ? record["@type"] : [record["@type"]]; - if (type.some((entry) => entry === "NewsArticle" || entry === "Article")) { + const scripts = [ + ...document.querySelectorAll("script[type='application/ld+json']"), + ].slice(0, MAX_JSON_LD_SCRIPTS); + for (const script of scripts) { + const parsed = parseBoundedJsonLd(script.textContent ?? ""); + if (!parsed) continue; + visitBoundedJson(parsed, (record) => { + const type = Array.isArray(record["@type"]) ? record["@type"] : [record["@type"]]; + if (!type.some((entry) => entry === "NewsArticle" || entry === "Article")) return; const authorValues = Array.isArray(record.author) ? record.author : [record.author]; for (const author of authorValues) { const name = @@ -98,19 +149,7 @@ function jsonLdAuthors(document: Document, publication: string | null): string[] const cleaned = cleanAuthorCandidate(typeof name === "string" ? name : null, publication); if (cleaned) authors.push(cleaned); } - } - - if (record["@graph"]) visit(record["@graph"]); - }; - - for (const script of document.querySelectorAll( - "script[type='application/ld+json']", - )) { - try { - visit(JSON.parse(script.textContent ?? "")); - } catch { - // Ignore malformed structured data and continue with visible metadata. - } + }); } return [...new Set(authors)]; @@ -123,13 +162,20 @@ function toIsoDate(value: string | null): string | null { } function getCanonicalUrl(document: Document, fallbackUrl: string): string { + const actualUrl = normalizeCanonicalUrl(fallbackUrl); + if (!actualUrl) { + throw new ArticleExtractionError("The current page does not have a valid web URL."); + } const canonical = document.querySelector("link[rel='canonical']")?.href; const socialUrl = firstMeta(document, ["meta[property='og:url']"]); - for (const candidate of [canonical, socialUrl, fallbackUrl]) { - const normalized = normalizeCanonicalUrl(candidate ?? fallbackUrl, fallbackUrl); - if (normalized) return normalized; + const actualOrigin = new URL(actualUrl).origin; + for (const candidate of [canonical, socialUrl]) { + const normalized = normalizeCanonicalUrl(candidate ?? "", actualUrl); + // Page metadata is untrusted. A cross-origin canonical must never replace + // the identity of the tab the user explicitly asked us to analyze. + if (normalized && new URL(normalized).origin === actualOrigin) return normalized; } - throw new ArticleExtractionError("The current page does not have a valid web URL."); + return actualUrl; } type ArticleRoot = { @@ -138,24 +184,18 @@ type ArticleRoot = { }; function hasStructuredArticleData(document: Document): boolean { - const visit = (value: unknown): boolean => { - if (Array.isArray(value)) return value.some(visit); - if (!value || typeof value !== "object") return false; - const record = value as Record; - const types = Array.isArray(record["@type"]) ? record["@type"] : [record["@type"]]; - if (types.some((entry) => entry === "NewsArticle" || entry === "Article")) return true; - return record["@graph"] !== undefined && visit(record["@graph"]); - }; - - return [ - ...document.querySelectorAll("script[type='application/ld+json']"), - ].some((script) => { - try { - return visit(JSON.parse(script.textContent ?? "")); - } catch { - return false; - } - }); + return [...document.querySelectorAll("script[type='application/ld+json']")] + .slice(0, MAX_JSON_LD_SCRIPTS) + .some((script) => { + const parsed = parseBoundedJsonLd(script.textContent ?? ""); + return ( + parsed !== null && + visitBoundedJson(parsed, (record) => { + const types = Array.isArray(record["@type"]) ? record["@type"] : [record["@type"]]; + return types.some((entry) => entry === "NewsArticle" || entry === "Article"); + }) + ); + }); } function hasArticleMetadata(document: Document): boolean { @@ -186,7 +226,7 @@ function chooseArticleRoot(document: Document, canonicalUrl: string): ArticleRoo const scored = candidates .map((element) => { const paragraphs = [...element.querySelectorAll("p")].filter( - (paragraph) => cleanText(paragraph.textContent).length >= 40, + (paragraph) => isVisibleContent(paragraph) && cleanText(paragraph.textContent).length >= 40, ); const textLength = paragraphs.reduce( (total, paragraph) => total + cleanText(paragraph.textContent).length, @@ -236,7 +276,7 @@ function chooseArticleRoot(document: Document, canonicalUrl: string): ArticleRoo // metadata distinguishes it from a generic page shell. const body = document.body ?? document.documentElement; const bodyParagraphs = [...body.querySelectorAll("p")].filter( - (paragraph) => cleanText(paragraph.textContent).length >= 40, + (paragraph) => isVisibleContent(paragraph) && cleanText(paragraph.textContent).length >= 40, ); if (metadata && !genericPath && bodyParagraphs.length > 0) { return { element: body, status: "article" }; @@ -316,7 +356,7 @@ function extractParagraphs(root: Element): { let truncated = false; for (const element of elements) { - if (element.closest(excludedAncestorSelector)) continue; + if (!isVisibleContent(element)) continue; const originalText = cleanText(element.textContent); const tagName = element.tagName.toLocaleLowerCase("en-US"); const kind = @@ -384,7 +424,7 @@ function extractLinks( for (const anchor of root.querySelectorAll("a[href]")) { if (links.length >= ARTICLE_MAX_LINKS) break; - if (anchor.closest(excludedAncestorSelector)) continue; + if (!isVisibleContent(anchor)) continue; const normalized = normalizeCanonicalUrl(anchor.href, canonicalUrl); if (!normalized || normalized === canonicalUrl || seen.has(normalized)) continue; seen.add(normalized); From e4869c9b3b503fb1ef79f531939f24f3d19797d8 Mon Sep 17 00:00:00 2001 From: austin Date: Mon, 3 Aug 2026 11:46:15 -0500 Subject: [PATCH 2/6] feat: add V2 research depth and free evidence routing --- apps/extension/entrypoints/offscreen/main.ts | 142 +++++- .../src/providers/free-evidence.test.ts | 99 ++++ apps/extension/src/providers/free-evidence.ts | 472 ++++++++++++++++++ apps/extension/src/runtime/messages.test.ts | 45 ++ apps/extension/src/runtime/messages.ts | 11 +- .../src/runtime/report-reuse.test.ts | 10 + apps/extension/src/runtime/report-reuse.ts | 20 +- .../src/storage/preferences-store.ts | 10 +- packages/contracts/src/evidence.ts | 2 +- packages/contracts/src/index.test.ts | 18 + packages/contracts/src/index.ts | 79 ++- packages/contracts/src/limits.ts | 42 +- packages/contracts/src/preferences.ts | 15 +- packages/intelligence/src/budgets.ts | 22 +- .../src/compass/calculate.test.ts | 95 ++++ .../intelligence/src/compass/calculate.ts | 142 ++++-- .../intelligence/src/evidence/adjudication.ts | 8 +- .../intelligence/src/evidence/sufficiency.ts | 2 +- packages/intelligence/src/pipeline.ts | 16 +- packages/intelligence/src/planning/lens.ts | 6 +- .../intelligence/src/synthesis/perspective.ts | 4 +- 21 files changed, 1176 insertions(+), 84 deletions(-) create mode 100644 apps/extension/src/providers/free-evidence.test.ts create mode 100644 apps/extension/src/providers/free-evidence.ts create mode 100644 packages/intelligence/src/compass/calculate.test.ts diff --git a/apps/extension/entrypoints/offscreen/main.ts b/apps/extension/entrypoints/offscreen/main.ts index 812ec15..ba57837 100644 --- a/apps/extension/entrypoints/offscreen/main.ts +++ b/apps/extension/entrypoints/offscreen/main.ts @@ -21,6 +21,7 @@ import { } from "../../src/runtime/messages"; import { ExaEvidenceRetriever } from "../../src/providers/exa-evidence"; import { NativeChatGptEvidenceRetriever } from "../../src/providers/chatgpt-evidence"; +import { FreeEvidenceRetriever } from "../../src/providers/free-evidence"; import { IndexedDbAnalysisArtifactStore } from "../../src/storage/analysis-artifacts"; import { EvidenceCache } from "../../src/storage/evidence-cache"; import { describeError, redactText, serializeRedacted } from "../../src/runtime/redaction"; @@ -29,13 +30,66 @@ const activeJobs = new Map(); const artifactStore = new IndexedDbAnalysisArtifactStore(); const telemetryTails = new Map>(); +const runCaches = new Map(); +const OFFSCREEN_IDLE_CLOSE_MS = 5_000; +let offscreenCloseTimer: ReturnType | null = null; function artifactKey(jobId: string, runToken: string): string { return `${jobId}:${runToken}`; } +function runCacheScope( + jobId: string, + runToken: string, + cacheScope: string | null | undefined, +): string { + const providerScope = (cacheScope?.trim() || "global").slice(0, 190); + return `run:${providerScope}:${jobId}:${runToken}`.slice(0, 256); +} + +function cacheForRun( + jobId: string, + runToken: string, + cacheScope: string | null | undefined, +): EvidenceCache { + const key = artifactKey(jobId, runToken); + const existing = runCaches.get(key); + if (existing) return existing; + const cache = new EvidenceCache(runCacheScope(jobId, runToken, cacheScope)); + runCaches.set(key, cache); + return cache; +} + +async function clearRunCache(jobId: string, runToken: string): Promise { + const key = artifactKey(jobId, runToken); + const cache = runCaches.get(key); + runCaches.delete(key); + await cache?.clear().catch((error: unknown) => { + console.warn("[perspectica] run-scoped evidence cache cleanup failed", describeError(error)); + }); +} + +function cancelScheduledOffscreenClose(): void { + if (offscreenCloseTimer !== null) { + clearTimeout(offscreenCloseTimer); + offscreenCloseTimer = null; + } +} + +function scheduleOffscreenClose(): void { + cancelScheduledOffscreenClose(); + offscreenCloseTimer = setTimeout(() => { + offscreenCloseTimer = null; + if (activeJobs.size > 0) return; + void chrome.offscreen.closeDocument().catch((error: unknown) => { + console.debug("[perspectica] offscreen document was already closed", describeError(error)); + }); + }, OFFSCREEN_IDLE_CLOSE_MS); +} + function queueTelemetry( jobId: string, + runToken: string, entry: Omit & { timestamp?: string }, ): Promise { const sanitized: AnalysisLogInput = { @@ -49,7 +103,7 @@ function queueTelemetry( const previous = telemetryTails.get(jobId) ?? Promise.resolve(); const next: Promise = previous .catch(() => undefined) - .then(() => sendInternal({ type: "internal.analysis.log", jobId, entry: sanitized })) + .then(() => sendInternal({ type: "internal.analysis.log", jobId, runToken, entry: sanitized })) .then(() => undefined) .catch((error: unknown) => { console.warn("[perspectica] telemetry persistence failed", describeError(error)); @@ -83,8 +137,8 @@ async function sendInternal(request: InternalRequestInput, attempts = 3): Pro throw lastError instanceof Error ? lastError : new Error("The extension runtime is unavailable."); } -function logTelemetry(jobId: string, telemetry: PipelineTelemetry): void { - void queueTelemetry(jobId, { +function logTelemetry(jobId: string, runToken: string, telemetry: PipelineTelemetry): void { + void queueTelemetry(jobId, runToken, { level: "info", scope: "pipeline", event: "phase.snapshot", @@ -106,9 +160,11 @@ function logTelemetry(jobId: string, telemetry: PipelineTelemetry): void { async function createRetriever( jobId: string, + runToken: string, modelId: string, reasoningEffort: ReasoningEffort, providerKind: SearchProviderKind, + cacheScope: string | null | undefined, ): Promise<{ model: ReturnType>; retriever: EvidenceRetriever }> { const chatgpt = createChatGPT({ credentials: () => sendInternal({ type: "internal.auth.getTokens" }), @@ -116,6 +172,7 @@ async function createRetriever( reasoningEffort, textVerbosity: "low", }); + const cache = cacheForRun(jobId, runToken, cacheScope); if (providerKind === "exa") { const secret = await sendInternal<{ apiKey: string }>({ type: "internal.providers.getSecret", @@ -127,7 +184,7 @@ async function createRetriever( secret.apiKey, undefined, (diagnostics) => { - void queueTelemetry(jobId, { + void queueTelemetry(jobId, runToken, { level: diagnostics.outcome === "failed" ? "error" : "debug", scope: "provider.exa", event: `mission.${diagnostics.outcome}`, @@ -135,7 +192,25 @@ async function createRetriever( payload: serializeRedacted(diagnostics), }); }, - new EvidenceCache(), + cache, + ), + }; + } + if (providerKind === "free") { + return { + model: chatgpt(modelId), + retriever: new FreeEvidenceRetriever( + globalThis.fetch.bind(globalThis), + (diagnostics) => { + void queueTelemetry(jobId, runToken, { + level: diagnostics.outcome === "failed" ? "error" : "debug", + scope: "provider.free", + event: `mission.${diagnostics.outcome}`, + message: `Free research mission ${diagnostics.missionId} ${diagnostics.outcome}.`, + payload: serializeRedacted(diagnostics), + }); + }, + cache, ), }; } @@ -145,7 +220,7 @@ async function createRetriever( chatgpt, modelId, (diagnostics) => { - void queueTelemetry(jobId, { + void queueTelemetry(jobId, runToken, { level: diagnostics.outcome === "failed" ? "error" : "debug", scope: "provider.chatgpt", event: `global-search.${diagnostics.outcome}`, @@ -153,7 +228,7 @@ async function createRetriever( payload: serializeRedacted(diagnostics), }); }, - new EvidenceCache(), + cache, ), }; } @@ -164,6 +239,7 @@ async function testSearchProvider( { type: "offscreen.providers.test" } >, ): Promise<{ available: true; sourceCount: number }> { + if (command.provider === "free") return { available: true, sourceCount: 0 }; const chatgpt = createChatGPT({ credentials: () => sendInternal({ type: "internal.auth.getTokens" }), defaultModel: command.preferences.model, @@ -171,7 +247,7 @@ async function testSearchProvider( textVerbosity: "low", }); if (command.provider !== "chatgpt") - throw new Error("Only ChatGPT web search is tested in the analysis runtime."); + throw new Error("Exa API connectivity is tested from the settings key flow."); const controller = new AbortController(); const timeout = setTimeout(() => controller.abort(), 45_000); try { @@ -217,6 +293,7 @@ async function runJob( >, ): Promise { activeJobs.get(jobId)?.controller.abort(); + cancelScheduledOffscreenClose(); const controller = new AbortController(); activeJobs.set(jobId, { controller, runToken: command.runToken }); let sequence = command.initialSequence; @@ -229,9 +306,11 @@ async function runJob( }; const { model, retriever } = await createRetriever( jobId, + command.runToken, preferences.model, preferences.reasoningEffort, command.searchProvider, + command.cacheScope, ); for await (const event of analyzeArticle({ article: command.request.article, @@ -241,8 +320,9 @@ async function runJob( modelVersion: preferences.model, reasoningEffort: preferences.reasoningEffort, mode: preferences.mode, + depth: preferences.depth, signal: controller.signal, - onTelemetry: (telemetry) => logTelemetry(jobId, telemetry), + onTelemetry: (telemetry) => logTelemetry(jobId, command.runToken, telemetry), onArtifacts: async (artifacts) => { for (const key of completedArtifacts.keys()) { if (key !== artifactKey(jobId, command.runToken)) completedArtifacts.delete(key); @@ -251,7 +331,7 @@ async function runJob( await artifactStore.set(jobId, command.runToken, artifacts); }, })) { - void queueTelemetry(jobId, { + void queueTelemetry(jobId, command.runToken, { timestamp: event.emittedAt, level: event.type === "analysis.failed" ? "error" : "info", scope: "pipeline.events", @@ -300,7 +380,12 @@ async function runJob( } } } finally { - if (activeJobs.get(jobId)?.controller === controller) activeJobs.delete(jobId); + const ownsActiveJob = activeJobs.get(jobId)?.controller === controller; + if (ownsActiveJob) activeJobs.delete(jobId); + if (ownsActiveJob || activeJobs.get(jobId)?.runToken !== command.runToken) { + await clearRunCache(jobId, command.runToken); + } + if (activeJobs.size === 0) scheduleOffscreenClose(); } } @@ -312,6 +397,7 @@ async function runRetryJob( >, ): Promise { activeJobs.get(jobId)?.controller.abort(); + cancelScheduledOffscreenClose(); const controller = new AbortController(); activeJobs.set(jobId, { controller, runToken: command.runToken }); let sequence = command.initialSequence; @@ -344,9 +430,11 @@ async function runRetryJob( }; const { model, retriever } = await createRetriever( jobId, + command.runToken, preferences.model, preferences.reasoningEffort, command.searchProvider, + command.cacheScope, ); for await (const event of retryArticleSections({ artifacts, @@ -354,9 +442,9 @@ async function runRetryJob( adjudicator: createModelEvidenceAdjudicator(model), sections: command.sections, signal: controller.signal, - onTelemetry: (telemetry) => logTelemetry(jobId, telemetry), + onTelemetry: (telemetry) => logTelemetry(jobId, command.runToken, telemetry), })) { - void queueTelemetry(jobId, { + void queueTelemetry(jobId, command.runToken, { timestamp: event.emittedAt, level: event.type === "analysis.failed" ? "error" : "info", scope: "pipeline.events", @@ -401,7 +489,12 @@ async function runRetryJob( } } } finally { - if (activeJobs.get(jobId)?.controller === controller) activeJobs.delete(jobId); + const ownsActiveJob = activeJobs.get(jobId)?.controller === controller; + if (ownsActiveJob) activeJobs.delete(jobId); + if (ownsActiveJob || activeJobs.get(jobId)?.runToken !== command.runToken) { + await clearRunCache(jobId, command.runToken); + } + if (activeJobs.size === 0) scheduleOffscreenClose(); } } @@ -414,14 +507,18 @@ chrome.runtime.onMessage.addListener((raw, sender, sendResponse) => { return false; } if (command.data.type === "offscreen.providers.test") { - void testSearchProvider(command.data).then( - (result) => sendResponse({ ok: true, data: result }), - (error: unknown) => - sendResponse({ - ok: false, - error: publicError(error, "ChatGPT web search is not available for this account."), - }), - ); + void testSearchProvider(command.data) + .then( + (result) => sendResponse({ ok: true, data: result }), + (error: unknown) => + sendResponse({ + ok: false, + error: publicError(error, "ChatGPT web search is not available for this account."), + }), + ) + .finally(() => { + if (activeJobs.size === 0) scheduleOffscreenClose(); + }); return true; } if (command.data.type === "offscreen.analysis.cancel") { @@ -430,6 +527,7 @@ chrome.runtime.onMessage.addListener((raw, sender, sendResponse) => { if (accepted && active) { active.controller.abort(); activeJobs.delete(command.data.jobId); + scheduleOffscreenClose(); } sendResponse({ accepted, cancelled: accepted }); return false; diff --git a/apps/extension/src/providers/free-evidence.test.ts b/apps/extension/src/providers/free-evidence.test.ts new file mode 100644 index 0000000..e742993 --- /dev/null +++ b/apps/extension/src/providers/free-evidence.test.ts @@ -0,0 +1,99 @@ +import { describe, expect, it, vi } from "vitest"; +import type { RetrievalPlan } from "@perspectica/contracts/evidence"; +import { FreeEvidenceRetriever, isSafePublisherUrl } from "./free-evidence"; + +const plan: RetrievalPlan = { + missions: [ + { + id: "mission-free", + claimIds: ["claim-1"], + purpose: "independent-verification", + queryVariants: ["example public report"], + priority: 1, + estimatedCost: 1, + freshness: "recent", + preferredSourceTypes: ["independent-reporting"], + includeDomains: [], + excludeDomains: [], + canServeSections: ["supporting"], + }, + ], + maxSources: 2, + maxConcurrency: 1, + deadlineAt: Date.now() + 20_000, +}; + +describe("free V2 evidence retrieval", () => { + it("allows only public HTTPS publisher URLs", () => { + expect(isSafePublisherUrl("https://example.com/report")).toBe(true); + expect(isSafePublisherUrl("http://example.com/report")).toBe(false); + expect(isSafePublisherUrl("https://127.0.0.1/report")).toBe(false); + expect(isSafePublisherUrl("https://[::1]/report")).toBe(false); + expect(isSafePublisherUrl("https://user:secret@example.com/report")).toBe(false); + }); + + it("turns discovery into source text only after a public page read", async () => { + const fetchImplementation = vi.fn(async (input: URL | RequestInfo) => { + const url = String(input); + if (url.includes("gdeltproject")) { + return new Response( + JSON.stringify({ + articles: [{ url: "https://news.example/report?utm_source=test", title: "Report" }], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + } + if (url.includes("duckduckgo")) { + return new Response(JSON.stringify({}), { + status: 200, + headers: { "content-type": "application/json" }, + }); + } + return new Response( + "

This is sufficiently long public source text for exact downstream excerpt validation.

", + { status: 200, headers: { "content-type": "text/html" } }, + ); + }); + const retriever = new FreeEvidenceRetriever(fetchImplementation as typeof fetch); + const batches = []; + for await (const batch of retriever.retrieve(plan, new AbortController().signal)) + batches.push(batch); + + expect(batches).toHaveLength(1); + expect(batches[0]?.candidates[0]).toMatchObject({ + provider: "free", + sourceUrl: "https://news.example/report", + contentKind: "source-text", + }); + expect(batches[0]?.candidates[0]?.discoveryExcerpt).toBeTruthy(); + expect(batches[0]?.candidates[0]).not.toHaveProperty("relationship"); + }); + + it("keeps discovery-only summaries non-quoteable", async () => { + const fetchImplementation = vi.fn(async (input: URL | RequestInfo) => { + const url = String(input); + if (url.includes("gdeltproject")) { + return new Response( + JSON.stringify({ articles: [{ url: "https://news.example/report", title: "Report" }] }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + } + if (url.includes("duckduckgo")) { + return new Response(JSON.stringify({}), { + status: 200, + headers: { "content-type": "application/json" }, + }); + } + return new Response("not html", { status: 200, headers: { "content-type": "text/plain" } }); + }); + const retriever = new FreeEvidenceRetriever(fetchImplementation as typeof fetch); + const batches = []; + for await (const batch of retriever.retrieve(plan, new AbortController().signal)) + batches.push(batch); + + expect(batches[0]?.candidates[0]).toMatchObject({ + contentKind: "search-summary", + discoveryExcerpt: null, + }); + }); +}); diff --git a/apps/extension/src/providers/free-evidence.ts b/apps/extension/src/providers/free-evidence.ts new file mode 100644 index 0000000..f30f3c0 --- /dev/null +++ b/apps/extension/src/providers/free-evidence.ts @@ -0,0 +1,472 @@ +import { EvidenceBatchSchema, EvidenceCandidateSchema } from "@perspectica/contracts/evidence"; +import type { + EvidenceBatch, + EvidenceCandidate, + EvidenceRetriever, + RetrievalPlan, +} from "@perspectica/contracts/evidence"; +import { normalizeCanonicalUrl } from "@perspectica/contracts/url"; +import { runPriorityTasksStream, sourceIdFor } from "@perspectica/intelligence"; +import type { EvidenceResultCache } from "../storage/evidence-cache"; + +const GDELT_MIN_INTERVAL_MS = 5_000; +const CACHE_TTL_MS = 10 * 60_000; +const MAX_RESULT_CONTENT = 14_000; +const MAX_RESPONSE_BYTES = 1_500_000; +const MAX_FETCH_URLS = 4; + +export interface FreeEvidenceDiagnostics { + missionId: string; + durationMs: number; + resultCount: number; + cacheHit: boolean; + outcome: "ready" | "failed"; + error?: string; +} + +function abortError(): DOMException { + return new DOMException("The operation was aborted.", "AbortError"); +} + +function isPrivateIpLiteral(hostname: string): boolean { + const host = hostname.toLowerCase().replace(/^\[|\]$/g, ""); + if (host === "localhost" || host.endsWith(".localhost") || host.endsWith(".local")) return true; + if ( + host === "::1" || + host.startsWith("fe80:") || + host.startsWith("fc") || + host.startsWith("fd") + ) { + return true; + } + const ipv4 = host.match(/^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/); + if (!ipv4) return false; + const octets = ipv4.slice(1).map(Number) as [number, number, number, number]; + if (octets.some((part) => part > 255)) return true; + return ( + octets[0] === 0 || + octets[0] === 10 || + octets[0] === 127 || + (octets[0] === 100 && octets[1] >= 64 && octets[1] <= 127) || + (octets[0] === 169 && octets[1] === 254) || + (octets[0] === 172 && octets[1] >= 16 && octets[1] <= 31) || + (octets[0] === 192 && octets[1] === 168) + ); +} + +/** Only public HTTPS pages may become quoteable source-text candidates. */ +export function isSafePublisherUrl(value: string): boolean { + try { + const url = new URL(value); + return ( + url.protocol === "https:" && + !url.username && + !url.password && + !isPrivateIpLiteral(url.hostname) + ); + } catch { + return false; + } +} + +function text(value: unknown, fallback = ""): string { + return typeof value === "string" ? value.replace(/\s+/g, " ").trim() : fallback; +} + +function boundedTextFromHtml(html: string): string { + if (typeof DOMParser === "undefined") { + return html + .replace(/<(script|style|noscript|template)[^>]*>[\s\S]*?<\/\1>/gi, " ") + .replace(/<[^>]+>/g, " ") + .replace(/\s+/g, " ") + .trim() + .slice(0, MAX_RESULT_CONTENT); + } + const document = new DOMParser().parseFromString(html, "text/html"); + for (const selector of [ + "script", + "style", + "noscript", + "template", + "nav", + "footer", + "aside", + "form", + "[hidden]", + "[aria-hidden='true']", + "[inert]", + ]) { + document.querySelectorAll(selector).forEach((element) => element.remove()); + } + const root = document.querySelector("article, main, [role='main']") ?? document.body; + return [...root.querySelectorAll("h1, h2, h3, p, blockquote")] + .map((element) => text(element.textContent)) + .filter((paragraph) => paragraph.length >= 40) + .join("\n\n") + .slice(0, MAX_RESULT_CONTENT); +} + +async function readBounded(response: Response): Promise { + const declared = Number(response.headers.get("content-length") ?? 0); + if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) { + throw new Error("The source response is too large to read safely."); + } + const value = await response.text(); + if (value.length > MAX_RESPONSE_BYTES) + throw new Error("The source response is too large to read safely."); + return value; +} + +function withAbort( + signal: AbortSignal | undefined, + timeoutMs: number, +): { signal: AbortSignal; done: () => void } { + const controller = new AbortController(); + const timer = setTimeout( + () => controller.abort(new DOMException("Request timed out.", "TimeoutError")), + timeoutMs, + ); + const abort = () => controller.abort(signal?.reason); + if (signal?.aborted) abort(); + else signal?.addEventListener("abort", abort, { once: true }); + return { + signal: controller.signal, + done: () => { + clearTimeout(timer); + signal?.removeEventListener("abort", abort); + }, + }; +} + +function publication(url: string): string { + try { + return new URL(url).hostname.replace(/^www\./, ""); + } catch { + return "Unknown publication"; + } +} + +function sourceType(url: string): EvidenceCandidate["sourceType"] { + return /\.(?:gov|mil|edu)(?:\.|\/|$)/i.test(url) || + /(?:record|filing|bill|data|report|pdf)/i.test(url) + ? "primary-record" + : "independent-reporting"; +} + +function publishedAt(value: unknown): string | null { + const candidate = text(value); + if (!candidate) return null; + const parsed = new Date(candidate); + return Number.isNaN(parsed.valueOf()) ? null : parsed.toISOString(); +} + +interface DiscoveryResult { + url: string; + title: string; + publishedAt: string | null; + note: string; +} + +export class FreeEvidenceRetriever implements EvidenceRetriever { + private readonly cache = new Map(); + private lastGdeltAt = 0; + + constructor( + private readonly fetchImplementation: typeof fetch = globalThis.fetch.bind(globalThis), + private readonly onDiagnostics?: (diagnostics: FreeEvidenceDiagnostics) => void, + private readonly persistentCache?: EvidenceResultCache, + ) {} + + private async delayForGdelt(signal: AbortSignal): Promise { + const waitMs = Math.max(0, this.lastGdeltAt + GDELT_MIN_INTERVAL_MS - Date.now()); + if (waitMs === 0) return; + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, waitMs); + const onAbort = () => { + clearTimeout(timer); + reject(abortError()); + }; + signal.addEventListener("abort", onAbort, { once: true }); + }); + } + + private async gdelt( + query: string, + maxResults: number, + signal: AbortSignal, + ): Promise { + await this.delayForGdelt(signal); + this.lastGdeltAt = Date.now(); + const endpoint = new URL("https://api.gdeltproject.org/api/v2/doc/doc"); + endpoint.searchParams.set("query", query.slice(0, 400)); + endpoint.searchParams.set("mode", "artlist"); + endpoint.searchParams.set("format", "json"); + endpoint.searchParams.set("maxrecords", String(Math.min(10, Math.max(1, maxResults)))); + const timed = withAbort(signal, 15_000); + try { + const response = await this.fetchImplementation(endpoint, { + signal: timed.signal, + credentials: "omit", + }); + if (!response.ok) return []; + const body = (JSON.parse(await readBounded(response)) as { articles?: unknown[] }).articles; + return (Array.isArray(body) ? body : []).flatMap((raw) => { + if (!raw || typeof raw !== "object") return []; + const record = raw as Record; + const url = normalizeCanonicalUrl(text(record.url)); + if (!url || !isSafePublisherUrl(url)) return []; + const title = text(record.title, "Untitled source"); + return [ + { + url, + title, + publishedAt: publishedAt(record.seendate), + note: [title, text(record.seendate), text(record.domain)].filter(Boolean).join(" · "), + }, + ]; + }); + } catch (error) { + if (signal.aborted) throw error; + return []; + } finally { + timed.done(); + } + } + + private async duckDuckGo(query: string, signal: AbortSignal): Promise { + const endpoint = new URL("https://api.duckduckgo.com/"); + endpoint.searchParams.set("q", query.slice(0, 400)); + endpoint.searchParams.set("format", "json"); + endpoint.searchParams.set("no_html", "1"); + endpoint.searchParams.set("skip_disambig", "1"); + const timed = withAbort(signal, 10_000); + try { + const response = await this.fetchImplementation(endpoint, { + signal: timed.signal, + credentials: "omit", + }); + if (!response.ok) return []; + const body = JSON.parse(await readBounded(response)) as Record; + const url = normalizeCanonicalUrl(text(body.AbstractURL)); + const note = text(body.AbstractText); + if (!url || !note || !isSafePublisherUrl(url)) return []; + return [ + { + url, + title: text(body.Heading, publication(url)), + publishedAt: null, + note, + }, + ]; + } catch (error) { + if (signal.aborted) throw error; + return []; + } finally { + timed.done(); + } + } + + private async readPublisher(url: string, signal: AbortSignal): Promise { + if (!isSafePublisherUrl(url)) return null; + const timed = withAbort(signal, 18_000); + try { + const response = await this.fetchImplementation(url, { + signal: timed.signal, + credentials: "omit", + redirect: "follow", + }); + const contentType = response.headers.get("content-type")?.toLocaleLowerCase("en-US") ?? ""; + if (!response.ok || !contentType.includes("text/html")) return null; + const finalUrl = normalizeCanonicalUrl(response.url || url); + if (!finalUrl || !isSafePublisherUrl(finalUrl)) return null; + return boundedTextFromHtml(await readBounded(response)); + } catch (error) { + if (signal.aborted) throw error; + return null; + } finally { + timed.done(); + } + } + + private async searchMission( + mission: RetrievalPlan["missions"][number], + plan: RetrievalPlan, + signal: AbortSignal, + ): Promise { + const startedAt = Date.now(); + const query = mission.queryVariants[0]?.trim() || "independent context"; + const cacheKey = JSON.stringify({ + provider: "free", + query, + mission: mission.purpose, + include: mission.includeDomains, + exclude: mission.excludeDomains, + }); + const cached = this.cache.get(cacheKey); + if (cached && cached.expiresAt > Date.now()) { + const candidates = cached.results.map((candidate) => ({ + ...candidate, + missionId: mission.id, + })); + return EvidenceBatchSchema.parse({ + missionId: mission.id, + provider: "free", + candidates, + coveredMissionIds: [mission.id], + status: "completed", + error: null, + searched: false, + cacheHit: true, + durationMs: Date.now() - startedAt, + }); + } + const persisted = await this.persistentCache?.get(cacheKey); + const persistedCandidates = (persisted ?? []) + .map((candidate) => EvidenceCandidateSchema.safeParse(candidate)) + .flatMap((parsed) => + parsed.success && parsed.data.provider === "free" ? [parsed.data] : [], + ); + if (persistedCandidates.length > 0) { + this.cache.set(cacheKey, { + expiresAt: Date.now() + CACHE_TTL_MS, + results: persistedCandidates, + }); + const candidates = persistedCandidates.map((candidate) => ({ + ...candidate, + missionId: mission.id, + })); + return EvidenceBatchSchema.parse({ + missionId: mission.id, + provider: "free", + candidates, + coveredMissionIds: [mission.id], + status: "completed", + error: null, + searched: false, + cacheHit: true, + durationMs: Date.now() - startedAt, + }); + } + + const [gdelt, answer] = await Promise.all([ + this.gdelt(query, Math.min(plan.maxSources, 10), signal), + this.duckDuckGo(query, signal), + ]); + const discoveries = [...new Map([...gdelt, ...answer].map((item) => [item.url, item])).values()] + .filter( + (item) => + !mission.excludeDomains.some((domain) => { + const host = new URL(item.url).hostname.replace(/^www\./, ""); + const excluded = domain + .trim() + .toLocaleLowerCase("en-US") + .replace(/^www\./, ""); + return host === excluded || host.endsWith(`.${excluded}`); + }), + ) + .slice(0, Math.min(MAX_FETCH_URLS, Math.max(1, plan.maxSources))); + const fetched = await Promise.all( + discoveries.map(async (item) => ({ + item, + content: await this.readPublisher(item.url, signal), + })), + ); + const candidates = fetched.map( + ({ item, content }) => + ({ + id: sourceIdFor(item.url, "candidate"), + missionId: mission.id, + sourceUrl: item.url, + title: item.title || publication(item.url), + publication: publication(item.url), + publishedAt: item.publishedAt, + sourceType: sourceType(item.url), + contentKind: content ? ("source-text" as const) : ("search-summary" as const), + content: ( + content || + item.note || + `Free discovery returned ${publication(item.url)}.` + ).slice(0, 20_000), + discoveryContext: content ? null : item.note || null, + discoveryExcerpt: content ? content.slice(0, 360) : null, + providerScore: null, + provider: "free" as const, + }) satisfies EvidenceCandidate, + ); + const parsedCandidates = candidates.flatMap((candidate) => { + const parsed = EvidenceCandidateSchema.safeParse(candidate); + return parsed.success ? [parsed.data] : []; + }); + this.cache.set(cacheKey, { expiresAt: Date.now() + CACHE_TTL_MS, results: parsedCandidates }); + await this.persistentCache?.set(cacheKey, parsedCandidates, CACHE_TTL_MS); + this.onDiagnostics?.({ + missionId: mission.id, + durationMs: Date.now() - startedAt, + resultCount: parsedCandidates.length, + cacheHit: false, + outcome: "ready", + }); + return EvidenceBatchSchema.parse({ + missionId: mission.id, + provider: "free", + candidates: parsedCandidates, + coveredMissionIds: [mission.id], + status: "completed", + error: null, + searched: true, + cacheHit: false, + durationMs: Date.now() - startedAt, + }); + } + + async *retrieve(plan: RetrievalPlan, signal: AbortSignal): AsyncIterable { + const controller = new AbortController(); + const abortFromParent = () => controller.abort(signal.reason); + if (signal.aborted) abortFromParent(); + else signal.addEventListener("abort", abortFromParent, { once: true }); + const deadlineTimer = setTimeout( + () => + controller.abort(new DOMException("Evidence retrieval deadline reached.", "TimeoutError")), + Math.max(0, plan.deadlineAt - Date.now()), + ); + const tasks = [...plan.missions].map((mission) => ({ + priority: mission.priority, + run: async (): Promise => { + const startedAt = Date.now(); + try { + return await this.searchMission(mission, plan, controller.signal); + } catch (error) { + if (signal.aborted) throw error; + const message = error instanceof Error ? error.message : String(error); + this.onDiagnostics?.({ + missionId: mission.id, + durationMs: Date.now() - startedAt, + resultCount: 0, + cacheHit: false, + outcome: "failed", + error: message, + }); + return EvidenceBatchSchema.parse({ + missionId: mission.id, + provider: "free", + candidates: [], + coveredMissionIds: [mission.id], + status: "failed", + error: message, + searched: true, + cacheHit: false, + durationMs: Date.now() - startedAt, + }); + } + }, + })); + try { + for await (const batch of runPriorityTasksStream(tasks, plan.maxConcurrency, signal)) + yield batch; + } finally { + clearTimeout(deadlineTimer); + signal.removeEventListener("abort", abortFromParent); + controller.abort(); + } + } +} diff --git a/apps/extension/src/runtime/messages.test.ts b/apps/extension/src/runtime/messages.test.ts index 52f70a9..f1d8dbf 100644 --- a/apps/extension/src/runtime/messages.test.ts +++ b/apps/extension/src/runtime/messages.test.ts @@ -45,6 +45,20 @@ describe("extension runtime protocol", () => { }, }).success, ).toBe(true); + expect( + OffscreenCommandSchema.safeParse({ + type: "offscreen.providers.test", + protocol: PERSPECTICA_RUNTIME_PROTOCOL, + provider: "free", + cacheScope: "account-scope-v1", + preferences: { + model: "gpt-5.6-luna", + reasoningEffort: "medium", + mode: "quick", + depth: "quick", + }, + }).success, + ).toBe(true); }); it("uses an explicit offscreen handshake to prevent mixed unpacked builds", () => { @@ -98,5 +112,36 @@ describe("extension runtime protocol", () => { expect(InternalRequestSchema.safeParse(base).success).toBe(true); expect(InternalRequestSchema.safeParse({ ...base, sequence: 0 }).success).toBe(false); expect(InternalRequestSchema.safeParse({ ...base, runToken: "" }).success).toBe(false); + expect( + InternalRequestSchema.safeParse({ + type: "internal.analysis.log", + requestId: "request-4", + jobId: "job-1", + runToken: "run-1", + entry: { + timestamp: "2026-07-29T12:00:00.000Z", + level: "info", + scope: "pipeline", + event: "phase.snapshot", + message: "snapshot", + payload: null, + }, + }).success, + ).toBe(true); + expect( + InternalRequestSchema.safeParse({ + type: "internal.analysis.log", + requestId: "request-5", + jobId: "job-1", + entry: { + timestamp: "2026-07-29T12:00:00.000Z", + level: "info", + scope: "pipeline", + event: "phase.snapshot", + message: "snapshot", + payload: null, + }, + }).success, + ).toBe(false); }); }); diff --git a/apps/extension/src/runtime/messages.ts b/apps/extension/src/runtime/messages.ts index 2c55668..4fbf5bb 100644 --- a/apps/extension/src/runtime/messages.ts +++ b/apps/extension/src/runtime/messages.ts @@ -10,6 +10,7 @@ const requiredText = z.string().trim().min(1); const requestId = requiredText.max(128); const jobId = requiredText.max(128); const runToken = requiredText.max(128); +const cacheScope = requiredText.max(256).nullable().optional(); const sequence = z.number().int().nonnegative(); // Increment this whenever the side panel, service worker, and offscreen @@ -18,7 +19,7 @@ const sequence = z.number().int().nonnegative(); // MV3 service worker or offscreen document. export const PERSPECTICA_RUNTIME_PROTOCOL = 6; -export const SearchProviderSchema = z.enum(["exa", "chatgpt"]); +export const SearchProviderSchema = z.enum(["free", "chatgpt", "exa"]); export type SearchProviderKind = z.infer; export const ExtensionPreferencesSchema = AnalysisPreferencesSchema.extend({ @@ -32,7 +33,8 @@ export const DEFAULT_EXTENSION_PREFERENCES: ExtensionPreferences = { model: "gpt-5.6-luna", reasoningEffort: "medium", mode: "balanced", - searchProvider: "exa", + depth: "balanced", + searchProvider: "free", rememberChatGpt: true, }; @@ -128,6 +130,7 @@ export const AnalysisResumeDataSchema = z.object({ runToken, request: OffscreenAnalysisRequestSchema, searchProvider: SearchProviderSchema, + cacheScope, }); export type AnalysisResumeData = z.infer; @@ -200,6 +203,7 @@ export const InternalRequestSchema = z.discriminatedUnion("type", [ baseRequest.extend({ type: z.literal("internal.analysis.log"), jobId, + runToken, entry: AnalysisLogInputSchema, }), baseRequest.extend({ @@ -232,6 +236,7 @@ export const OffscreenCommandSchema = z.discriminatedUnion("type", [ initialSequence: sequence.default(0), request: OffscreenAnalysisRequestSchema, searchProvider: SearchProviderSchema, + cacheScope, }), z.object({ type: z.literal("offscreen.analysis.cancel"), @@ -247,6 +252,7 @@ export const OffscreenCommandSchema = z.discriminatedUnion("type", [ initialSequence: sequence.default(0), request: OffscreenAnalysisRequestSchema, searchProvider: SearchProviderSchema, + cacheScope, sections: z.array(ReportSectionSchema).min(1).max(6), }), z.object({ @@ -254,6 +260,7 @@ export const OffscreenCommandSchema = z.discriminatedUnion("type", [ protocol: z.literal(PERSPECTICA_RUNTIME_PROTOCOL), provider: SearchProviderSchema, preferences: AnalysisPreferencesSchema.extend({ mode: AnalysisModeSchema }), + cacheScope, }), ]); export type OffscreenCommand = z.infer; diff --git a/apps/extension/src/runtime/report-reuse.test.ts b/apps/extension/src/runtime/report-reuse.test.ts index ae914c9..434761d 100644 --- a/apps/extension/src/runtime/report-reuse.test.ts +++ b/apps/extension/src/runtime/report-reuse.test.ts @@ -13,6 +13,8 @@ const article = { const preferences = { model: "gpt-5.6-luna" as const, reasoningEffort: "medium" as const, + mode: "balanced" as const, + depth: "balanced" as const, searchProvider: "exa" as const, }; const configFingerprint = createAnalysisConfigFingerprint(preferences); @@ -37,6 +39,12 @@ function job(overrides: Partial = {}): AnalysisJob { } describe("canReuseAnalysisJob", () => { + it("includes V2 pipeline versions and depth in the reuse fingerprint", () => { + expect(configFingerprint).toContain("analysis-config-v2"); + expect(configFingerprint).toContain("intelligence-graph-v2"); + expect(configFingerprint).toContain(":balanced:balanced:exa"); + }); + it("reuses a report when tab, canonical URL, and fingerprint match", () => { expect(canReuseAnalysisJob(job(), article, 7, configFingerprint)).toBe(true); }); @@ -70,6 +78,8 @@ describe("canReuseAnalysisJob", () => { const changed = [ createAnalysisConfigFingerprint({ ...preferences, model: "gpt-5.6-sol" }), createAnalysisConfigFingerprint({ ...preferences, reasoningEffort: "high" }), + createAnalysisConfigFingerprint({ ...preferences, mode: "deep" }), + createAnalysisConfigFingerprint({ ...preferences, depth: "verified" }), createAnalysisConfigFingerprint({ ...preferences, searchProvider: "chatgpt" }), ]; diff --git a/apps/extension/src/runtime/report-reuse.ts b/apps/extension/src/runtime/report-reuse.ts index 9d05670..7d87895 100644 --- a/apps/extension/src/runtime/report-reuse.ts +++ b/apps/extension/src/runtime/report-reuse.ts @@ -1,4 +1,10 @@ import { normalizeCanonicalUrl, type ArticleDocument } from "@perspectica/contracts"; +import { + ARTICLE_INDEX_VERSION, + INTELLIGENCE_PIPELINE_VERSION, + INTELLIGENCE_PROMPT_VERSION, +} from "@perspectica/contracts/limits"; +import type { AnalysisMode } from "@perspectica/contracts/preferences"; import type { AnalysisJob, ExtensionPreferences } from "./messages"; const REUSABLE_JOB_STATUSES = new Set([ @@ -11,16 +17,26 @@ const REUSABLE_JOB_STATUSES = new Set([ /** Stable identity for every preference that can materially change a report. */ export function createAnalysisConfigFingerprint( - preferences: Pick, + preferences: Pick & + Partial>, ): string { return [ - "analysis-config-v1", + "analysis-config-v2", + ARTICLE_INDEX_VERSION, + INTELLIGENCE_PIPELINE_VERSION, + INTELLIGENCE_PROMPT_VERSION, preferences.model, preferences.reasoningEffort, + canonicalMode(preferences.mode), + preferences.depth ?? "balanced", preferences.searchProvider, ].join(":"); } +function canonicalMode(mode: ExtensionPreferences["mode"] | "fast" | undefined): AnalysisMode { + return mode === "fast" ? "quick" : (mode ?? "balanced"); +} + /** Detect navigation using the browser's actual URL, not page canonical metadata. */ export function tabUrlChanged(beforeUrl: string, afterUrl: string): boolean { const before = normalizeCanonicalUrl(beforeUrl); diff --git a/apps/extension/src/storage/preferences-store.ts b/apps/extension/src/storage/preferences-store.ts index 2841103..ee0cbb5 100644 --- a/apps/extension/src/storage/preferences-store.ts +++ b/apps/extension/src/storage/preferences-store.ts @@ -11,8 +11,14 @@ export class PreferencesStore { constructor(private readonly storage: JsonStorageArea) {} async get(): Promise { - const parsed = ExtensionPreferencesSchema.safeParse(await this.storage.get(PREFERENCES_KEY)); - return parsed.success ? parsed.data : DEFAULT_EXTENSION_PREFERENCES; + const raw = await this.storage.get(PREFERENCES_KEY); + const parsed = ExtensionPreferencesSchema.safeParse(raw); + if (!parsed.success) return DEFAULT_EXTENSION_PREFERENCES; + // Parsing normalizes the retired `fast` mode to canonical `quick`; persist + // the normalized record so subsequent resume/fingerprint paths agree. + if (JSON.stringify(raw) !== JSON.stringify(parsed.data)) + await this.storage.set(PREFERENCES_KEY, parsed.data); + return parsed.data; } async set(preferences: ExtensionPreferences): Promise { diff --git a/packages/contracts/src/evidence.ts b/packages/contracts/src/evidence.ts index 12e2394..4817ccc 100644 --- a/packages/contracts/src/evidence.ts +++ b/packages/contracts/src/evidence.ts @@ -6,7 +6,7 @@ const text = z.string().trim().min(1); export const ContentKindSchema = z.enum(["source-text", "search-summary"]); export type ContentKind = z.infer; -export const EvidenceProviderSchema = z.enum(["exa", "chatgpt"]); +export const EvidenceProviderSchema = z.enum(["free", "chatgpt", "exa"]); export type EvidenceProvider = z.infer; export const SourceTypeV2Schema = z.enum([ diff --git a/packages/contracts/src/index.test.ts b/packages/contracts/src/index.test.ts index 2e03ce1..724b002 100644 --- a/packages/contracts/src/index.test.ts +++ b/packages/contracts/src/index.test.ts @@ -3,8 +3,11 @@ import { ARTICLE_MAX_CONTENT_CHARS, ArticleDocumentSchema, ExternalSourceSchema, + RESEARCH_PROFILES, + ResearchDepthSchema, normalizeCanonicalUrl, } from "./index"; +import { EvidenceProviderSchema } from "./evidence"; const baseArticle = { fingerprint: "fixture", @@ -152,3 +155,18 @@ describe("ExternalSourceSchema", () => { ).toBe(false); }); }); + +describe("V2 research depth and providers", () => { + it("exposes the four bounded depth profiles", () => { + expect(ResearchDepthSchema.options).toEqual(["quick", "balanced", "deep", "verified"]); + expect(RESEARCH_PROFILES.quick.maxModelSteps).toBe(4); + expect(RESEARCH_PROFILES.verified.maxOutputTokens).toBe(4_000); + expect(RESEARCH_PROFILES.verified.specialistTimeoutMs).toBe(180_000); + }); + + it("accepts all canonical evidence providers", () => { + for (const provider of ["free", "chatgpt", "exa"] as const) { + expect(EvidenceProviderSchema.parse(provider)).toBe(provider); + } + }); +}); diff --git a/packages/contracts/src/index.ts b/packages/contracts/src/index.ts index 6ad0b76..b292a4d 100644 --- a/packages/contracts/src/index.ts +++ b/packages/contracts/src/index.ts @@ -171,7 +171,7 @@ export type SourceCitationKind = z.infer; * source validation so discarded evidence cannot affect the final score. */ export const PoliticalContextWeightingSchema = z.object({ - articleWeight: z.number().min(0.4).max(0.6), + articleWeight: z.number().min(0.5).max(0.6), publicationHistory: z.number().min(0).max(1), journalistWork: z.number().min(0).max(1), comparableCoverage: z.number().min(0).max(1), @@ -468,6 +468,81 @@ export const SourceListResultSchema = z.object({ }); export type SourceListResult = z.infer; +/** Bounded research depth selected independently from the analysis model. */ +export const ResearchDepthSchema = z.enum(["quick", "balanced", "deep", "verified"]); +export type ResearchDepth = z.infer; + +/** + * The depth table is intentionally explicit. Consumers should not infer + * provider, model, or deadline behavior from a display label. + */ +export const ResearchProfileSchema = z.object({ + depth: ResearchDepthSchema, + maxSearchQueries: z.number().int().nonnegative().max(24), + maxSearchResults: z.number().int().nonnegative().max(100), + maxSourceReads: z.number().int().nonnegative().max(24), + maxModelSteps: z.number().int().positive().max(12), + maxOutputTokens: z.number().int().positive().max(6_000), + maxArticleContextChars: z.number().int().positive().max(60_000), + maxSourceContextChars: z.number().int().positive().max(20_000), + specialistTimeoutMs: z.number().int().positive().max(180_000), +}); +export type ResearchProfile = z.infer; + +export const RESEARCH_PROFILES: Readonly> = { + quick: { + depth: "quick", + maxSearchQueries: 4, + maxSearchResults: 12, + maxSourceReads: 4, + maxModelSteps: 4, + maxOutputTokens: 1_600, + maxArticleContextChars: 24_000, + maxSourceContextChars: 6_000, + specialistTimeoutMs: 45_000, + }, + balanced: { + depth: "balanced", + maxSearchQueries: 8, + maxSearchResults: 24, + maxSourceReads: 8, + maxModelSteps: 6, + maxOutputTokens: 2_400, + maxArticleContextChars: 48_000, + maxSourceContextChars: 10_000, + specialistTimeoutMs: 90_000, + }, + deep: { + depth: "deep", + maxSearchQueries: 12, + maxSearchResults: 40, + maxSourceReads: 12, + maxModelSteps: 8, + maxOutputTokens: 3_200, + maxArticleContextChars: 48_000, + maxSourceContextChars: 14_000, + specialistTimeoutMs: 120_000, + }, + verified: { + depth: "verified", + maxSearchQueries: 18, + maxSearchResults: 60, + maxSourceReads: 16, + maxModelSteps: 8, + maxOutputTokens: 4_000, + maxArticleContextChars: 48_000, + maxSourceContextChars: 18_000, + specialistTimeoutMs: 180_000, + }, +}; + +export function researchProfileFor(depth: ResearchDepth = "balanced"): ResearchProfile { + return RESEARCH_PROFILES[depth]; +} + +export const RESEARCH_DEPTH_PROFILES = RESEARCH_PROFILES; +export const getResearchProfile = researchProfileFor; + export const AnalysisMetadataSchema = z.object({ analysisId: requiredText, articleFingerprint: requiredText, @@ -478,6 +553,7 @@ export const AnalysisMetadataSchema = z.object({ reasoningEffort: z.enum(["none", "low", "medium", "high", "xhigh", "max"]), startedAt: z.string().datetime({ offset: true }), contentType: ContentTypeSchema, + researchDepth: ResearchDepthSchema.optional(), }); export type AnalysisMetadata = z.infer; @@ -490,6 +566,7 @@ export type AnalysisReasoningEffort = z.infer; diff --git a/packages/contracts/src/limits.ts b/packages/contracts/src/limits.ts index 735c9ed..1ede26e 100644 --- a/packages/contracts/src/limits.ts +++ b/packages/contracts/src/limits.ts @@ -3,13 +3,21 @@ export const INTELLIGENCE_PIPELINE_VERSION = "intelligence-graph-v2" as const; export const INTELLIGENCE_PROMPT_VERSION = "lens-planner-v2" as const; export const ANALYSIS_LIMITS = { - fast: { + quick: { deepPassageCharacters: 16_000, maxClaims: 4, maxMissions: 4, maxSources: 6, maxConcurrency: 2, totalDeadlineMs: 45_000, + maxSearchQueries: 4, + maxSearchResults: 12, + maxSourceReads: 4, + maxModelSteps: 4, + maxArticleContextChars: 24_000, + maxSourceContextChars: 6_000, + specialistTimeoutMs: 45_000, + modelOutputTokens: 1_600, }, balanced: { deepPassageCharacters: 22_000, @@ -18,6 +26,14 @@ export const ANALYSIS_LIMITS = { maxSources: 8, maxConcurrency: 2, totalDeadlineMs: 70_000, + maxSearchQueries: 8, + maxSearchResults: 24, + maxSourceReads: 8, + maxModelSteps: 6, + maxArticleContextChars: 48_000, + maxSourceContextChars: 10_000, + specialistTimeoutMs: 90_000, + modelOutputTokens: 2_400, }, deep: { deepPassageCharacters: 30_000, @@ -26,6 +42,30 @@ export const ANALYSIS_LIMITS = { maxSources: 12, maxConcurrency: 2, totalDeadlineMs: 110_000, + maxSearchQueries: 12, + maxSearchResults: 40, + maxSourceReads: 12, + maxModelSteps: 8, + maxArticleContextChars: 48_000, + maxSourceContextChars: 14_000, + specialistTimeoutMs: 120_000, + modelOutputTokens: 3_200, + }, + verified: { + deepPassageCharacters: 30_000, + maxClaims: 8, + maxMissions: 8, + maxSources: 16, + maxConcurrency: 2, + totalDeadlineMs: 180_000, + maxSearchQueries: 18, + maxSearchResults: 60, + maxSourceReads: 16, + maxModelSteps: 8, + maxArticleContextChars: 48_000, + maxSourceContextChars: 18_000, + specialistTimeoutMs: 180_000, + modelOutputTokens: 4_000, }, } as const; diff --git a/packages/contracts/src/preferences.ts b/packages/contracts/src/preferences.ts index 17f1077..4b99fa5 100644 --- a/packages/contracts/src/preferences.ts +++ b/packages/contracts/src/preferences.ts @@ -2,8 +2,15 @@ import { z } from "zod"; import { AnalysisModelSchema, AnalysisReasoningEffortSchema } from "./index"; import { ANALYSIS_LIMITS } from "./limits"; -export const AnalysisModeSchema = z.enum(["fast", "balanced", "deep"]); -export type AnalysisMode = z.infer; +const CanonicalAnalysisModeSchema = z.enum(["quick", "balanced", "deep", "verified"]); + +/** `fast` was the pre-depth name for `quick`; parse it only at migration edges. */ +export const AnalysisModeSchema = z.preprocess( + (value) => (value === "fast" ? "quick" : value), + CanonicalAnalysisModeSchema, +); +export type AnalysisMode = z.infer; +export type LegacyAnalysisMode = "fast"; export const V2AnalysisPreferencesSchema = z.object({ model: AnalysisModelSchema, @@ -12,6 +19,6 @@ export const V2AnalysisPreferencesSchema = z.object({ }); export type V2AnalysisPreferences = z.infer; -export function budgetForMode(mode: AnalysisMode) { - return ANALYSIS_LIMITS[mode]; +export function budgetForMode(mode: AnalysisMode | LegacyAnalysisMode) { + return ANALYSIS_LIMITS[mode === "fast" ? "quick" : mode]; } diff --git a/packages/intelligence/src/budgets.ts b/packages/intelligence/src/budgets.ts index 821d890..6f064d3 100644 --- a/packages/intelligence/src/budgets.ts +++ b/packages/intelligence/src/budgets.ts @@ -1,6 +1,8 @@ import type { AnalysisReasoningEffort } from "@perspectica/contracts"; import { ANALYSIS_LIMITS, type AnalysisBudgetMode } from "@perspectica/contracts/limits"; +export type LegacyAnalysisBudgetMode = "fast"; + export interface AnalysisBudget { mode: AnalysisBudgetMode; deepPassageCharacters: number; @@ -9,19 +11,31 @@ export interface AnalysisBudget { maxSources: number; maxConcurrency: number; totalDeadlineMs: number; + maxSearchQueries: number; + maxSearchResults: number; + maxSourceReads: number; + maxModelSteps: number; + maxArticleContextChars: number; + maxSourceContextChars: number; + specialistTimeoutMs: number; modelOutputTokens: number; reasoningEffort: AnalysisReasoningEffort; } +function canonicalMode(mode: AnalysisBudgetMode | LegacyAnalysisBudgetMode): AnalysisBudgetMode { + return mode === "fast" ? "quick" : mode; +} + export function resolveAnalysisBudget( - mode: AnalysisBudgetMode = "balanced", + mode: AnalysisBudgetMode | LegacyAnalysisBudgetMode = "balanced", reasoningEffort: AnalysisReasoningEffort = "medium", ): AnalysisBudget { - const limits = ANALYSIS_LIMITS[mode]; + const canonical = canonicalMode(mode); + const limits = ANALYSIS_LIMITS[canonical]; return { ...limits, - mode, + mode: canonical, reasoningEffort, - modelOutputTokens: mode === "fast" ? 1_600 : mode === "deep" ? 3_000 : 2_200, + modelOutputTokens: limits.modelOutputTokens, }; } diff --git a/packages/intelligence/src/compass/calculate.test.ts b/packages/intelligence/src/compass/calculate.test.ts new file mode 100644 index 0000000..713b482 --- /dev/null +++ b/packages/intelligence/src/compass/calculate.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from "vitest"; +import { projectCompassWithContext } from "./calculate"; + +const index = { + paragraphs: { + "p-1": { id: "p-1", text: "A policy framing sentence.", speaker: null }, + }, + sentences: { + "s-1": { id: "s-1", paragraphId: "p-1", text: "A policy framing sentence.", speaker: null }, + }, +} as never; + +const plan = { + articleSignals: { + compass: [ + { + paragraphIds: ["p-1"], + sentenceIds: ["s-1"], + score: -1, + direction: "left", + strength: 1, + explanation: "Article framing.", + attributed: false, + }, + ], + }, +} as never; + +function context(score: number) { + return { + status: "ready" as const, + summary: "Verified context.", + signals: [ + { + id: "context-1", + sourceKind: "publication-history" as const, + subject: "Example publication", + score, + direction: score < 0 ? ("left" as const) : ("right" as const), + strength: 1, + relevance: 1, + explanation: "Bounded publication context.", + sourceTitle: "Publication history", + publication: "Example", + url: "https://example.com/context", + citationKind: "source-excerpt" as const, + excerpt: "A verified context excerpt.", + }, + ], + weighting: { + articleWeight: 0.5, + publicationHistory: 1, + journalistWork: 0, + comparableCoverage: 0, + topicContext: 0, + rationale: "Article and context each contribute at most half.", + }, + }; +} + +describe("V2 one-dimensional compass", () => { + it("limits context-assisted influence to fifty percent", () => { + const result = projectCompassWithContext(index, plan, context(1)); + expect(result.basis).toBe("context-assisted"); + expect(result.influence).toEqual({ + article: 0.5, + publication: 0.5, + journalist: 0, + comparableCoverage: 0, + topicContext: 0, + }); + expect(result.score).toBe(0); + }); + + it("uses Unclear only when article and context signals are both absent", () => { + const result = projectCompassWithContext(index, { articleSignals: { compass: [] } } as never, { + status: "empty", + summary: "No context.", + signals: [], + }); + expect(result.label).toBe("unclear"); + expect(result.score).toBeNull(); + }); + + it("permits a low-confidence context-led numeric placement", () => { + const result = projectCompassWithContext( + index, + { articleSignals: { compass: [] } } as never, + context(-1), + ); + expect(result.basis).toBe("context-led"); + expect(result.score).toBe(-1); + expect(result.confidence).toBe("low"); + }); +}); diff --git a/packages/intelligence/src/compass/calculate.ts b/packages/intelligence/src/compass/calculate.ts index 7ab6432..19705d4 100644 --- a/packages/intelligence/src/compass/calculate.ts +++ b/packages/intelligence/src/compass/calculate.ts @@ -27,6 +27,90 @@ function label(score: number): CompassResult["label"] { return "far-right"; } +function roundedScore(value: number): number { + return Math.round(clamp(value, -3, 3) * 100) / 100; +} + +function contextScore(context: PoliticalContextResult): { + score: number; + weight: number; +} { + const weighted = context.signals.reduce( + (total, signal) => total + signal.score * signal.strength * signal.relevance, + 0, + ); + const weight = context.signals.reduce( + (total, signal) => total + signal.strength * signal.relevance, + 0, + ); + return { score: weight > 0 ? roundedScore(weighted / weight) : 0, weight }; +} + +function contextCounts(context: PoliticalContextResult) { + return { + publication: context.signals.filter((signal) => signal.sourceKind === "publication-history") + .length, + journalist: context.signals.filter((signal) => signal.sourceKind === "journalist-work").length, + comparableCoverage: context.signals.filter( + (signal) => signal.sourceKind === "comparable-coverage", + ).length, + topicContext: context.signals.filter((signal) => signal.sourceKind === "topic-context").length, + }; +} + +function contextInfluence( + context: PoliticalContextResult, + contextWeight: number, +): Pick< + CompassResult["influence"], + "publication" | "journalist" | "comparableCoverage" | "topicContext" +> { + const counts = contextCounts(context); + const weighting = context.weighting; + const raw = { + publication: weighting?.publicationHistory ?? 0.65, + journalist: weighting?.journalistWork ?? 0.1, + comparableCoverage: weighting?.comparableCoverage ?? 0.2, + topicContext: weighting?.topicContext ?? 0.05, + }; + const available = (Object.keys(counts) as Array).filter( + (kind) => counts[kind] > 0, + ); + const totalWeight = available.reduce((sum, kind) => sum + raw[kind], 0); + const totalCount = available.reduce((sum, kind) => sum + counts[kind], 0); + const denominator = totalWeight > 0 ? totalWeight : totalCount; + return { + publication: + contextWeight * + (denominator > 0 + ? totalWeight > 0 + ? raw.publication / denominator + : counts.publication / denominator + : 0), + journalist: + contextWeight * + (denominator > 0 + ? totalWeight > 0 + ? raw.journalist / denominator + : counts.journalist / denominator + : 0), + comparableCoverage: + contextWeight * + (denominator > 0 + ? totalWeight > 0 + ? raw.comparableCoverage / denominator + : counts.comparableCoverage / denominator + : 0), + topicContext: + contextWeight * + (denominator > 0 + ? totalWeight > 0 + ? raw.topicContext / denominator + : counts.topicContext / denominator + : 0), + }; +} + const display: Record = { "far-left": "Far left", left: "Left", @@ -132,49 +216,43 @@ export function projectCompassWithContext( context: PoliticalContextResult, ): CompassResult { const article = projectArticleCompass(index, plan); - if (context.status !== "ready" || context.signals.length === 0 || article.score === null) { + if (context.status !== "ready" || context.signals.length === 0) { return CompassResultSchema.parse({ ...article, context }); } - const weightedContext = context.signals.reduce( - (total, signal) => total + signal.score * signal.strength * signal.relevance, - 0, + const contextual = contextScore(context); + if (contextual.weight <= 0) return CompassResultSchema.parse({ ...article, context }); + // Direct article signals anchor the placement, while validated context can + // contribute exactly the remaining half. Context-only placement is allowed + // when the article contains no endorsed spectrum signal, but remains low + // confidence and is never presented as article-owned framing. + const hasArticle = article.score !== null; + const score = roundedScore( + hasArticle ? (article.score ?? 0) * 0.5 + contextual.score * 0.5 : contextual.score, ); - const contextWeight = context.signals.reduce( - (total, signal) => total + signal.strength * signal.relevance, - 0, - ); - if (contextWeight <= 0) return CompassResultSchema.parse({ ...article, context }); - const score = - Math.round(clamp(article.score * 0.6 + (weightedContext / contextWeight) * 0.4, -3, 3) * 100) / - 100; const placement = label(score); - const counts = { - publication: context.signals.filter((signal) => signal.sourceKind === "publication-history") - .length, - journalist: context.signals.filter((signal) => signal.sourceKind === "journalist-work").length, - comparableCoverage: context.signals.filter( - (signal) => signal.sourceKind === "comparable-coverage", - ).length, - topicContext: context.signals.filter((signal) => signal.sourceKind === "topic-context").length, - }; - const total = Math.max(1, context.signals.length); + const articleWeight = hasArticle ? 0.5 : 0; + const contextWeight = hasArticle ? 0.5 : 1; + const confidenceScore = hasArticle + ? Math.min(0.9, Math.round((article.confidenceScore * 0.7 + 0.16) * 100) / 100) + : Math.min( + 0.44, + Math.round((0.22 + Math.min(0.12, context.signals.length * 0.04)) * 100) / 100, + ); return CompassResultSchema.parse({ ...article, label: placement, displayLabel: display[placement], score, - confidenceScore: Math.min(0.86, Math.round((article.confidenceScore * 0.7 + 0.16) * 100) / 100), - confidence: - article.confidenceScore >= 0.7 ? "high" : article.confidenceScore >= 0.42 ? "medium" : "low", - explanation: `${article.explanation} Bounded publication, journalist, comparable-coverage, and topic-context evidence shifted the contextual estimate cautiously toward ${display[placement].toLocaleLowerCase("en-US")}.`, - basis: "context-assisted", + confidenceScore, + confidence: confidenceScore >= 0.75 ? "high" : confidenceScore >= 0.45 ? "medium" : "low", + explanation: hasArticle + ? `${article.explanation} Bounded publication, journalist, comparable-coverage, and topic-context evidence shifted the contextual estimate cautiously toward ${display[placement].toLocaleLowerCase("en-US")}.` + : `The available publication, journalist, comparable-coverage, and topic-context research is closest to ${display[placement].toLocaleLowerCase("en-US")} on the political spectrum.`, + basis: hasArticle ? "context-assisted" : "context-led", context, influence: { - article: 0.6, - publication: (0.4 * counts.publication) / total, - journalist: (0.4 * counts.journalist) / total, - comparableCoverage: (0.4 * counts.comparableCoverage) / total, - topicContext: (0.4 * counts.topicContext) / total, + article: articleWeight, + ...contextInfluence(context, contextWeight), }, }); } diff --git a/packages/intelligence/src/evidence/adjudication.ts b/packages/intelligence/src/evidence/adjudication.ts index 6de3e06..c0a617b 100644 --- a/packages/intelligence/src/evidence/adjudication.ts +++ b/packages/intelligence/src/evidence/adjudication.ts @@ -91,11 +91,11 @@ export function createModelEvidenceAdjudicator( system: SYSTEM_PROMPT, prompt, maxRetries: 1, - maxOutputTokens: Math.min(3_200, Math.max(1_200, input.budget.modelOutputTokens)), + maxOutputTokens: input.budget.modelOutputTokens, timeout: { - totalMs: Math.min(45_000, input.budget.totalDeadlineMs), - firstChunkMs: 30_000, - chunkMs: 15_000, + totalMs: Math.min(input.budget.specialistTimeoutMs, input.budget.totalDeadlineMs), + firstChunkMs: Math.min(30_000, input.budget.specialistTimeoutMs), + chunkMs: Math.min(15_000, input.budget.specialistTimeoutMs), }, }); const decisions = result.output?.decisions ?? []; diff --git a/packages/intelligence/src/evidence/sufficiency.ts b/packages/intelligence/src/evidence/sufficiency.ts index a603ced..e64cddc 100644 --- a/packages/intelligence/src/evidence/sufficiency.ts +++ b/packages/intelligence/src/evidence/sufficiency.ts @@ -23,7 +23,7 @@ export function evaluateSufficiency( const coverage = plan.claims.length === 0 ? 1 : covered.size / plan.claims.length; const stop = assertions.length >= budget.maxSources || - (coverage >= (budget.mode === "fast" ? 0.6 : budget.mode === "balanced" ? 0.75 : 0.9) && + (coverage >= (budget.mode === "quick" ? 0.6 : budget.mode === "balanced" ? 0.75 : 0.9) && completedMissions >= Math.min(2, plan.missions.length)) || completedMissions >= plan.missions.length; return { diff --git a/packages/intelligence/src/pipeline.ts b/packages/intelligence/src/pipeline.ts index 055051c..0ac4712 100644 --- a/packages/intelligence/src/pipeline.ts +++ b/packages/intelligence/src/pipeline.ts @@ -1,5 +1,9 @@ import { PipelineEventSchema, type PipelineEvent } from "@perspectica/contracts/events"; -import { AnalysisMetadataSchema, type ArticleDocument } from "@perspectica/contracts"; +import { + AnalysisMetadataSchema, + type ArticleDocument, + type ResearchDepth, +} from "@perspectica/contracts"; import type { ArticleIndex } from "@perspectica/contracts/article"; import type { EvidenceRetriever } from "@perspectica/contracts/evidence"; import type { AnalysisPlan, ReportSection } from "@perspectica/contracts/report"; @@ -8,7 +12,11 @@ import { INTELLIGENCE_PROMPT_VERSION, } from "@perspectica/contracts/limits"; import { buildArticleIndex } from "@perspectica/extraction"; -import { resolveAnalysisBudget, type AnalysisBudget } from "./budgets"; +import { + resolveAnalysisBudget, + type AnalysisBudget, + type LegacyAnalysisBudgetMode, +} from "./budgets"; import { createAnalysisPlan } from "./planning/lens"; import { EvidenceLedger } from "./evidence/source-ledger"; import { retryMissingSections as retryRetrieval, runRetrieval } from "./retrieval/coordinator"; @@ -25,7 +33,8 @@ export interface AnalysisInput { analysisId?: string; modelVersion?: string; reasoningEffort?: "none" | "low" | "medium" | "high" | "xhigh" | "max"; - mode?: AnalysisBudget["mode"]; + mode?: AnalysisBudget["mode"] | LegacyAnalysisBudgetMode; + depth?: ResearchDepth; signal?: AbortSignal; now?: () => Date; createId?: () => string; @@ -111,6 +120,7 @@ export async function* analyzeArticle(input: AnalysisInput): AsyncGenerator { const event = createEvent(type, analysisId, data, now); diff --git a/packages/intelligence/src/planning/lens.ts b/packages/intelligence/src/planning/lens.ts index bbdcd01..191565e 100644 --- a/packages/intelligence/src/planning/lens.ts +++ b/packages/intelligence/src/planning/lens.ts @@ -434,9 +434,9 @@ export async function createAnalysisPlan( maxOutputTokens: budget.modelOutputTokens, maxRetries: 1, timeout: { - totalMs: Math.min(45_000, budget.totalDeadlineMs), - firstChunkMs: 30_000, - chunkMs: 15_000, + totalMs: Math.min(budget.specialistTimeoutMs, budget.totalDeadlineMs), + firstChunkMs: Math.min(30_000, budget.specialistTimeoutMs), + chunkMs: Math.min(15_000, budget.specialistTimeoutMs), }, }); for await (const _ of result.partialOutputStream) options.signal?.throwIfAborted(); diff --git a/packages/intelligence/src/synthesis/perspective.ts b/packages/intelligence/src/synthesis/perspective.ts index e69f259..00044d7 100644 --- a/packages/intelligence/src/synthesis/perspective.ts +++ b/packages/intelligence/src/synthesis/perspective.ts @@ -95,7 +95,7 @@ export function synthesizePerspective( ]; }), weighting: { - articleWeight: 0.6, + articleWeight: 0.5, publicationHistory: contextAssertions.filter( (assertion) => assertion.context?.sourceKind === "publication-history", @@ -113,7 +113,7 @@ export function synthesizePerspective( (assertion) => assertion.context?.sourceKind === "topic-context", ).length / Math.max(contextAssertions.length, 1), rationale: - "Article-owned signals remain the anchor; accepted contextual signals contribute no more than forty percent.", + "Article-owned signals remain the anchor; accepted contextual signals contribute no more than fifty percent.", }, }, ); From 3d73db657206fa149da11d21beeacd53b2ae4376 Mon Sep 17 00:00:00 2001 From: austin Date: Mon, 3 Aug 2026 11:50:49 -0500 Subject: [PATCH 3/6] feat: transplant editorial compact sidepanel UX --- apps/extension/entrypoints/sidepanel/App.tsx | 781 ++++++++++++------ .../entrypoints/sidepanel/BrandHeader.tsx | 90 +- .../sidepanel/DepthControl.test.ts | 13 + .../entrypoints/sidepanel/DepthControl.tsx | 66 ++ .../entrypoints/sidepanel/SettingsScreen.tsx | 577 ++++++++----- apps/extension/entrypoints/sidepanel/api.ts | 5 + .../entrypoints/sidepanel/footnotes.test.ts | 24 + .../entrypoints/sidepanel/footnotes.tsx | 152 ++++ .../entrypoints/sidepanel/styles.css | 322 +++++++- 9 files changed, 1579 insertions(+), 451 deletions(-) create mode 100644 apps/extension/entrypoints/sidepanel/DepthControl.test.ts create mode 100644 apps/extension/entrypoints/sidepanel/DepthControl.tsx create mode 100644 apps/extension/entrypoints/sidepanel/footnotes.test.ts create mode 100644 apps/extension/entrypoints/sidepanel/footnotes.tsx diff --git a/apps/extension/entrypoints/sidepanel/App.tsx b/apps/extension/entrypoints/sidepanel/App.tsx index 6d8cfdf..5fe797c 100644 --- a/apps/extension/entrypoints/sidepanel/App.tsx +++ b/apps/extension/entrypoints/sidepanel/App.tsx @@ -16,6 +16,7 @@ import type { ReaderCopy, ReaderCitation, SourceListResult, + ResearchDepth, } from "@perspectica/contracts"; import type { ReportSection } from "@perspectica/contracts/report"; import { AnalysisProgress } from "./AnalysisProgress"; @@ -29,7 +30,26 @@ import { Section } from "./Section"; import { SettingsScreen } from "./SettingsScreen"; import type { SettingsPreferences } from "./SettingsScreen"; import { SearchSetupScreen } from "./SearchSetupScreen"; +import { ResearchDepthControl } from "./DepthControl"; import { DEFAULT_ANALYSIS_PREFERENCES } from "./preferences"; +import { + acceptedReaderCopy, + buildFootnoteLedger, + FootnoteMarker, + InlineFootnotes, + SectionFootnotes, + canonicalCitationKey, + type CitationTarget, +} from "./footnotes"; +export { + acceptedReaderCopy, + buildFootnoteLedger, + canonicalCitationKey, + FootnoteMarker, + InlineFootnotes, + SectionFootnotes, +} from "./footnotes"; +export type { CitationTarget, FootnoteEntry } from "./footnotes"; import { beginExtraction, beginTargetedRetry, @@ -37,6 +57,7 @@ import { failReport, isAnalysisActive, } from "./report-state"; +import type { ReportPhase } from "./report-state"; import { ReportStore } from "./report-store"; import type { ReportSectionKey } from "./report-store"; import { @@ -44,6 +65,7 @@ import { clearAnalysisLogs, getAnalysisLogs, getRuntimeState, + isResumableJob, streamAnalysis, type AnalysisStreamStatus, testSearchProvider, @@ -103,7 +125,13 @@ function formatDate(value: string | null): string | null { }); } -export function SourceLink({ source }: { source: ExternalSource }) { +export function SourceLink({ + source, + footnote, +}: { + source: ExternalSource; + footnote?: { scope: string; number: number; citation: CitationTarget }; +}) { const isSearchSummary = source.citationKind === "search-summary"; return (
@@ -115,6 +143,13 @@ export function SourceLink({ source }: { source: ExternalSource }) { {source.title} + {footnote ? ( + + ) : null} {source.publication ? ` · ${source.publication}` : ""} {source.publishedAt ? ` · ${new Date(source.publishedAt).toLocaleDateString()}` : ""} @@ -132,57 +167,29 @@ export function SourceLink({ source }: { source: ExternalSource }) { ); } -interface CitationTarget { - id: string; - title: string; - publication: string; - url: string; - citationKind?: "source-excerpt" | "search-summary"; -} - -function InlineCitations({ - citationIds, - citations, -}: { - citationIds: string[]; - citations: ReadonlyMap; -}) { - const accepted = citationIds.flatMap((id) => { - const citation = citations.get(id); - return citation ? [citation] : []; - }); - if (accepted.length === 0) return null; - return ( - - {accepted.map((citation) => ( - - - {citation.publication || citation.title} - - {citation.citationKind === "search-summary" ? ( - Web-search summary - ) : null} - - ))} - - ); -} - function ReaderCopyBody({ copy, citations, labels, + sectionId = "report", }: { copy: ReaderCopy; citations: ReadonlyMap; labels?: ReadonlyMap; + sectionId?: string; }) { + const accepted = acceptedReaderCopy(copy, citations); + if (!accepted) return null; + const ledger = buildFootnoteLedger( + accepted.findings.flatMap((finding) => finding.citationIds), + citations, + ); return (

- +

- {copy.findings.map((finding) => { + {accepted.findings.map((finding) => { const findingLabels = [ ...new Set( finding.citationIds.flatMap((id) => (labels?.has(id) ? [labels.get(id)!] : [])), @@ -195,7 +202,12 @@ function ReaderCopyBody({ ) : null}

- +

{finding.keySourceNote ? ( @@ -205,6 +217,7 @@ function ReaderCopyBody({
); })} + ); } @@ -218,52 +231,71 @@ function evidenceCitationMap(sources: ExternalSource[]): Map - ); +function EvidenceBody({ result, sectionId }: { result: EvidenceSectionData; sectionId: string }) { + const citations = evidenceCitationMap(result.sources); + if (result.readerCopy && acceptedReaderCopy(result.readerCopy, citations)) { + return ; } + const ledger = buildFootnoteLedger( + result.sources.map((source) => source.id), + citations, + ); return ( <>

- {result.sources.map((source) => ( - - ))} + {result.sources.map((source) => { + const number = ledger.numbers.get(source.id); + return ( + + ); + })} + ); } function JournalistBody({ result }: { result: JournalistContextResult }) { - if (result.readerCopy) { + const citations = new Map( + result.findings.map((finding) => [ + finding.id, + { + id: finding.id, + title: finding.sourceTitle, + publication: finding.publication, + url: finding.url, + publishedAt: null, + citationKind: finding.citationKind, + }, + ]), + ); + if (result.readerCopy && acceptedReaderCopy(result.readerCopy, citations)) { return ( [ - finding.id, - { - id: finding.id, - title: finding.sourceTitle, - publication: finding.publication, - url: finding.url, - citationKind: finding.citationKind, - }, - ]), - ) - } + sectionId="journalist-context" + citations={citations} /> ); } + const ledger = buildFootnoteLedger( + result.findings.map((finding) => finding.id), + citations, + ); return ( <>

@@ -279,6 +311,13 @@ function JournalistBody({ result }: { result: JournalistContextResult }) { {finding.sourceTitle} + {ledger.numbers.has(finding.id) ? ( + + ) : null} {finding.publication ? ` · ${finding.publication}` : ""} {finding.citationKind === "search-summary" ? ( Web-search summary @@ -291,35 +330,64 @@ function JournalistBody({ result }: { result: JournalistContextResult }) { ) : null} ))} + ); } function AdditionalContextBody({ result }: { result: AdditionalContextResult }) { - if (result.readerCopy) { + const citations = evidenceCitationMap(result.sources); + if (result.readerCopy && acceptedReaderCopy(result.readerCopy, citations)) { return ( - + ); } + const ledger = buildFootnoteLedger( + result.sources.map((source) => source.id), + citations, + ); return ( <>

- {result.sources.map((source) => ( - - ))} + {result.sources.map((source) => { + const number = ledger.numbers.get(source.id); + return ( + + ); + })} + ); } function SourceListBody({ result }: { result: SourceListResult }) { - if (result.sources.length === 0) { + const seen = new Set(); + const sources = result.sources.filter((source) => { + const key = canonicalCitationKey(source.url); + if (seen.has(key)) return false; + seen.add(key); + return true; + }); + if (sources.length === 0) { return

No cited works were found in the article.

; } return (
    - {result.sources.map((source) => ( + {sources.map((source) => (
  1. @@ -332,9 +400,185 @@ function SourceListBody({ result }: { result: SourceListResult }) { ); } +interface AnalyzeScreenProps { + metadata: ArticleMetadataForPreview | null; + onAnalyze: () => void; + onOpenSettings: () => void; + menuItems?: ReadonlyArray<{ label: string; onSelect: () => void }>; + researchDepth?: ResearchDepth; + onResearchDepthChange?: (depth: ResearchDepth) => void; +} + +type ArticleMetadataForPreview = { + title: string; + author: string | null; + publication: string | null; + publishedAt: string | null; + contentType: string; +}; + +export function AnalyzeScreen({ + metadata, + onAnalyze, + onOpenSettings, + menuItems, + researchDepth = "balanced" as ResearchDepth, + onResearchDepthChange, +}: AnalyzeScreenProps) { + const publishedAt = formatDate(metadata?.publishedAt ?? null); + return ( +
    + +
    +

    Article preview

    +

    + {metadata?.title ?? "Ready to read this article?"} +

    + {metadata ? ( +

    + {metadata.publication ?? "Current article"} + {metadata.author ? ` · By ${compactByline(metadata.author, metadata.publication)}` : ""} + {publishedAt ? ` · ${publishedAt}` : ""} +

    + ) : ( +

    + Perspectica previews the active article locally, then researches it only when you choose + Analyze. +

    + )} +
    + Research begins only when you select Analyze. + Your article stays local until then. +
    + + +
    +
    + ); +} + +export function DiagnosticsScreen({ onBack }: { onBack: () => void }) { + const [status, setStatus] = useState<"idle" | "copying" | "copied" | "error">("idle"); + const [clearStatus, setClearStatus] = useState<"idle" | "clearing" | "cleared" | "error">("idle"); + const [manual, setManual] = useState(null); + const textRef = useRef(null); + const copy = async () => { + setStatus("copying"); + try { + const logs = await getAnalysisLogs(); + if (await copyText(logs.text)) { + setStatus("copied"); + } else { + setManual(logs.text); + setStatus("error"); + requestAnimationFrame(() => { + textRef.current?.focus(); + textRef.current?.select(); + }); + } + } catch { + setStatus("error"); + } + }; + const clear = async () => { + if (!window.confirm("Clear saved Perspectica diagnostics from this device?")) return; + setClearStatus("clearing"); + try { + await clearAnalysisLogs(); + setManual(null); + setClearStatus("cleared"); + } catch { + setClearStatus("error"); + } + }; + return ( +
    + +
    +

    Support

    +

    + Diagnostics +

    +

    + Copy a sanitized activity log when you need help. Article text and credentials are not + included. +

    +
    + + +

    + {status === "error" ? "Automatic copy was blocked; use the selected log below." : ""} +

    + {manual ? ( +