Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions packages/contracts/src/evidence.ts
Original file line number Diff line number Diff line change
Expand Up @@ -72,13 +72,13 @@ export type EvidenceContextSignal = z.infer<typeof EvidenceContextSignalSchema>;
export const EvidenceAdjudicationSchema = z.object({
candidateId: text.max(160),
missionId: text.max(160),
claimId: text.max(160).nullable(),
claimId: text.max(160).nullable().default(null),
relationship: EvidenceRelationshipV2Schema,
statement: text.max(1_200),
excerpt: text.max(4_000).nullable(),
excerpt: text.max(4_000).nullable().default(null),
confidence: z.number().min(0).max(1),
relevance: z.number().min(0).max(1),
context: EvidenceContextSignalSchema.nullable(),
context: EvidenceContextSignalSchema.nullable().default(null),
});
export type EvidenceAdjudication = z.infer<typeof EvidenceAdjudicationSchema>;

Expand Down
69 changes: 69 additions & 0 deletions packages/extraction/src/article-index.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
import { ArticleDocumentSchema } from "@perspectica/contracts";
import { describe, expect, it } from "vitest";
import { buildArticleIndex } from "./article-index";

describe("ArticleIndex link classification", () => {
it("keeps sharing, embedded-audio, and app links out of editorial sources", () => {
const article = ArticleDocumentSchema.parse({
fingerprint: "article-links",
canonicalUrl: "https://www.foxnews.com/story",
title: "A story",
author: "Reporter",
publication: "Fox News",
publishedAt: null,
language: "en",
contentType: "news",
paragraphs: [
{
id: "p1",
index: 0,
kind: "paragraph",
speaker: null,
text: "The report described the event and cited an outside record.",
},
],
links: [
{
id: "flipboard",
label: "Flipboard",
url: "https://share.flipboard.com/bookmarklet/popout?url=https%3A%2F%2Fwww.foxnews.com%2Fstory",
paragraphId: null,
},
{
id: "beyondwords",
label: "beyondwords.io",
url: "https://beyondwords.io/player/example",
paragraphId: null,
},
{
id: "app",
label: "CLICK HERE TO DOWNLOAD THE FOX NEWS APP",
url: "https://foxnews.onelink.me/example",
paragraphId: null,
},
{
id: "primary",
label: "Public record",
url: "https://example.gov/record",
paragraphId: "p1",
},
],
extraction: {
extractorVersion: "test",
extractedAt: "2026-08-03T00:00:00.000Z",
wordCount: 10,
articleStatus: "article",
contentChars: 62,
contentTruncated: false,
},
});

const index = buildArticleIndex(article);
expect(Object.fromEntries(index.links.map((link) => [link.id, link.classification]))).toEqual({
flipboard: "social",
beyondwords: "promotional",
app: "promotional",
primary: "likely-primary",
});
});
});
11 changes: 10 additions & 1 deletion packages/extraction/src/article-index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,16 @@ const SOCIAL_HOSTS = new Set([
"twitter.com",
"x.com",
"youtube.com",
"flipboard.com",
"sharethis.com",
"addtoany.com",
"pinterest.com",
"t.me",
"telegram.me",
]);
const NON_EDITORIAL_HOSTS = new Set(["beyondwords.io"]);
const PROMOTIONAL_PATTERN =
/\b(?:subscribe|newsletter|advert(?:isement)?|sponsor|shop|store|download|app|onelink|affiliate)\b/i;
const NAVIGATION_PATH =
/(?:^|\/)(?:search|tag|topic|category|author|login|account|subscribe)(?:\/|$)/i;
const COMMON_CAPITALIZED = new Set([
Expand Down Expand Up @@ -195,7 +204,7 @@ function linkClassification(
if ([...SOCIAL_HOSTS].some((domain) => linkHost === domain || linkHost.endsWith(`.${domain}`))) {
return "social";
}
if (/\b(?:subscribe|newsletter|advert|sponsor|shop|store)\b/i.test(`${link.label} ${link.url}`)) {
if (NON_EDITORIAL_HOSTS.has(linkHost) || PROMOTIONAL_PATTERN.test(`${link.label} ${link.url}`)) {

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win

Match non-editorial host subdomains.

NON_EDITORIAL_HOSTS.has(linkHost) only matches beyondwords.io. A link such as https://www.beyondwords.io/... classifies as external and can reach Works Cited through the external-source projection. Match the apex domain and its subdomains, as SOCIAL_HOSTS already does.

Proposed fix
-  if (NON_EDITORIAL_HOSTS.has(linkHost) || PROMOTIONAL_PATTERN.test(`${link.label} ${link.url}`)) {
+  if (
+    [...NON_EDITORIAL_HOSTS].some(
+      (domain) => linkHost === domain || linkHost.endsWith(`.${domain}`),
+    ) ||
+    PROMOTIONAL_PATTERN.test(`${link.label} ${link.url}`)
+  ) {
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
if (NON_EDITORIAL_HOSTS.has(linkHost) || PROMOTIONAL_PATTERN.test(`${link.label} ${link.url}`)) {
if (
[...NON_EDITORIAL_HOSTS].some(
(domain) => linkHost === domain || linkHost.endsWith(`.${domain}`),
) ||
PROMOTIONAL_PATTERN.test(`${link.label} ${link.url}`)
) {
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@packages/extraction/src/article-index.ts` at line 207, Update the
non-editorial host check in the link classification logic around
NON_EDITORIAL_HOSTS and PROMOTIONAL_PATTERN so it matches both each configured
apex domain and its subdomains, consistent with the SOCIAL_HOSTS matching
behavior. Ensure hosts such as www.beyondwords.io are treated as non-editorial
before they can enter the external-source projection.

return "promotional";
}
if (
Expand Down
18 changes: 15 additions & 3 deletions packages/intelligence/src/evidence/adjudication.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ import type { AnalysisPlan } from "@perspectica/contracts/report";
import type { AnalysisBudget } from "../budgets";

const AdjudicationOutputSchema = z.object({
decisions: z.array(EvidenceAdjudicationSchema).max(96),
decisions: z.array(EvidenceAdjudicationSchema).max(8),

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🗄️ Data Integrity & Integration | 🟠 Major | ⚡ Quick win

Enforce the eight-decision limit on the public result.

Line 13 limits each adjudicator.adjudicate response. Lines 115-126 append every batch response and return the aggregate without a cap. Multiple batches can therefore return more than eight decisions, which conflicts with the bounded-output objective.

Apply one aggregate limit and a deterministic selection rule. Add a regression test with multiple batches.

Suggested aggregate limit
+const MAX_DECISIONS = 8;
+
 const AdjudicationOutputSchema = z.object({
-  decisions: z.array(EvidenceAdjudicationSchema).max(8),
+  decisions: z.array(EvidenceAdjudicationSchema).max(MAX_DECISIONS),
 });
...
-  return decisions;
+  return decisions.slice(0, MAX_DECISIONS);

Also applies to: 115-126

🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@packages/intelligence/src/evidence/adjudication.ts` at line 13, Update the
aggregation logic in adjudicator.adjudicate so the public result is capped at
eight decisions, not just each individual batch response. After combining batch
results, apply a deterministic selection rule (preserving the established
decision order) before returning the aggregate, and add a regression test
covering multiple batches that would otherwise exceed the limit.

});

export interface EvidenceAdjudicationInput {
Expand Down Expand Up @@ -51,7 +51,7 @@ function buildPrompt(input: EvidenceAdjudicationInput): string {
.slice(0, Math.max(input.budget.maxSources * 4, 24))
.map(
(candidate) =>
`CANDIDATE ${candidate.id} mission=${candidate.missionId ?? "global-search"} url=${candidate.sourceUrl} title=${compact(candidate.title, 300)} kind=${candidate.contentKind} sourceType=${candidate.sourceType}\nCONTENT: ${compact(candidate.content, 4_000)}\nDISCOVERY: ${compact(candidate.discoveryContext ?? "", 2_000)}`,
`CANDIDATE ${candidate.id} mission=${candidate.missionId ?? "global-search"} url=${candidate.sourceUrl} title=${compact(candidate.title, 300)} kind=${candidate.contentKind} sourceType=${candidate.sourceType}\nCONTENT: ${compact(candidate.content, 2_400)}\nDISCOVERY: ${compact(candidate.discoveryContext ?? "", 600)}`,
)
.join("\n\n");
return [
Expand All @@ -73,6 +73,7 @@ const SYSTEM_PROMPT = [
"Supports, contradicts, and qualifies require an exact planned claim and a clear source-content anchor.",
"Use context only for journalist-work, publication-history, comparable-coverage, or topic-context that is explicitly present in the candidate.",
"Return no decision for irrelevant, ambiguous, self-referential, or discovery-only candidates. Do not write generic discovery prose such as 'surfaced a relevant source'.",
"Return at most one short decision per candidate. Keep source excerpts contiguous and as short as possible (preferably under 600 characters).",
"Every candidateId and missionId must be copied exactly from the input. Do not invent IDs.",
].join(" ");

Expand Down Expand Up @@ -111,5 +112,16 @@ export async function adjudicateEvidence(
input: EvidenceAdjudicationInput & { adjudicator?: EvidenceAdjudicator },
): Promise<EvidenceAdjudication[]> {
if (!input.adjudicator || input.candidates.length === 0) return [];
return input.adjudicator.adjudicate(input);
const maxCandidatesPerCall = 6;
const decisions: EvidenceAdjudication[] = [];
for (let offset = 0; offset < input.candidates.length; offset += maxCandidatesPerCall) {
const batch = input.candidates.slice(offset, offset + maxCandidatesPerCall);
decisions.push(
Comment on lines +115 to +119
...(await input.adjudicator.adjudicate({
...input,
candidates: batch,
})),
);
}
return decisions;
Comment on lines +115 to +126

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🩺 Stability & Availability | 🟠 Major | ⚡ Quick win

🧩 Analysis chain

🏁 Script executed:

#!/usr/bin/env bash
set -euo pipefail

rg -n -C 8 \
  -e '\btotalDeadlineMs\b' \
  -e '\bAbortController\b' \
  -e '\bdeadline\b' \
  -e '\badjudicateEvidence\b' \
  packages/intelligence packages/contracts --glob '*.ts'

Repository: drperky20/perspectica

Length of output: 18822


🏁 Script executed:

#!/usr/bin/env bash
set -euo pipefail

echo "== files =="
git ls-files 'packages/intelligence/src/**/*.ts' | sed -n '1,120p'

echo
echo "== adjudication.ts outline =="
ast-grep outline packages/intelligence/src/evidence/adjudication.ts --view expanded || true

echo
echo "== adjudication.ts 1-150 =="
cat -n packages/intelligence/src/evidence/adjudication.ts | sed -n '1,150p'

echo
echo "== budgets.ts =="
cat -n packages/intelligence/src/budgets.ts | sed -n '1,120p'

echo
echo "== exact pipeline/adjudicator usages =="
rg -n -C 4 'adjudicateEvidence|budget: resolveAnalysisBudget|resolveAnalysisBudget\(' packages/intelligence/src --glob '*.ts'

Repository: drperky20/perspectica

Length of output: 15415


Carry a remaining deadline for each adjudication batch.

adjudicateEvidence creates 6-candidate batches and passes the same input.budget into each batch. createModelEvidenceAdjudicator uses input.budget.totalDeadlineMs for every generateText timeout, so multiple batches plus one retry each can exceed the pipeline deadline. Track a single absolute deadline and clamp each batch timeout to the remaining time before passing it to adjudicator.adjudicate.

🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@packages/intelligence/src/evidence/adjudication.ts` around lines 115 - 126,
Update adjudicateEvidence to compute one absolute deadline from
input.budget.totalDeadlineMs, derive the remaining time before each batch, and
pass a budget with totalDeadlineMs clamped to that remaining duration into
input.adjudicator.adjudicate. Preserve the existing six-candidate batching and
decision aggregation while ensuring every batch and retry shares the pipeline
deadline.

}
22 changes: 22 additions & 0 deletions packages/intelligence/src/pipeline.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -175,6 +175,28 @@ describe("V2 intelligence pipeline", () => {
expect(events.at(-1)?.type).toBe("analysis.cancelled");
});

it("turns an empty provider error into a valid terminal failure event", async () => {
const events = [];
const emptyErrorAdjudicator: EvidenceAdjudicator = {
async adjudicate() {
throw new Error("");
},
};
for await (const event of analyzeArticle({
article: article(),
retriever: retriever(),
adjudicator: emptyErrorAdjudicator,
mode: "fast",
reasoningEffort: "low",
}))
events.push(event);

expect(events.at(-1)?.type).toBe("analysis.failed");
const failure = events.at(-1);
if (failure?.type === "analysis.failed")
expect(failure.data.message).toBe("The analysis pipeline failed.");
});

it("retries only requested lanes from the existing artifacts", async () => {
let artifacts: AnalysisArtifacts | undefined;
const emptyRetriever = {
Expand Down
9 changes: 7 additions & 2 deletions packages/intelligence/src/pipeline.ts
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,11 @@ function isAbortError(error: unknown, signal?: AbortSignal): boolean {
);
}

function pipelineErrorMessage(error: unknown, fallback: string): string {
const message = error instanceof Error ? error.message : typeof error === "string" ? error : "";
return (message.trim() || fallback).slice(0, 1_000);
}

export async function* analyzeArticle(input: AnalysisInput): AsyncGenerator<PipelineEvent> {
const now = input.now ?? (() => new Date());
const analysisId = input.analysisId ?? randomId("analysis");
Expand Down Expand Up @@ -279,7 +284,7 @@ export async function* analyzeArticle(input: AnalysisInput): AsyncGenerator<Pipe
yield emit("analysis.cancelled", { message: "Analysis cancelled." });
return;
}
const message = error instanceof Error ? error.message : "The analysis pipeline failed.";
const message = pipelineErrorMessage(error, "The analysis pipeline failed.");
yield emit("phase.changed", { phase: "failed", message });
yield emit("analysis.failed", { message, retryable: true });
}
Expand Down Expand Up @@ -414,7 +419,7 @@ export async function* retryArticleSections(
yield emit("analysis.cancelled", { message: "Retry cancelled." });
return;
}
const message = error instanceof Error ? error.message : "The targeted retry failed.";
const message = pipelineErrorMessage(error, "The targeted retry failed.");
yield emit("phase.changed", { phase: "failed", message });
yield emit("analysis.failed", { message, retryable: true });
}
Expand Down
2 changes: 1 addition & 1 deletion packages/intelligence/src/report/projector.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,6 @@ describe("report source projection", () => {
],
} as ArticleIndex);

expect(result.sources.map((source) => source.id)).toEqual(["external", "primary", "same"]);
expect(result.sources.map((source) => source.id)).toEqual(["external", "primary"]);
});
});
5 changes: 1 addition & 4 deletions packages/intelligence/src/report/projector.ts
Original file line number Diff line number Diff line change
Expand Up @@ -65,10 +65,7 @@ export function projectBias(plan: AnalysisPlan): BiasResult {

export function projectSourceList(article: ArticleIndex): SourceListResult {
const links = article.links
.filter((link) =>
["external", "likely-primary", "same-publication"].includes(link.classification),
)
.filter((link) => link.classification !== "same-publication" || Boolean(link.paragraphId))
.filter((link) => ["external", "likely-primary"].includes(link.classification))
.map((link) => ({
id: link.id,
label: link.label,
Expand Down
Loading