From 90baa7887e784a2f34856aa430fbfb239b451374 Mon Sep 17 00:00:00 2001 From: kevin9327 <5299031+kevin9327@users.noreply.github.com> Date: Wed, 9 Sep 2026 13:29:11 +0900 Subject: [PATCH] fix(audit): analyze XHTML and mixed-case HTML content types crawlPage sends `Accept: text/html,application/xhtml+xml` but classified a document as HTML only when its content-type contained the exact lowercase substring `text/html`. A page served as application/xhtml+xml or TEXT/HTML was stored as a non-HTML asset: no title, no meta description, no headings, no links, so the crawl also never left the start URL. --- .../site-audit-workflow-helpers.test.ts | 54 +++++++++++++++++++ .../workflows/site-audit-workflow-helpers.ts | 11 +++- 2 files changed, 63 insertions(+), 2 deletions(-) diff --git a/src/server/workflows/site-audit-workflow-helpers.test.ts b/src/server/workflows/site-audit-workflow-helpers.test.ts index a3c33ebb2..5b37f56f2 100644 --- a/src/server/workflows/site-audit-workflow-helpers.test.ts +++ b/src/server/workflows/site-audit-workflow-helpers.test.ts @@ -307,3 +307,57 @@ describe("crawlPage", () => { expect(result.status, result.stderr).toBe(0); }, 25_000); }); + +const LINKED_PAGE_HTML = `A page title +

Hi

Body text.

next`; + +function serveAs(contentType: string) { + vi.stubGlobal("fetch", () => + Promise.resolve( + new Response(LINKED_PAGE_HTML, { + status: 200, + headers: { "content-type": contentType }, + }), + ), + ); +} + +function summarize(page: Awaited>) { + return { + isHtml: page?.isHtml, + title: page?.title, + linkCount: page?.links.length, + }; +} + +describe("crawlPage content-type classification", () => { + // The crawler sends `Accept: text/html,application/xhtml+xml`, and media + // types are case-insensitive, so both spellings must be analyzed. + it.each([ + "text/html; charset=utf-8", + "application/xhtml+xml; charset=utf-8", + "TEXT/HTML", + ])("analyzes a page served as %s", async (contentType) => { + serveAs(contentType); + + const page = await crawl(); + + expect(summarize(page)).toEqual({ + isHtml: true, + title: "A page title", + linkCount: 1, + }); + }); + + it("records a non-HTML document without analyzing it", async () => { + serveAs("application/pdf"); + + const page = await crawl(); + + expect(summarize(page)).toEqual({ + isHtml: false, + title: "", + linkCount: 0, + }); + }); +}); diff --git a/src/server/workflows/site-audit-workflow-helpers.ts b/src/server/workflows/site-audit-workflow-helpers.ts index 7d207aea1..138e283c5 100644 --- a/src/server/workflows/site-audit-workflow-helpers.ts +++ b/src/server/workflows/site-audit-workflow-helpers.ts @@ -138,8 +138,15 @@ export async function crawlPage( }); } - const contentType = response.headers.get("content-type") ?? ""; - const isHtml = contentType.includes("text/html"); + // Media types are case-insensitive, and the crawl Accept header asks for + // application/xhtml+xml as well — a page served as either is a document + // the analyzer can read. + const contentType = ( + response.headers.get("content-type") ?? "" + ).toLowerCase(); + const isHtml = + contentType.includes("text/html") || + contentType.includes("application/xhtml+xml"); // Cap what we read: the first 1 MiB still contains the SEO metadata and // navigation needed by the audit in normal documents. const body = isHtml ? await readTextUpTo(response, MAX_HTML_BYTES) : "";