From 90baa7887e784a2f34856aa430fbfb239b451374 Mon Sep 17 00:00:00 2001
From: kevin9327 <5299031+kevin9327@users.noreply.github.com>
Date: Wed, 9 Sep 2026 13:29:11 +0900
Subject: [PATCH] fix(audit): analyze XHTML and mixed-case HTML content types
crawlPage sends `Accept: text/html,application/xhtml+xml` but classified a
document as HTML only when its content-type contained the exact lowercase
substring `text/html`. A page served as application/xhtml+xml or TEXT/HTML was
stored as a non-HTML asset: no title, no meta description, no headings, no
links, so the crawl also never left the start URL.
---
.../site-audit-workflow-helpers.test.ts | 54 +++++++++++++++++++
.../workflows/site-audit-workflow-helpers.ts | 11 +++-
2 files changed, 63 insertions(+), 2 deletions(-)
diff --git a/src/server/workflows/site-audit-workflow-helpers.test.ts b/src/server/workflows/site-audit-workflow-helpers.test.ts
index a3c33ebb2..5b37f56f2 100644
--- a/src/server/workflows/site-audit-workflow-helpers.test.ts
+++ b/src/server/workflows/site-audit-workflow-helpers.test.ts
@@ -307,3 +307,57 @@ describe("crawlPage", () => {
expect(result.status, result.stderr).toBe(0);
}, 25_000);
});
+
+const LINKED_PAGE_HTML = `
A page title
+Hi
Body text.
next`;
+
+function serveAs(contentType: string) {
+ vi.stubGlobal("fetch", () =>
+ Promise.resolve(
+ new Response(LINKED_PAGE_HTML, {
+ status: 200,
+ headers: { "content-type": contentType },
+ }),
+ ),
+ );
+}
+
+function summarize(page: Awaited>) {
+ return {
+ isHtml: page?.isHtml,
+ title: page?.title,
+ linkCount: page?.links.length,
+ };
+}
+
+describe("crawlPage content-type classification", () => {
+ // The crawler sends `Accept: text/html,application/xhtml+xml`, and media
+ // types are case-insensitive, so both spellings must be analyzed.
+ it.each([
+ "text/html; charset=utf-8",
+ "application/xhtml+xml; charset=utf-8",
+ "TEXT/HTML",
+ ])("analyzes a page served as %s", async (contentType) => {
+ serveAs(contentType);
+
+ const page = await crawl();
+
+ expect(summarize(page)).toEqual({
+ isHtml: true,
+ title: "A page title",
+ linkCount: 1,
+ });
+ });
+
+ it("records a non-HTML document without analyzing it", async () => {
+ serveAs("application/pdf");
+
+ const page = await crawl();
+
+ expect(summarize(page)).toEqual({
+ isHtml: false,
+ title: "",
+ linkCount: 0,
+ });
+ });
+});
diff --git a/src/server/workflows/site-audit-workflow-helpers.ts b/src/server/workflows/site-audit-workflow-helpers.ts
index 7d207aea1..138e283c5 100644
--- a/src/server/workflows/site-audit-workflow-helpers.ts
+++ b/src/server/workflows/site-audit-workflow-helpers.ts
@@ -138,8 +138,15 @@ export async function crawlPage(
});
}
- const contentType = response.headers.get("content-type") ?? "";
- const isHtml = contentType.includes("text/html");
+ // Media types are case-insensitive, and the crawl Accept header asks for
+ // application/xhtml+xml as well — a page served as either is a document
+ // the analyzer can read.
+ const contentType = (
+ response.headers.get("content-type") ?? ""
+ ).toLowerCase();
+ const isHtml =
+ contentType.includes("text/html") ||
+ contentType.includes("application/xhtml+xml");
// Cap what we read: the first 1 MiB still contains the SEO metadata and
// navigation needed by the audit in normal documents.
const body = isHtml ? await readTextUpTo(response, MAX_HTML_BYTES) : "";