From bb427e0e848786e62a1fc481593f9ce4a5d918fa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 06:33:34 +0200 Subject: [PATCH 1/7] feat: add ui-helper (windows/displays/permissions) and captureScreen API - src/native/ui-helper.swift: read-only CoreGraphics introspection helper (CGGetOnlineDisplayList so asleep displays stay visible; no AppKit) - scripts: ui-helper joins vision-helper/pdf-helper in the prebuilt pipeline - src/ui.ts: listWindows / listDisplays / checkPermissions / captureScreen (screencapture-based, returns paths + geometry, never image bytes) - README: UI API section with privacy invariant (eyes, not hands) Co-Authored-By: Claude Fable 5 --- README.md | 28 ++++- scripts/build-native-cross.js | 2 +- scripts/install-native.js | 2 +- src/index.ts | 11 ++ src/native/ui-helper.swift | 111 ++++++++++++++++ src/ui.ts | 230 ++++++++++++++++++++++++++++++++++ 6 files changed, 381 insertions(+), 3 deletions(-) create mode 100644 src/native/ui-helper.swift create mode 100644 src/ui.ts diff --git a/README.md b/README.md index 08f48ff..ece4458 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ Uses macOS's built-in [Vision framework](https://developer.apple.com/documentati npm install macos-vision ``` -The native Swift binaries (`vision-helper`, `pdf-helper`) are downloaded as prebuilt artifacts from the matching GitHub Release (signed by SHA-256). If the download fails (no network, custom registry, unpublished version), the postinstall falls back to compiling locally with `swiftc` — that's the only path that needs Xcode Command Line Tools. Set `MACOS_VISION_SKIP_DOWNLOAD=1` to force local compilation. +The native Swift binaries (`vision-helper`, `pdf-helper`, `ui-helper`) are downloaded as prebuilt artifacts from the matching GitHub Release (signed by SHA-256). If the download fails (no network, custom registry, unpublished version), the postinstall falls back to compiling locally with `swiftc` — that's the only path that needs Xcode Command Line Tools. Set `MACOS_VISION_SKIP_DOWNLOAD=1` to force local compilation. ## What you get @@ -28,6 +28,7 @@ The native Swift binaries (`vision-helper`, `pdf-helper`) are downloaded as preb | Image classification | Apple Vision | offline | | Layout inference (lines, paragraphs, reading order) | heuristic in TypeScript | offline | | PDF rasterization | PDFKit (`pdf-helper`) | offline | +| Screen capture + window / display / permission introspection | `screencapture` + CoreGraphics (`ui-helper`) | offline | | **Image / PDF → Markdown** | Apple Vision OCR + local LLM via Ollama | local LLM call | --- @@ -144,6 +145,31 @@ for (const block of layout) { --- +## API — UI (screen capture, windows, permissions) + +Read-only introspection of the desktop plus PNG captures, meant as the "eyes" of UI-testing agents. Requires **Screen Recording** permission for the host process (System Settings → Privacy & Security → Screen Recording). + +```js +import { listWindows, listDisplays, checkPermissions, captureScreen } from 'macos-vision'; + +const perms = await checkPermissions(); // { screenRecording, accessibility } +const displays = await listDisplays(); // bounds in screen points + backing scale +const windows = await listWindows(); // on-screen app windows, front-to-back + +// Capture the frontmost Safari window (app name: exact or case-insensitive prefix) +const shot = await captureScreen({ app: 'Safari' }); +// { path, pixelWidth, pixelHeight, frame: { x, y, w, h }, scale, capturedAt, target } + +// Other targets: { windowId }, { rect: { x, y, w, h } }, { displayId } (default: main display) +const region = await captureScreen({ rect: { x: 0, y: 0, w: 800, h: 600 }, outPath: './region.png' }); +``` + +All coordinates are **global screen points with a top-left origin** — the same space `CGEvent` clicks use — so `frame` + an OCR block's normalized bbox maps straight to a click point. + +**Privacy invariant:** these functions return paths, geometry, and text — never image bytes. Captures are written to disk (`$TMPDIR/macos-vision/` unless `outPath` is given) and the caller owns cleanup. The library never synthesizes input: *eyes, not hands*. + +--- + ## API — Markdown pipeline (VisionScribe) `VisionScribe` converts an image or PDF to Markdown by combining Apple Vision OCR with a local LLM (via Ollama). The LLM never sees the image — it only formats text that Vision already extracted. This keeps image processing local and reduces the risk of vision-model hallucinations, but Markdown reconstruction is still best-effort and depends on the local model and document complexity. diff --git a/scripts/build-native-cross.js b/scripts/build-native-cross.js index f5aab7f..61309d4 100644 --- a/scripts/build-native-cross.js +++ b/scripts/build-native-cross.js @@ -11,7 +11,7 @@ const TARGETS = [ { arch: 'arm64', swift: 'arm64-apple-macos12' }, { arch: 'x64', swift: 'x86_64-apple-macos12' }, ]; -const HELPERS = ['vision-helper', 'pdf-helper']; +const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; for (const { arch, swift } of TARGETS) { const outDir = path.join(root, 'bin', `darwin-${arch}`); diff --git a/scripts/install-native.js b/scripts/install-native.js index 59c8633..0ab610c 100644 --- a/scripts/install-native.js +++ b/scripts/install-native.js @@ -17,7 +17,7 @@ const root = path.resolve(__dirname, '..'); const pkg = JSON.parse(readFileSync(path.join(root, 'package.json'), 'utf8')); const binDir = path.join(root, 'bin'); -const HELPERS = ['vision-helper', 'pdf-helper']; +const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; // 0. Skip if all binaries already exist (cached install) if (HELPERS.every((h) => existsSync(path.join(binDir, h)))) { diff --git a/src/index.ts b/src/index.ts index ea18b8f..a6bd3e0 100644 --- a/src/index.ts +++ b/src/index.ts @@ -338,3 +338,14 @@ export { inferLayout, sortBlocksByReadingOrder } from './layout.js'; // ─── Markdown pipeline (VisionScribe) ────────────────────────────────────────── export { VisionScribe, OllamaUnavailableError } from './markdown/index.js'; export type { VisionScribeOptions, ParagraphGroup } from './markdown/index.js'; + +// ─── UI: screen capture, windows, displays, permissions ──────────────────────── +export { listWindows, listDisplays, checkPermissions, captureScreen } from './ui.js'; +export type { + WindowInfo, + DisplayInfo, + PermissionsInfo, + ScreenFrame, + CaptureResult, + CaptureOptions, +} from './ui.js'; diff --git a/src/native/ui-helper.swift b/src/native/ui-helper.swift new file mode 100644 index 0000000..1e75792 --- /dev/null +++ b/src/native/ui-helper.swift @@ -0,0 +1,111 @@ +import ApplicationServices +import CoreGraphics +import Foundation + +// ui-helper — read-only screen/window/permission introspection for macos-vision-mcp. +// Emits JSON on stdout. Never captures pixels itself (capture goes through +// /usr/sbin/screencapture); never synthesizes input. Eyes, not hands. + +// ─── Result structs ────────────────────────────────────────────────────────── + +struct WindowInfo: Codable { + let windowId: Int + let app: String + let pid: Int + let title: String + // Global screen points, top-left origin (matches CGEvent click coordinates). + let x: Double; let y: Double; let w: Double; let h: Double + let layer: Int + let isOnScreen: Bool +} + +struct DisplayInfo: Codable { + let displayId: Int + let isMain: Bool + // Global screen points, top-left origin. + let x: Double; let y: Double; let w: Double; let h: Double + let scale: Double +} + +struct PermissionsInfo: Codable { + let screenRecording: Bool + let accessibility: Bool +} + +func encodeJSON(_ value: T) -> String { + guard let data = try? JSONEncoder().encode(value), + let str = String(data: data, encoding: .utf8) else { return "[]" } + return str +} + +// ─── Modes ─────────────────────────────────────────────────────────────────── + +let args = CommandLine.arguments + +if args.contains("--permissions") { + let info = PermissionsInfo( + screenRecording: CGPreflightScreenCaptureAccess(), + accessibility: AXIsProcessTrusted() + ) + print(encodeJSON(info)) + exit(0) +} + +if args.contains("--displays") { + var count: UInt32 = 0 + var ids = [CGDirectDisplayID](repeating: 0, count: 16) + // Online (not Active) list: an asleep display is still online, and captures/ + // window queries keep working while it sleeps — report it. + CGGetOnlineDisplayList(16, &ids, &count) + var results: [DisplayInfo] = [] + for i in 0.. 0 { + scale = Double(mode.pixelWidth) / Double(mode.width) + } + results.append(DisplayInfo( + displayId: Int(id), + isMain: CGDisplayIsMain(id) != 0, + x: bounds.origin.x, y: bounds.origin.y, + w: bounds.size.width, h: bounds.size.height, + scale: scale + )) + } + print(encodeJSON(results)) + exit(0) +} + +if args.contains("--windows") { + let options: CGWindowListOption = [.optionOnScreenOnly, .excludeDesktopElements] + guard let list = CGWindowListCopyWindowInfo(options, kCGNullWindowID) as? [[String: Any]] else { + print("[]") + exit(0) + } + var results: [WindowInfo] = [] + for w in list { + guard let boundsDict = w[kCGWindowBounds as String] as? [String: Double] else { continue } + let layer = w[kCGWindowLayer as String] as? Int ?? 0 + // Layer 0 = normal app windows; skip menu bar, dock, overlays unless --all. + if layer != 0 && !args.contains("--all") { continue } + let width = boundsDict["Width"] ?? 0 + let height = boundsDict["Height"] ?? 0 + if width < 40 || height < 40 { continue } // status items, tooltips + results.append(WindowInfo( + windowId: w[kCGWindowNumber as String] as? Int ?? 0, + app: w[kCGWindowOwnerName as String] as? String ?? "", + pid: w[kCGWindowOwnerPID as String] as? Int ?? 0, + title: w[kCGWindowName as String] as? String ?? "", + x: boundsDict["X"] ?? 0, y: boundsDict["Y"] ?? 0, + w: width, h: height, + layer: layer, + isOnScreen: (w[kCGWindowIsOnscreen as String] as? Bool) ?? true + )) + } + print(encodeJSON(results)) + exit(0) +} + +print("Usage: ui-helper [--windows [--all] | --displays | --permissions]") +exit(1) diff --git a/src/ui.ts b/src/ui.ts new file mode 100644 index 0000000..2ab3d1e --- /dev/null +++ b/src/ui.ts @@ -0,0 +1,230 @@ +// UI layer: screen capture + window/display/permission introspection. +// +// Privacy invariant: no function in this module ever returns image bytes. +// Captures land on disk; only paths, geometry, and metadata flow back. +// The library never synthesizes input — eyes, not hands. + +import { execFile } from 'child_process'; +import { promisify } from 'util'; +import { existsSync, mkdirSync } from 'fs'; +import { open } from 'fs/promises'; +import { tmpdir } from 'os'; +import { resolve, dirname, join } from 'path'; +import { fileURLToPath } from 'url'; + +const execFileAsync = promisify(execFile); +const __dirname = dirname(fileURLToPath(import.meta.url)); +const UI_BIN_PATH = resolve(__dirname, '../bin/ui-helper'); +const UI_HELPER_TIMEOUT_MS = 15_000; +const SCREENCAPTURE_TIMEOUT_MS = 15_000; + +// ─── Types ─────────────────────────────────────────────────────────────────── + +export interface WindowInfo { + /** CGWindowID — stable for the window's lifetime */ + windowId: number; + /** Owning application name, e.g. 'Safari' */ + app: string; + pid: number; + /** Window title (may be empty) */ + title: string; + /** Global screen points, top-left origin (CGEvent click space) */ + x: number; + y: number; + w: number; + h: number; + /** 0 = normal app window; menu bar / dock / overlays only with `listWindows(true)` */ + layer: number; + isOnScreen: boolean; +} + +export interface DisplayInfo { + displayId: number; + isMain: boolean; + /** Global screen points, top-left origin */ + x: number; + y: number; + w: number; + h: number; + /** Backing scale factor (2 on Retina) */ + scale: number; +} + +export interface PermissionsInfo { + screenRecording: boolean; + accessibility: boolean; +} + +/** Rectangle in global screen points, top-left origin (CGEvent click space). */ +export interface ScreenFrame { + x: number; + y: number; + w: number; + h: number; +} + +export interface CaptureResult { + /** Absolute path to the PNG on disk */ + path: string; + /** Pixel dimensions of the PNG. */ + pixelWidth: number; + pixelHeight: number; + /** Screen region the image covers, in global screen points (top-left origin). */ + frame: ScreenFrame; + /** pixelWidth / frame.w — ≈2 on Retina */ + scale: number; + /** ISO timestamp */ + capturedAt: string; + /** Human-readable description of what was captured */ + target: string; +} + +export interface CaptureOptions { + windowId?: number; + /** App name (exact or case-insensitive prefix) — its frontmost window wins. */ + app?: string; + /** Region in global screen points, top-left origin. */ + rect?: ScreenFrame; + displayId?: number; + /** Where to write the PNG. Default: `$TMPDIR/macos-vision/capture-.png` */ + outPath?: string; +} + +// ─── ui-helper ─────────────────────────────────────────────────────────────── + +async function runUi(...args: string[]): Promise { + const { stdout } = await execFileAsync(UI_BIN_PATH, args, { timeout: UI_HELPER_TIMEOUT_MS }); + return JSON.parse(stdout) as T; +} + +/** On-screen windows, front-to-back. Pass `true` to include menu bar, dock and overlays. */ +export function listWindows(includeAll = false): Promise { + return runUi('--windows', ...(includeAll ? ['--all'] : [])); +} + +/** Online displays (including asleep ones) with bounds and backing scale. */ +export function listDisplays(): Promise { + return runUi('--displays'); +} + +/** Screen Recording / Accessibility status for the current host process. */ +export function checkPermissions(): Promise { + return runUi('--permissions'); +} + +// ─── Capture ───────────────────────────────────────────────────────────────── + +function capturePath(outPath?: string): string { + if (outPath) return resolve(outPath); + const dir = join(tmpdir(), 'macos-vision'); + mkdirSync(dir, { recursive: true }); + return join(dir, `capture-${Date.now()}.png`); +} + +/** Width/height straight from the PNG IHDR header — no subprocess. */ +async function pngPixelSize(path: string): Promise<{ w: number; h: number }> { + const fh = await open(path, 'r'); + try { + const buf = Buffer.alloc(8); + await fh.read(buf, 0, 8, 16); // IHDR starts at byte 8; width/height at 16/20 + return { w: buf.readUInt32BE(0), h: buf.readUInt32BE(4) }; + } finally { + await fh.close(); + } +} + +async function resolveWindow(opts: CaptureOptions): Promise { + const windows = await listWindows(); + if (opts.windowId != null) { + const win = windows.find((w) => w.windowId === opts.windowId); + if (!win) throw new Error(`Window ${opts.windowId} not found (use listWindows())`); + return win; + } + if (opts.app) { + const q = opts.app.toLowerCase(); + // CGWindowList is front-to-back — first match is the frontmost window of that app. + // App-name matching only: title search would silently capture the wrong app. + const win = windows.find((w) => w.app.toLowerCase() === q || w.app.toLowerCase().startsWith(q)); + if (!win) { + const apps = [...new Set(windows.map((w) => w.app))].join(', '); + throw new Error(`No on-screen window matches app "${opts.app}". Visible apps: ${apps}`); + } + return win; + } + throw new Error('window target requires windowId or app'); +} + +// Screen Recording can only change for this process via an app restart, so one +// successful preflight is valid for the process lifetime. +let screenRecordingOk = false; + +/** + * Captures a window (`windowId` / `app`), a region (`rect`) or a display + * (`displayId`, default: main) to a PNG via `/usr/sbin/screencapture`. + * Returns the file path and geometry — never the image bytes. + */ +export async function captureScreen(opts: CaptureOptions = {}): Promise { + if (!screenRecordingOk) { + const perms = await checkPermissions(); + if (!perms.screenRecording) { + throw new Error( + 'Screen Recording permission missing. Grant it to the host application ' + + '(Terminal / Claude Desktop / Cursor) in System Settings → Privacy & Security → Screen Recording, then restart it.' + ); + } + screenRecordingOk = true; + } + + const mode = opts.windowId != null || opts.app ? 'window' : opts.rect ? 'region' : 'display'; + const out = capturePath(opts.outPath); + const args = ['-x', '-t', 'png']; + let frame: ScreenFrame; + let targetDesc: string; + + if (mode === 'window') { + const win = await resolveWindow(opts); + args.push('-o', '-l', String(win.windowId)); // -o: no window shadow + frame = { x: win.x, y: win.y, w: win.w, h: win.h }; + targetDesc = `window ${win.windowId} (${win.app}${win.title ? `: ${win.title}` : ''})`; + } else if (mode === 'region') { + const { x, y, w, h } = opts.rect!; + args.push(`-R${x},${y},${w},${h}`); + frame = { x, y, w, h }; + targetDesc = `region ${x},${y} ${w}×${h}`; + } else { + const displays = await listDisplays(); + const display = + opts.displayId != null + ? displays.find((d) => d.displayId === opts.displayId) + : (displays.find((d) => d.isMain) ?? displays[0]); + if (!display) throw new Error(`Display ${opts.displayId} not found`); + const idx = displays.indexOf(display) + 1; // screencapture -D is 1-based ordinal + args.push('-D', String(idx)); + frame = { x: display.x, y: display.y, w: display.w, h: display.h }; + targetDesc = `display ${display.displayId}`; + } + + args.push(out); + try { + await execFileAsync('/usr/sbin/screencapture', args, { timeout: SCREENCAPTURE_TIMEOUT_MS }); + } catch (err) { + const stderr = (err as { stderr?: string }).stderr?.trim(); + throw new Error( + `screencapture failed for ${targetDesc}${stderr ? ` (${stderr})` : ''}. ` + + 'Common causes: the screen is locked or asleep, or the window was closed.' + ); + } + if (!existsSync(out)) { + throw new Error(`screencapture produced no file for ${targetDesc} — is the window on screen?`); + } + const px = await pngPixelSize(out); + return { + path: out, + pixelWidth: px.w, + pixelHeight: px.h, + frame, + scale: frame.w > 0 ? px.w / frame.w : 1, + capturedAt: new Date().toISOString(), + target: targetDesc, + }; +} From 97edf24860f6ce40c9128df18702a8fadc7771dd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 06:57:10 +0200 Subject: [PATCH 2/7] feat: extended Vision API, OCR tuning, document structure, entities and pixel ops vision-helper (macOS 13 target, newer features behind #available): - OCR options: --lang, --auto-lang, --no-correction, --custom-words, --fast, --roi (results mapped back to full-image space), --min-text-height - --document-structure (macOS 26, RecognizeDocumentsRequest): title, paragraphs, tables with spans, lists, barcodes, detected data with positions - --entities (NSDataDetector over stdin), --text-rects, --compare (feature print distance), --image-info, --languages, --capabilities - people/scenes: --face-landmarks (+capture quality, head pose), --humans, --body-pose, --hand-pose, --animals, --animal-pose, --horizon, --contours, --saliency attention|objectness (+heatmap), --aesthetics (15+), --smudge (26+, reports supported:false when the model is absent on this hardware) - pixel ops returning paths: --crop, --document-crop (CIPerspectiveCorrection), --foreground-mask (14+), --person-mask TypeScript: - src/vision.ts: typed wrappers for everything above, UnsupportedOnThisMacOSError - ocr(): TextRecognitionOptions, opt-in content-hash cache, onProgress for PDFs - captureScreen(): sha256 of the PNG - CLI: OCR tuning flags, --structure/--entities/--text-regions/--humans/ --face-landmarks/--saliency/--aesthetics/--info/--capabilities/--languages - tests: test/vision.test.ts (22 cases); README; changeset (minor); node >= 20 Co-Authored-By: Claude Fable 5 --- .changeset/ui-helper-and-extended-vision.md | 13 + README.md | 104 ++- package.json | 2 +- scripts/build-native-cross.js | 4 +- src/cli.ts | 93 +- src/index.ts | 170 +++- src/native/vision-helper.swift | 894 +++++++++++++++++++- src/ui.ts | 8 +- src/vision.ts | 589 +++++++++++++ test/vision.test.ts | 214 +++++ 10 files changed, 2057 insertions(+), 34 deletions(-) create mode 100644 .changeset/ui-helper-and-extended-vision.md create mode 100644 src/vision.ts create mode 100644 test/vision.test.ts diff --git a/.changeset/ui-helper-and-extended-vision.md b/.changeset/ui-helper-and-extended-vision.md new file mode 100644 index 0000000..b8f32ea --- /dev/null +++ b/.changeset/ui-helper-and-extended-vision.md @@ -0,0 +1,13 @@ +--- +"macos-vision": minor +--- + +feat: ui-helper (windows / displays / permissions / captureScreen) and extended Vision API + +- New native `ui-helper` (third prebuilt binary): `listWindows`, `listDisplays`, `checkPermissions`, `captureScreen` (returns path + geometry + sha256, never bytes). +- OCR tuning: `languages`, `autoDetectLanguage`, `languageCorrection`, `customWords`, `fast`, `regionOfInterest`, `minTextHeight`; opt-in content-hash `cache`; `onProgress` for PDFs. +- `recognizeDocument` (macOS 26+): native paragraphs, tables, lists, title, barcodes and detected data with positions. +- `extractEntities` (links, e-mails, phones, addresses, dates), `detectTextRegions`, `compareImages`, `imageInfo`, `visionCapabilities`, `supportedOcrLanguages`. +- People & scenes: `detectFaceLandmarks`, `detectHumans`, `detectBodyPose`, `detectHandPose`, `detectAnimals`, `detectAnimalPose`, `detectHorizon`, `detectSaliency`, `detectContours`, `imageAesthetics`, `detectLensSmudge`. +- Pixel ops that return paths: `cropImage`, `cropDocument` (perspective-corrected), `extractForeground`, `personMask`. +- Native helpers now target macOS 13 (newer features gate on `#available`). Node >= 20. diff --git a/README.md b/README.md index ece4458..0b0781e 100644 --- a/README.md +++ b/README.md @@ -6,8 +6,8 @@ Uses macOS's built-in [Vision framework](https://developer.apple.com/documentati ## Requirements -- macOS 12+ (Apple Silicon or Intel) -- Node.js 18+ +- macOS 13+ (Apple Silicon or Intel) — some features need newer macOS (see `visionCapabilities()`) +- Node.js 20+ - [Ollama](https://ollama.com) running locally — only if you use the Markdown pipeline - Xcode Command Line Tools (`xcode-select --install`) — **only** needed as an offline fallback when prebuilt binaries cannot be downloaded @@ -29,6 +29,11 @@ The native Swift binaries (`vision-helper`, `pdf-helper`, `ui-helper`) are downl | Layout inference (lines, paragraphs, reading order) | heuristic in TypeScript | offline | | PDF rasterization | PDFKit (`pdf-helper`) | offline | | Screen capture + window / display / permission introspection | `screencapture` + CoreGraphics (`ui-helper`) | offline | +| Document structure — paragraphs, tables, lists, detected data (macOS 26+) | Apple Vision `RecognizeDocumentsRequest` | offline | +| Entities in text — links, e-mails, phones, addresses, dates | Foundation `NSDataDetector` | offline | +| Image similarity, saliency, contours, text regions | Apple Vision | offline | +| People — face landmarks, body / hand pose, person masks, subject cutout | Apple Vision | offline | +| Crop / deskew / perspective-correct documents | CoreImage | offline | | **Image / PDF → Markdown** | Apple Vision OCR + local LLM via Ollama | local LLM call | --- @@ -145,6 +150,90 @@ for (const block of layout) { --- +## API — Extended Vision + +Everything below follows the same conventions: normalized 0–1 coordinates with a **top-left origin**, results as JSON, and pixel-producing operations **write PNG files and return paths** — never image bytes. Check `visionCapabilities()` first: features gate on the macOS version. + +```js +import { + visionCapabilities, supportedOcrLanguages, ocr, + recognizeDocument, extractEntities, detectTextRegions, compareImages, imageInfo, + detectFaceLandmarks, detectHumans, detectBodyPose, detectHandPose, detectAnimals, + detectSaliency, detectContours, detectHorizon, imageAesthetics, detectLensSmudge, + cropImage, cropDocument, extractForeground, personMask, +} from 'macos-vision'; + +const caps = await visionCapabilities(); +// { helperVersion, macosVersion, ocrLanguages: ['en-US','pl-PL',…], features: { documentStructure, foregroundMask, … } } +``` + +### OCR tuning + +`ocr()` now accepts Vision's recognition knobs. Results for `regionOfInterest` are still reported in full-image coordinates. + +```js +const blocks = await ocr('invoice.png', { + format: 'blocks', + languages: ['pl-PL', 'en-US'], // priority order; see supportedOcrLanguages() + languageCorrection: false, // keep IBANs, IDs and hashes verbatim + customWords: ['Prorok', 'FV/2026/08'], + regionOfInterest: { x: 0, y: 0, width: 1, height: 0.25 }, + fast: false, // true → quicker, less accurate + cache: true, // ~/.cache/macos-vision/ocr, keyed by file sha256 + options +}); +await ocr('long.pdf', { onProgress: (done, total) => console.log(`${done}/${total}`) }); +``` + +### Document structure (macOS 26+) + +Native layout understanding — no heuristics, no LLM. Throws `UnsupportedOnThisMacOSError` on older systems. + +```js +const doc = await recognizeDocument('invoice.png', { languages: ['pl-PL'] }); +doc.title?.text; // 'Faktura VAT' +doc.paragraphs[0].lines; // [{ text, confidence, bbox }] +doc.tables[0].rows; // string[][] — cell texts by row +doc.tables[0].cells; // [{ text, row, col, rowSpan, colSpan, bbox }] +doc.lists[0].items; // [{ marker, text, bbox }] +doc.detectedData; // [{ type: 'money' | 'date' | 'email' | 'phone' | 'link' | …, text, value, bbox }] +``` + +### Text utilities + +```js +await extractEntities(text); // links / e-mails / phones / addresses / dates with offsets — any macOS +await detectTextRegions('shot.png'); // where text is, without reading it (fast ROI picker) +await compareImages('before.png', 'after.png'); // { distance } — 0 identical, > ~0.8 different content +await imageInfo('photo.jpg'); // { width, height, dpi, format, orientation, … } +``` + +### People, scenes, quality + +```js +await detectFaceLandmarks('photo.jpg'); // bbox + roll/yaw/pitch + captureQuality + landmark polylines +await detectHumans('photo.jpg'); // full-body boxes +await detectBodyPose('photo.jpg'); // { joints: { left_wrist_joint: { x, y, confidence }, … } } +await detectHandPose('photo.jpg'); // + chirality +await detectAnimals('photo.jpg'); // cats & dogs with labels +await detectAnimalPose('photo.jpg'); // macOS 14+ +await detectSaliency('photo.jpg', { mode: 'attention' | 'objectness', heatmapPath: 'heat.png' }); +await detectContours('chart.png', { maxPoints: 32 }); +await detectHorizon('landscape.jpg'); // { angleDegrees } | null +await imageAesthetics('photo.jpg'); // { overallScore, isUtility } — macOS 15+; isUtility = screenshot/receipt-like +await detectLensSmudge('photo.jpg'); // { confidence, supported } — macOS 26+, supported:false when the model is absent +``` + +### Pixel operations (return paths) + +```js +await cropImage('shot.png', { x: 0.5, y: 0, width: 0.5, height: 0.3 }); // zoom for a second OCR pass +await cropDocument('receipt-photo.jpg'); // detect + perspective-correct + deskew +await extractForeground('product.jpg', { tight: true }); // subject cutout with alpha (macOS 14+) +await personMask('photo.jpg'); // 8-bit mask, white = person +``` + +--- + ## API — UI (screen capture, windows, permissions) Read-only introspection of the desktop plus PNG captures, meant as the "eyes" of UI-testing agents. Requires **Screen Recording** permission for the host process (System Settings → Privacy & Security → Screen Recording). @@ -158,7 +247,7 @@ const windows = await listWindows(); // on-screen app windows, front-to-back // Capture the frontmost Safari window (app name: exact or case-insensitive prefix) const shot = await captureScreen({ app: 'Safari' }); -// { path, pixelWidth, pixelHeight, frame: { x, y, w, h }, scale, capturedAt, target } +// { path, pixelWidth, pixelHeight, sha256, frame: { x, y, w, h }, scale, capturedAt, target } // Other targets: { windowId }, { rect: { x, y, w, h } }, { displayId } (default: main display) const region = await captureScreen({ rect: { x: 0, y: 0, w: 800, h: 600 }, outPath: './region.png' }); @@ -294,6 +383,15 @@ The `VisionScribe` API, the system prompt, and the chunking strategy are unchang | `options.format` | `'text' \| 'blocks'` | `'text'` | Plain text or structured blocks with coordinates | | `options.startPage` | `number` | `1` | PDFs only — first page to OCR, 1-based. Ignored for images. | | `options.maxPages` | `number` | all | PDFs only — maximum number of pages to OCR. Ignored for images. | +| `options.onProgress` | `(done, total) => void` | — | PDFs only — called after each page. | +| `options.languages` | `string[]` | Vision default | BCP-47 codes in priority order. | +| `options.autoDetectLanguage` | `boolean` | `false` | Let Vision pick the language per run. | +| `options.languageCorrection` | `boolean` | `true` | Disable for codes, IDs, IBANs. | +| `options.customWords` | `string[]` | — | Vocabulary that overrides the language model. | +| `options.fast` | `boolean` | `false` | `.fast` recognition level. | +| `options.regionOfInterest` | `{ x, y, width, height }` | whole image | Normalized, top-left origin; output stays in full-image space. | +| `options.minTextHeight` | `number` | — | Ignore text shorter than this fraction of image height. | +| `options.cache` | `boolean` | `false` | Cache by content hash + options in `~/.cache/macos-vision/ocr`. | Returns `Promise` or `Promise`. diff --git a/package.json b/package.json index 8ae267e..f8a4993 100644 --- a/package.json +++ b/package.json @@ -87,7 +87,7 @@ "darwin" ], "engines": { - "node": ">=18.0.0" + "node": ">=20.0.0" }, "devDependencies": { "@changesets/cli": "^2.31.0", diff --git a/scripts/build-native-cross.js b/scripts/build-native-cross.js index 61309d4..e75420a 100644 --- a/scripts/build-native-cross.js +++ b/scripts/build-native-cross.js @@ -8,8 +8,8 @@ const __dirname = path.dirname(fileURLToPath(import.meta.url)); const root = path.resolve(__dirname, '..'); const TARGETS = [ - { arch: 'arm64', swift: 'arm64-apple-macos12' }, - { arch: 'x64', swift: 'x86_64-apple-macos12' }, + { arch: 'arm64', swift: 'arm64-apple-macos13' }, + { arch: 'x64', swift: 'x86_64-apple-macos13' }, ]; const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; diff --git a/src/cli.ts b/src/cli.ts index a8f15b5..285a287 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -15,7 +15,18 @@ import { DocumentBounds, classify, Classification, + recognizeDocument, + extractEntities, + detectTextRegions, + detectHumans, + detectFaceLandmarks, + detectSaliency, + imageAesthetics, + imageInfo, + visionCapabilities, + supportedOcrLanguages, } from './index.js'; +import type { TextRecognitionOptions } from './index.js'; const USAGE = ` Usage: macos-vision [options] @@ -30,6 +41,27 @@ Vision options: --classify Image classification --all Run all of the above +OCR tuning: + --lang Recognition languages, comma-separated BCP-47 (e.g. pl-PL,en-US) + --auto-lang Let Vision detect the language automatically + --no-correction Disable language-model correction (IDs, codes, IBANs) + --custom-words Domain vocabulary that should win over the language model + --fast Fast recognition level (quicker, less accurate) + --roi x,y,w,h Only read inside this normalized region (top-left origin) + --cache Cache OCR results by file hash in ~/.cache/macos-vision/ocr + +Extended analysis: + --structure Document structure: paragraphs, tables, lists, data (macOS 26+) + --entities Links / e-mails / phones / addresses / dates found in OCR text + --text-regions Where text is, without reading it + --humans Person bounding boxes + --face-landmarks Face landmarks, head pose, capture quality + --saliency Attention-based salient regions + --aesthetics Aesthetics score + utility flag (macOS 15+) + --info Pixel dimensions and image metadata + --capabilities What this machine supports (no input file needed) + --languages Supported OCR languages (no input file needed) + PDF page range (PDFs only; ignored for images): --start-page First page to process, 1-based (default: 1) --max-pages Maximum number of pages to process (default: all) @@ -75,6 +107,9 @@ const ollamaUrl = takeOpt('--ollama-url', argv); const outPath = takeOpt('-o', argv) ?? takeOpt('--output', argv); const startPageRaw = takeOpt('--start-page', argv); const maxPagesRaw = takeOpt('--max-pages', argv); +const langRaw = takeOpt('--lang', argv); +const customWordsRaw = takeOpt('--custom-words', argv); +const roiRaw = takeOpt('--roi', argv); function parsePageOpt(name: string, raw: string | undefined): number | undefined { if (raw === undefined) return undefined; @@ -95,16 +130,46 @@ if (maxPages !== undefined) pageRange.maxPages = maxPages; const flags = new Set(argv.filter((a) => a.startsWith('--'))); const fileArgs = argv.filter((a) => !a.startsWith('-')); -if (!fileArgs[0]) { +const textOptions: TextRecognitionOptions = {}; +if (langRaw) textOptions.languages = langRaw.split(','); +if (customWordsRaw) textOptions.customWords = customWordsRaw.split(','); +if (flags.has('--auto-lang')) textOptions.autoDetectLanguage = true; +if (flags.has('--no-correction')) textOptions.languageCorrection = false; +if (flags.has('--fast')) textOptions.fast = true; +if (roiRaw) { + const [x, y, width, height] = roiRaw.split(',').map(Number); + if ([x, y, width, height].some((n) => !Number.isFinite(n))) { + console.error(`Error: --roi expects x,y,w,h (got "${roiRaw}")`); + process.exit(1); + } + textOptions.regionOfInterest = { x, y, width, height }; +} +const ocrBase = { ...pageRange, ...textOptions, ...(flags.has('--cache') ? { cache: true } : {}) }; + +const metaOnly = flags.has('--capabilities') || flags.has('--languages'); + +if (metaOnly) { + (async () => { + if (flags.has('--capabilities')) + console.log(JSON.stringify(await visionCapabilities(), null, 2)); + if (flags.has('--languages')) + console.log(JSON.stringify(await supportedOcrLanguages(), null, 2)); + })().catch((err) => { + console.error(err instanceof Error ? err.message : String(err)); + process.exit(1); + }); +} else if (!fileArgs[0]) { console.error('Error: no image or PDF path provided.\n'); console.log(USAGE); process.exit(1); } -const inputPath = resolve(fileArgs[0]); +const inputPath = fileArgs[0] ? resolve(fileArgs[0]) : ''; // ─── Markdown pipeline ───────────────────────────────────────────────────────────── -if (flags.has('--markdown')) { +if (metaOnly) { + // already handled above +} else if (flags.has('--markdown')) { const toStdout = flags.has('--stdout'); const opts: { model?: string; ollamaUrl?: string } = {}; if (model) opts.model = model; @@ -151,6 +216,17 @@ if (flags.has('--markdown')) { const runRects = runAll || flags.has('--rectangles'); const runDoc = runAll || flags.has('--document'); const runClassify = runAll || flags.has('--classify'); + const extended: Array<[string, () => Promise]> = [ + ['--structure', () => recognizeDocument(inputPath, textOptions)], + ['--entities', async () => extractEntities((await ocr(inputPath, ocrBase)) as string)], + ['--text-regions', () => detectTextRegions(inputPath, textOptions)], + ['--humans', () => detectHumans(inputPath)], + ['--face-landmarks', () => detectFaceLandmarks(inputPath)], + ['--saliency', () => detectSaliency(inputPath)], + ['--aesthetics', () => imageAesthetics(inputPath)], + ['--info', () => imageInfo(inputPath)], + ]; + const runExtended = extended.filter(([flag]) => flags.has(flag)); // Default: OCR text when no feature flag is given const anyFeatureFlag = @@ -161,21 +237,22 @@ if (flags.has('--markdown')) { flags.has('--barcodes') || flags.has('--rectangles') || flags.has('--document') || - flags.has('--classify'); + flags.has('--classify') || + runExtended.length > 0; const useDefault = !anyFeatureFlag; (async () => { try { if (useDefault || runOcr) { - const text = await ocr(inputPath, pageRange); + const text = await ocr(inputPath, ocrBase); console.log(text as string); } if (runBlocks) { const blocks = (await ocr(inputPath, { + ...ocrBase, format: 'blocks', - ...pageRange, })) as VisionBlock[]; console.log(JSON.stringify(blocks, null, 2)); } @@ -204,6 +281,10 @@ if (flags.has('--markdown')) { const labels = (await classify(inputPath)) as Classification[]; console.log(JSON.stringify(labels, null, 2)); } + + for (const [, fn] of runExtended) { + console.log(JSON.stringify(await fn(), null, 2)); + } } catch (error) { console.error('Error:', error); process.exit(1); diff --git a/src/index.ts b/src/index.ts index a6bd3e0..89069ff 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2,7 +2,11 @@ import { execFile } from 'child_process'; import { promisify } from 'util'; import { resolve, dirname, extname, dirname as pathDirname } from 'path'; import { fileURLToPath } from 'url'; -import { open } from 'fs/promises'; +import { open, readFile, writeFile, mkdir } from 'fs/promises'; +import { createHash } from 'crypto'; +import { homedir } from 'os'; +import { textOptionArgs } from './vision.js'; +import type { TextRecognitionOptions } from './vision.js'; const execFileAsync = promisify(execFile); const __dirname = dirname(fileURLToPath(import.meta.url)); @@ -10,6 +14,7 @@ const BIN_PATH = resolve(__dirname, '../bin/vision-helper'); const PDF_BIN_PATH = resolve(__dirname, '../bin/pdf-helper'); const BINARY_TIMEOUT_MS = 30_000; const PDF_RASTERIZE_TIMEOUT_MS = 120_000; +const OCR_CACHE_DIR = resolve(homedir(), '.cache', 'macos-vision', 'ocr'); async function run(flag: string, imagePath: string): Promise { const { stdout } = await execFileAsync(BIN_PATH, [flag, resolve(imagePath)], { @@ -57,6 +62,8 @@ export interface PdfPageRangeOptions { startPage?: number; /** Maximum number of pages to process. Default: all pages from `startPage`. Ignored for non-PDF inputs. */ maxPages?: number; + /** Called after each page is OCR'd (PDF inputs only). `done` counts from 1. */ + onProgress?: (done: number, total: number) => void; } function buildPdfArgs(absPath: string, options: PdfPageRangeOptions): string[] { @@ -107,24 +114,68 @@ export async function rasterizePdf( async function ocrPdf( pdfPath: string, format: 'text' | 'blocks', - range: PdfPageRangeOptions = {} + options: OcrOptions = {} ): Promise { - const { pages } = await rasterizePdf(pdfPath, range); + // eslint-disable-next-line @typescript-eslint/no-unused-vars + const { startPage, maxPages, onProgress, format: _format, ...textOptions } = options; + const { pages } = await rasterizePdf(pdfPath, { startPage, maxPages }); if (format === 'blocks') { const all: VisionBlock[] = []; - for (const { page, path: pagePath } of pages) { - const blocks = (await ocr(pagePath, { format: 'blocks' })) as VisionBlock[]; + for (const [i, { page, path: pagePath }] of pages.entries()) { + const blocks = (await ocr(pagePath, { ...textOptions, format: 'blocks' })) as VisionBlock[]; all.push(...blocks.map((b) => ({ ...b, page }))); + onProgress?.(i + 1, pages.length); } return all; } const texts: string[] = []; - for (const { path: pagePath } of pages) { - texts.push((await ocr(pagePath)) as string); + for (const [i, { path: pagePath }] of pages.entries()) { + texts.push((await ocr(pagePath, { ...textOptions, format: 'text' })) as string); + onProgress?.(i + 1, pages.length); } return texts.join('\n\n--- Page Break ---\n\n'); } +// ─── OCR result cache ──────────────────────────────────────────────────────── + +/** SHA-256 of a file's bytes, hex. */ +export async function fileSha256(filePath: string): Promise { + return createHash('sha256') + .update(await readFile(filePath)) + .digest('hex'); +} + +async function cacheKey( + absPath: string, + format: string, + opts: TextRecognitionOptions +): Promise { + const content = await fileSha256(absPath); + const optsKey = JSON.stringify({ + format, + ...opts, + regionOfInterest: opts.regionOfInterest ?? null, + }); + return createHash('sha256').update(content).update(optsKey).digest('hex'); +} + +async function readCache(key: string): Promise { + try { + return JSON.parse(await readFile(resolve(OCR_CACHE_DIR, `${key}.json`), 'utf8')) as T; + } catch { + return undefined; + } +} + +async function writeCache(key: string, value: unknown): Promise { + try { + await mkdir(OCR_CACHE_DIR, { recursive: true }); + await writeFile(resolve(OCR_CACHE_DIR, `${key}.json`), JSON.stringify(value)); + } catch { + // cache is best-effort + } +} + // ─── OCR ───────────────────────────────────────────────────────────────────── export interface VisionBlock { @@ -144,9 +195,15 @@ export interface VisionBlock { page?: number; } -export interface OcrOptions extends PdfPageRangeOptions { +export interface OcrOptions extends PdfPageRangeOptions, TextRecognitionOptions { /** Return plain text (default) or structured blocks with coordinates */ format?: 'text' | 'blocks'; + /** + * Cache results in `~/.cache/macos-vision/ocr/` keyed by file content hash + options. + * Repeated OCR of the same bytes (e.g. an agent re-reading a screenshot) returns instantly. + * Default false. + */ + cache?: boolean; } export async function ocr( @@ -162,17 +219,31 @@ export async function ocr( options: OcrOptions = {} ): Promise { const absPath = resolve(imagePath); - const { format = 'text', startPage, maxPages } = options; + const { + format = 'text', + cache = false, + startPage, + maxPages, + onProgress, + ...textOptions + } = options; // ── PDF fast-path: rasterize via pdf-helper, then OCR each page ────── if (await isPdf(absPath)) { - return ocrPdf(absPath, format, { startPage, maxPages }); + return ocrPdf(absPath, format, { ...textOptions, cache, startPage, maxPages, onProgress }); + } + + const key = cache ? await cacheKey(absPath, format, textOptions) : undefined; + if (key) { + const hit = await readCache(key); + if (hit !== undefined) return hit; } + const textArgs = textOptionArgs(textOptions); - // ── Existing image path (unchanged) ───────────────────────────────── if (format === 'blocks') { - const { stdout } = await execFileAsync(BIN_PATH, ['--json', absPath], { + const { stdout } = await execFileAsync(BIN_PATH, ['--json', ...textArgs, absPath], { timeout: BINARY_TIMEOUT_MS, + maxBuffer: 64 * 1024 * 1024, }); const raw: Array<{ t: string; @@ -182,7 +253,7 @@ export async function ocr( h: number; confidence: number; }> = JSON.parse(stdout); - return raw.map((b) => ({ + const blocks = raw.map((b) => ({ text: b.t, x: b.x, y: b.y, @@ -190,10 +261,17 @@ export async function ocr( height: b.h, confidence: b.confidence, })); + if (key) await writeCache(key, blocks); + return blocks; } - const { stdout } = await execFileAsync(BIN_PATH, [absPath], { timeout: BINARY_TIMEOUT_MS }); - return stdout.trim(); + const { stdout } = await execFileAsync(BIN_PATH, [...textArgs, absPath], { + timeout: BINARY_TIMEOUT_MS, + maxBuffer: 64 * 1024 * 1024, + }); + const text = stdout.trim(); + if (key) await writeCache(key, text); + return text; } // ─── Face detection ────────────────────────────────────────────────────── @@ -349,3 +427,65 @@ export type { CaptureResult, CaptureOptions, } from './ui.js'; + +// ─── Extended Vision API ─────────────────────────────────────────────────────── +export { + visionCapabilities, + supportedOcrLanguages, + imageInfo, + detectTextRegions, + compareImages, + extractEntities, + recognizeDocument, + detectLensSmudge, + imageAesthetics, + detectHorizon, + detectFaceLandmarks, + detectHumans, + detectBodyPose, + detectHandPose, + detectAnimalPose, + detectAnimals, + detectSaliency, + detectContours, + cropImage, + cropDocument, + extractForeground, + personMask, + UnsupportedOnThisMacOSError, +} from './vision.js'; +export type { + NormalizedRect, + TextRecognitionOptions, + VisionCapabilities, + ImageInfo, + TextRegion, + ImageComparison, + TextEntity, + DocumentStructure, + DocText, + DocLine, + DocBox, + DocTable, + DocCell, + DocList, + DocListItem, + DocDetectedData, + DocBarcode, + LensSmudge, + AestheticsScore, + Horizon, + FaceLandmarks, + HumanBox, + Keypoint, + Pose, + Animal, + SaliencyOptions, + Saliency, + ContourOptions, + Contour, + Contours, + CropResult, + MaskResult, + ForegroundOptions, +} from './vision.js'; diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index 3b43128..af9937c 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -1,6 +1,10 @@ import Vision import AppKit import Foundation +import CoreGraphics +import CoreImage +import ImageIO +import UniformTypeIdentifiers // ─── Result structs ────────────────────────────────────────────────────────── @@ -52,6 +56,123 @@ func encodeJSON(_ value: T) -> String { return str } +struct TextRectResult: Codable { + let x: Double; let y: Double; let w: Double; let h: Double + let confidence: Float +} + +struct CompareResult: Codable { + let distance: Double +} + +struct SmudgeResult: Codable { + let confidence: Float + let supported: Bool +} + +struct EntityResult: Codable { + let type: String + let text: String + let start: Int + let end: Int + let value: String? + let components: [String: String]? +} + +struct Capabilities: Codable { + let helperVersion: String + let macosVersion: String + let ocrLanguages: [String] + let features: [String: Bool] +} + +// Document structure (macOS 26+, RecognizeDocumentsRequest) +struct DocRegion: Codable { let x: Double; let y: Double; let w: Double; let h: Double } +struct DocLine: Codable { let text: String; let confidence: Float; let bbox: DocRegion } +struct DocText: Codable { + let text: String + let alignment: String? + let bbox: DocRegion + let lines: [DocLine] +} +struct DocCell: Codable { + let text: String + let row: Int; let col: Int + let rowSpan: Int; let colSpan: Int + let bbox: DocRegion +} +struct DocTable: Codable { + let rowCount: Int; let columnCount: Int + let rows: [[String]] + let cells: [DocCell] + let bbox: DocRegion +} +struct DocListItem: Codable { let marker: String; let text: String; let bbox: DocRegion } +struct DocList: Codable { let items: [DocListItem]; let bbox: DocRegion } +struct DocData: Codable { let type: String; let text: String; let value: String?; let bbox: DocRegion } +struct DocBarcode: Codable { let type: String; let value: String; let bbox: DocRegion } +struct DocStructure: Codable { + let title: DocText? + let text: String + let paragraphs: [DocText] + let tables: [DocTable] + let lists: [DocList] + let barcodes: [DocBarcode] + let detectedData: [DocData] +} + +struct PointResult: Codable { let x: Double; let y: Double; let confidence: Float } +struct FaceLandmarksResult: Codable { + let x: Double; let y: Double; let w: Double; let h: Double + let confidence: Float + let roll: Double?; let yaw: Double?; let pitch: Double? + let captureQuality: Float? + let landmarks: [String: [[Double]]] +} +struct HumanResult: Codable { let x: Double; let y: Double; let w: Double; let h: Double; let confidence: Float } +struct PoseResult: Codable { let joints: [String: PointResult]; let confidence: Float; let chirality: String? } +struct AnimalResult: Codable { + let labels: [ClassificationResult] + let x: Double; let y: Double; let w: Double; let h: Double + let confidence: Float +} +struct HorizonResult: Codable { let angleDegrees: Double } +struct ContourResult: Codable { + let index: Int; let pointCount: Int; let childCount: Int + let x: Double; let y: Double; let w: Double; let h: Double + let points: [[Double]]? +} +struct ContoursResult: Codable { let totalContours: Int; let topLevel: [ContourResult] } +struct SaliencyResult: Codable { + let regions: [HumanResult] + let heatmapPath: String? +} +struct MaskResult: Codable { let instances: Int; let outPath: String } +struct AestheticsResult: Codable { let overallScore: Float; let isUtility: Bool } +struct CropResult: Codable { let outPath: String; let width: Int; let height: Int } +struct ImageInfoResult: Codable { + let width: Int; let height: Int + let hasAlpha: Bool; let bitsPerComponent: Int + let colorSpace: String?; let dpi: Double?; let orientation: Int?; let format: String? +} + +let HELPER_VERSION = "2" + +func macOSVersionString() -> String { + let v = ProcessInfo.processInfo.operatingSystemVersion + return "\(v.majorVersion).\(v.minorVersion).\(v.patchVersion)" +} + +func hasAestheticsAPI() -> Bool { + if #available(macOS 15.0, *) { return true } + return false +} + +func hasDocumentAPI() -> Bool { + if #available(macOS 26.0, *) { return true } + return false +} + // ─── Argument parsing ───────────────────────────────────────────────────────── let args = CommandLine.arguments @@ -61,11 +182,157 @@ let isBarcodes = args.contains("--barcodes") let isRectangles = args.contains("--rectangles") let isDocument = args.contains("--document") let isClassify = args.contains("--classify") +let isTextRects = args.contains("--text-rects") +let isCompare = args.contains("--compare") +let isSmudge = args.contains("--smudge") +let isStructure = args.contains("--document-structure") +let isEntities = args.contains("--entities") +let isLandmarks = args.contains("--face-landmarks") +let isHumans = args.contains("--humans") +let isBodyPose = args.contains("--body-pose") +let isHandPose = args.contains("--hand-pose") +let isAnimals = args.contains("--animals") +let isAnimalPose = args.contains("--animal-pose") +let isHorizon = args.contains("--horizon") +let isContours = args.contains("--contours") +let isSaliency = args.contains("--saliency") +let isForeground = args.contains("--foreground-mask") +let isPersonMask = args.contains("--person-mask") +let isAesthetics = args.contains("--aesthetics") +let isDocCrop = args.contains("--document-crop") +let isCrop = args.contains("--crop") +let isImageInfo = args.contains("--image-info") +let anyNewMode = isLandmarks || isHumans || isBodyPose || isHandPose || isAnimals || isAnimalPose || isHorizon + || isContours || isSaliency || isForeground || isPersonMask || isAesthetics || isDocCrop || isCrop || isImageInfo + +/// Value following a `--flag`, or nil. +func optValue(_ flag: String) -> String? { + guard let i = args.firstIndex(of: flag), i + 1 < args.count else { return nil } + return args[i + 1] +} + +// Flags that consume the next argument (so it is not mistaken for a file path). +let valueFlags: Set = ["--lang", "--custom-words", "--roi", "--min-text-height", "--out", "--saliency", "--crop", "--max-points"] +var consumed = Set() +for (i, a) in args.enumerated() where valueFlags.contains(a) && i + 1 < args.count { consumed.insert(i + 1) } +let fileArgs = args.enumerated() + .filter { i, a in i > 0 && !a.hasPrefix("--") && !consumed.contains(i) } + .map { $0.element } -let fileArgs = args.filter { !$0.hasPrefix("--") && !$0.contains("vision-helper") } +// ─── Modes that need no image ─────────────────────────────────────────────── + +if args.contains("--languages") { + let req = VNRecognizeTextRequest() + req.recognitionLevel = .accurate + let langs = (try? req.supportedRecognitionLanguages()) ?? [] + print(encodeJSON(langs)) + exit(0) +} + +if args.contains("--capabilities") { + let req = VNRecognizeTextRequest() + req.recognitionLevel = .accurate + let langs = (try? req.supportedRecognitionLanguages()) ?? [] + let caps = Capabilities( + helperVersion: HELPER_VERSION, + macosVersion: macOSVersionString(), + ocrLanguages: langs, + features: [ + "ocr": true, "ocrOptions": true, "faces": true, "barcodes": true, + "rectangles": true, "document": true, "classify": true, + "textRects": true, "compare": true, "entities": true, + "documentStructure": hasDocumentAPI(), "lensSmudge": hasDocumentAPI(), + "faceLandmarks": true, "humans": true, "bodyPose": true, "handPose": true, + "animals": true, "animalPose": true, "horizon": true, "contours": true, + "saliency": true, "foregroundMask": true, "personMask": true, + "aesthetics": hasAestheticsAPI(), "documentCrop": true, "crop": true, "imageInfo": true, + ] + ) + print(encodeJSON(caps)) + exit(0) +} + +// --entities: NSDataDetector over UTF-8 text read from stdin. No image involved. +if isEntities { + let data = FileHandle.standardInput.readDataToEndOfFile() + let text = String(decoding: data, as: UTF8.self) + let types: NSTextCheckingResult.CheckingType = [.link, .phoneNumber, .address, .date, .transitInformation] + guard let detector = try? NSDataDetector(types: types.rawValue) else { + fputs("ERROR: NSDataDetector unavailable\n", stderr) + exit(1) + } + let ns = text as NSString + var results: [EntityResult] = [] + let iso = ISO8601DateFormatter() + for m in detector.matches(in: text, options: [], range: NSRange(location: 0, length: ns.length)) { + let matched = ns.substring(with: m.range) + var type = "unknown" + var value: String? = nil + var comps: [String: String]? = nil + switch m.resultType { + case .link: + type = "link"; value = m.url?.absoluteString + if let u = m.url, u.scheme == "mailto" { type = "email"; value = String(u.absoluteString.dropFirst(7)) } + case .phoneNumber: + type = "phone"; value = m.phoneNumber + case .address: + type = "address" + if let c = m.addressComponents { + var d: [String: String] = [:] + for (k, v) in c { d[k.rawValue] = v } + comps = d + value = c[.street].map { s in [s, c[.city], c[.zip]].compactMap { $0 }.joined(separator: ", ") } + } + case .date: + type = "date"; value = m.date.map { iso.string(from: $0) } + if m.duration > 0 { comps = ["durationSeconds": String(Int(m.duration))] } + case .transitInformation: + type = "transit" + if let c = m.components { var d: [String: String] = [:]; for (k, v) in c { d[k.rawValue] = v }; comps = d } + default: break + } + results.append(EntityResult(type: type, text: matched, start: m.range.location, + end: m.range.location + m.range.length, value: value, components: comps)) + } + print(encodeJSON(results)) + exit(0) +} guard let imagePath = fileArgs.first else { - print("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify] ") + print("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify|--text-rects|--document-structure|--smudge|--compare ] ") + print(" vision-helper --languages | --capabilities | --entities < text.txt") + print("OCR options: --lang pl,en --auto-lang --no-correction --custom-words a,b --fast --roi x,y,w,h --min-text-height f") + exit(0) +} + +// ─── Image compare (feature print distance) ────────────────────────────────── + +if isCompare { + guard fileArgs.count >= 2 else { + fputs("ERROR: --compare needs two image paths\n", stderr) + exit(1) + } + func featurePrint(_ path: String) -> VNFeaturePrintObservation? { + guard let img = NSImage(contentsOf: URL(fileURLWithPath: path)), + let cg = img.cgImage(forProposedRect: nil, context: nil, hints: nil) else { + fputs("ERROR: Cannot open file: \(path)\n", stderr) + exit(1) + } + let req = VNGenerateImageFeaturePrintRequest() + let h = VNImageRequestHandler(cgImage: cg, options: [:]) + try? h.perform([req]) + return req.results?.first as? VNFeaturePrintObservation + } + guard let a = featurePrint(fileArgs[0]), let b = featurePrint(fileArgs[1]) else { + fputs("ERROR: Vision feature print failed\n", stderr) + exit(1) + } + var distance: Float = 0 + do { try a.computeDistance(&distance, to: b) } catch { + fputs("ERROR: Vision feature print distance failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + print(encodeJSON(CompareResult(distance: Double(distance)))) exit(0) } @@ -77,9 +344,40 @@ guard let image = NSImage(contentsOf: URL(fileURLWithPath: imagePath)), let handler = VNImageRequestHandler(cgImage: cgImage, options: [:]) +// ─── Shared OCR options ────────────────────────────────────────────────────── + +let ocrLanguages: [String] = optValue("--lang")?.split(separator: ",").map { String($0).trimmingCharacters(in: .whitespaces) }.filter { !$0.isEmpty } ?? [] +let ocrCustomWords: [String] = optValue("--custom-words")?.split(separator: ",").map { String($0) }.filter { !$0.isEmpty } ?? [] +let ocrAutoLang = args.contains("--auto-lang") +let ocrNoCorrection = args.contains("--no-correction") +let ocrFast = args.contains("--fast") +let ocrMinHeight: Float? = optValue("--min-text-height").flatMap { Float($0) } + +// Region of interest: normalized x,y,w,h with TOP-LEFT origin (same space as our output). +// Vision wants bottom-left origin, so flip. Results stay relative to the full image +// because the handler reports observations in full-image coordinates. +var roiRect: CGRect? = nil +if let roi = optValue("--roi") { + let p = roi.split(separator: ",").compactMap { Double($0) } + if p.count == 4 { + roiRect = CGRect(x: p[0], y: 1.0 - p[1] - p[3], width: p[2], height: p[3]) + } else { + fputs("ERROR: --roi expects x,y,w,h (normalized 0-1)\n", stderr) + exit(1) + } +} + +/// Vision reports observations relative to `regionOfInterest`; map back to full-image space. +func unROI(_ r: CGRect) -> CGRect { + guard let roi = roiRect else { return r } + return CGRect(x: roi.origin.x + r.origin.x * roi.width, + y: roi.origin.y + r.origin.y * roi.height, + width: r.width * roi.width, height: r.height * roi.height) +} + // ─── OCR (default + --json) ─────────────────────────────────────────────────── -if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !isClassify) { +if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !isClassify && !isTextRects && !isSmudge && !isStructure && !anyNewMode) { var ocrResults: [OCRResult] = [] var rawText = "" @@ -87,7 +385,7 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is guard let obs = req.results as? [VNRecognizedTextObservation] else { return } for o in obs { guard let c = o.topCandidates(1).first else { continue } - let box = o.boundingBox + let box = unROI(o.boundingBox) if isJsonMode { ocrResults.append(OCRResult( t: c.string, @@ -102,7 +400,13 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is } } } - request.recognitionLevel = .accurate + request.recognitionLevel = ocrFast ? .fast : .accurate + if !ocrLanguages.isEmpty { request.recognitionLanguages = ocrLanguages } + if ocrAutoLang { request.automaticallyDetectsLanguage = true } + request.usesLanguageCorrection = !ocrNoCorrection + if !ocrCustomWords.isEmpty { request.customWords = ocrCustomWords } + if let mh = ocrMinHeight { request.minimumTextHeight = mh } + if let r = roiRect { request.regionOfInterest = r } do { try handler.perform([request]) @@ -114,6 +418,586 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is exit(0) } +// ─── Text rectangles (fast "where is text" without recognition) ────────────── + +if isTextRects { + var results: [TextRectResult] = [] + let request = VNDetectTextRectanglesRequest { (req, _) in + guard let obs = req.results as? [VNTextObservation] else { return } + for o in obs { + let box = unROI(o.boundingBox) + results.append(TextRectResult( + x: Double(box.origin.x), + y: flipY(Double(box.origin.y), Double(box.size.height)), + w: Double(box.size.width), + h: Double(box.size.height), + confidence: o.confidence + )) + } + } + if let r = roiRect { request.regionOfInterest = r } + do { + try handler.perform([request]) + } catch { + fputs("ERROR: Vision text rectangle detection failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + print(encodeJSON(results)) + exit(0) +} + +// ─── macOS 26+: document structure & lens smudge (new Vision Swift API) ────── + +@available(macOS 26.0, *) +func region(_ r: NormalizedRegion) -> DocRegion { + let b = r.boundingBox + return DocRegion(x: Double(b.origin.x), y: flipY(Double(b.origin.y), Double(b.height)), + w: Double(b.width), h: Double(b.height)) +} + +@available(macOS 26.0, *) +func docText(_ t: DocumentObservation.Container.Text) -> DocText { + let align: String? + switch t.textAlignment { + case .some(.leading): align = "leading" + case .some(.trailing): align = "trailing" + case .some(.center): align = "center" + default: align = nil + } + return DocText( + text: t.transcript, + alignment: align, + bbox: region(t.boundingRegion), + lines: t.lines.map { DocLine(text: $0.transcript, confidence: $0.confidence, bbox: region($0.boundingRegion)) } + ) +} + +@available(macOS 26.0, *) +func docData(_ d: DocumentObservation.Container.DataDetectorMatch, in text: String) -> DocData { + var type = "unknown" + var value: String? = nil + switch d.match.details { + case .link(let l): type = "link"; value = l.url.absoluteString + case .emailAddress(let e): type = "email"; value = e.emailAddress + case .phoneNumber(let p): type = "phone"; value = p.phoneNumber + case .postalAddress(let a): type = "address"; value = a.fullAddress + case .calendarEvent(let c): type = "date"; value = c.startDate.map { ISO8601DateFormatter().string(from: $0) } + case .moneyAmount(let m): type = "money"; value = "\(m.amount) \(m.currency.identifier)" + case .flightNumber(let f): type = "flight"; value = "\(f.airlineCode)\(f.flightNumber)" + case .shipmentTrackingNumber(let s): type = "tracking"; value = s.trackingNumber + case .measurement(let m): type = "measurement"; value = String(m.value) + case .paymentIdentifier(let p): type = "payment"; value = p.identifier + @unknown default: break + } + let matched = d.match.range.map { String(text[$0]) } ?? "" + return DocData(type: type, text: matched, value: value, bbox: region(d.boundingRegion)) +} + +@available(macOS 26.0, *) +func runDocumentStructure() -> DocStructure? { + var req = RecognizeDocumentsRequest() + var opts = req.textRecognitionOptions + if !ocrLanguages.isEmpty { opts.recognitionLanguages = ocrLanguages.map { Locale.Language(identifier: $0) } } + if ocrAutoLang { opts.automaticallyDetectLanguage = true } + opts.useLanguageCorrection = !ocrNoCorrection + if !ocrCustomWords.isEmpty { opts.customWords = ocrCustomWords } + if let mh = ocrMinHeight { opts.minimumTextHeightFraction = mh } + req.textRecognitionOptions = opts + if let r = roiRect { req.regionOfInterest = NormalizedRect(normalizedRect: r) } + + let sema = DispatchSemaphore(value: 0) + var result: DocStructure? = nil + var failure: Error? = nil + Task { + do { + let observations = try await req.perform(on: cgImage) + if let doc = observations.first { + let c = doc.document + let full = c.text.transcript + var tables: [DocTable] = [] + for t in c.tables { + var cells: [DocCell] = [] + var rows: [[String]] = [] + for (ri, row) in t.rows.enumerated() { + var rowTexts: [String] = [] + for cell in row { + // A spanning cell appears in every row it covers; emit it once. + if cell.rowRange.lowerBound == ri { + cells.append(DocCell( + text: cell.content.text.transcript, + row: cell.rowRange.lowerBound, col: cell.columnRange.lowerBound, + rowSpan: cell.rowRange.count, colSpan: cell.columnRange.count, + bbox: region(cell.content.boundingRegion) + )) + } + rowTexts.append(cell.content.text.transcript) + } + rows.append(rowTexts) + } + tables.append(DocTable(rowCount: t.rows.count, columnCount: t.columns.count, + rows: rows, cells: cells, bbox: region(t.boundingRegion))) + } + let lists = c.lists.map { l in + DocList(items: l.items.map { DocListItem(marker: $0.markerString, text: $0.itemString, + bbox: region($0.content.boundingRegion)) }, + bbox: region(l.boundingRegion)) + } + let barcodes = c.barcodes.map { + DocBarcode(type: String(describing: $0.symbology), value: $0.payloadString ?? "", bbox: region($0.boundingRegion)) + } + result = DocStructure( + title: c.title.map(docText), + text: full, + paragraphs: c.paragraphs.map(docText), + tables: tables, + lists: lists, + barcodes: barcodes, + detectedData: c.text.detectedData.map { docData($0, in: full) } + ) + } else { + result = DocStructure(title: nil, text: "", paragraphs: [], tables: [], lists: [], barcodes: [], detectedData: []) + } + } catch { + failure = error + } + sema.signal() + } + sema.wait() + if let e = failure { + fputs("ERROR: Vision document recognition failed: \(e.localizedDescription)\n", stderr) + exit(1) + } + return result +} + +@available(macOS 26.0, *) +func runSmudge() -> SmudgeResult { + let req = DetectLensSmudgeRequest() + // VisionCore logs "Unable to find a valid E5 ..." straight to stdout on hardware + // without the smudge model. Divert stdout to a temp file while the request runs + // so the JSON we print afterwards stays clean, and use the noise to flag support. + let tmp = NSTemporaryDirectory() + "vision-helper-smudge-\(getpid()).log" + let savedStdout = dup(STDOUT_FILENO) + fflush(stdout) + let fd = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0o600) + if fd >= 0 { dup2(fd, STDOUT_FILENO); close(fd) } + let sema = DispatchSemaphore(value: 0) + var conf: Float = 0 + var failure: Error? = nil + Task { + do { conf = try await req.perform(on: cgImage).confidence } catch { failure = error } + sema.signal() + } + sema.wait() + fflush(stdout) + dup2(savedStdout, STDOUT_FILENO) + close(savedStdout) + let noise = (try? String(contentsOfFile: tmp, encoding: .utf8)) ?? "" + unlink(tmp) + if let e = failure { + fputs("ERROR: Vision lens smudge detection failed: \(e.localizedDescription)\n", stderr) + exit(1) + } + let supported = !noise.contains("Unable to find") + return SmudgeResult(confidence: conf, supported: supported) +} + +if isStructure { + if #available(macOS 26.0, *) { + if let s = runDocumentStructure() { print(encodeJSON(s)) } + exit(0) + } + fputs("ERROR: --document-structure requires macOS 26 or newer\n", stderr) + exit(2) +} + +if isSmudge { + if #available(macOS 26.0, *) { + print(encodeJSON(runSmudge())) + exit(0) + } + fputs("ERROR: --smudge requires macOS 26 or newer\n", stderr) + exit(2) +} + + +// ─── Pixel output helpers ──────────────────────────────────────────────────── + +func writePNG(_ cg: CGImage, to path: String) -> Bool { + let url = URL(fileURLWithPath: path) as CFURL + guard let dest = CGImageDestinationCreateWithURL(url, UTType.png.identifier as CFString, 1, nil) else { return false } + CGImageDestinationAddImage(dest, cg, nil) + return CGImageDestinationFinalize(dest) +} + +let ciContext = CIContext(options: nil) + +func cgFromPixelBuffer(_ pb: CVPixelBuffer) -> CGImage? { + let ci = CIImage(cvPixelBuffer: pb) + return ciContext.createCGImage(ci, from: ci.extent) +} + +func requireOut() -> String { + guard let out = optValue("--out") else { + fputs("ERROR: this mode requires --out \n", stderr) + exit(1) + } + return out +} + +func box(_ r: CGRect) -> (Double, Double, Double, Double) { + (Double(r.origin.x), flipY(Double(r.origin.y), Double(r.size.height)), Double(r.size.width), Double(r.size.height)) +} + +func jointsDict(_ points: [VNRecognizedPointKey: VNRecognizedPoint]) -> [String: PointResult] { + var d: [String: PointResult] = [:] + for (k, p) in points where p.confidence > 0 { + d[k.rawValue] = PointResult(x: Double(p.x), y: 1.0 - Double(p.y), confidence: p.confidence) + } + return d +} + +// ─── Image info (no Vision) ────────────────────────────────────────────────── + +if isImageInfo { + var dpi: Double? = nil, orientation: Int? = nil, format: String? = nil + if let src = CGImageSourceCreateWithURL(URL(fileURLWithPath: imagePath) as CFURL, nil) { + format = CGImageSourceGetType(src) as String? + if let props = CGImageSourceCopyPropertiesAtIndex(src, 0, nil) as? [CFString: Any] { + dpi = props[kCGImagePropertyDPIWidth] as? Double + orientation = props[kCGImagePropertyOrientation] as? Int + } + } + let info = ImageInfoResult( + width: cgImage.width, height: cgImage.height, + hasAlpha: cgImage.alphaInfo != .none && cgImage.alphaInfo != .noneSkipFirst && cgImage.alphaInfo != .noneSkipLast, + bitsPerComponent: cgImage.bitsPerComponent, + colorSpace: cgImage.colorSpace?.name as String?, + dpi: dpi, orientation: orientation, format: format + ) + print(encodeJSON(info)) + exit(0) +} + +// ─── Crop (normalized region, top-left origin) ─────────────────────────────── + +if isCrop { + let out = requireOut() + let p = optValue("--crop")?.split(separator: ",").compactMap { Double($0) } ?? [] + guard p.count == 4 else { + fputs("ERROR: --crop expects x,y,w,h (normalized 0-1)\n", stderr) + exit(1) + } + let W = Double(cgImage.width), H = Double(cgImage.height) + let rect = CGRect(x: p[0] * W, y: p[1] * H, width: p[2] * W, height: p[3] * H).integral + guard let cropped = cgImage.cropping(to: rect), writePNG(cropped, to: out) else { + fputs("ERROR: crop failed\n", stderr) + exit(1) + } + print(encodeJSON(CropResult(outPath: out, width: cropped.width, height: cropped.height))) + exit(0) +} + +// ─── Document crop (perspective-corrected) ─────────────────────────────────── + +if isDocCrop { + let out = requireOut() + let request = VNDetectDocumentSegmentationRequest() + try? handler.perform([request]) + guard let o = request.results?.first as? VNRectangleObservation else { + fputs("ERROR: no document detected\n", stderr) + exit(1) + } + let ci = CIImage(cgImage: cgImage) + let W = ci.extent.width, H = ci.extent.height + func p(_ pt: CGPoint) -> CGPoint { CGPoint(x: pt.x * W, y: pt.y * H) } + let filter = CIFilter(name: "CIPerspectiveCorrection")! + filter.setValue(ci, forKey: kCIInputImageKey) + filter.setValue(CIVector(cgPoint: p(o.topLeft)), forKey: "inputTopLeft") + filter.setValue(CIVector(cgPoint: p(o.topRight)), forKey: "inputTopRight") + filter.setValue(CIVector(cgPoint: p(o.bottomLeft)), forKey: "inputBottomLeft") + filter.setValue(CIVector(cgPoint: p(o.bottomRight)), forKey: "inputBottomRight") + guard let outCI = filter.outputImage, + let outCG = ciContext.createCGImage(outCI, from: outCI.extent), + writePNG(outCG, to: out) else { + fputs("ERROR: perspective correction failed\n", stderr) + exit(1) + } + print(encodeJSON(CropResult(outPath: out, width: outCG.width, height: outCG.height))) + exit(0) +} + +// ─── Face landmarks + capture quality ──────────────────────────────────────── + +if isLandmarks { + let lm = VNDetectFaceLandmarksRequest() + let cq = VNDetectFaceCaptureQualityRequest() + do { try handler.perform([lm, cq]) } catch { + fputs("ERROR: Vision face landmarks failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let faces = (lm.results ?? []) + let quals = (cq.results ?? []) + var results: [FaceLandmarksResult] = [] + for (i, f) in faces.enumerated() { + let b = box(f.boundingBox) + var marks: [String: [[Double]]] = [:] + if let l = f.landmarks { + let regions: [(String, VNFaceLandmarkRegion2D?)] = [ + ("faceContour", l.faceContour), ("leftEye", l.leftEye), ("rightEye", l.rightEye), + ("leftEyebrow", l.leftEyebrow), ("rightEyebrow", l.rightEyebrow), ("nose", l.nose), + ("noseCrest", l.noseCrest), ("medianLine", l.medianLine), ("outerLips", l.outerLips), + ("innerLips", l.innerLips), ("leftPupil", l.leftPupil), ("rightPupil", l.rightPupil), + ] + for (name, r) in regions { + guard let r = r else { continue } + // pointsInImage gives pixel coords (bottom-left origin); normalize + flip. + let pts = r.pointsInImage(imageSize: CGSize(width: cgImage.width, height: cgImage.height)) + marks[name] = pts.map { [Double($0.x) / Double(cgImage.width), 1.0 - Double($0.y) / Double(cgImage.height)] } + } + } + let q: Float? = i < quals.count ? quals[i].faceCaptureQuality : nil + results.append(FaceLandmarksResult( + x: b.0, y: b.1, w: b.2, h: b.3, confidence: f.confidence, + roll: f.roll.map { Double(truncating: $0) * 180 / .pi }, + yaw: f.yaw.map { Double(truncating: $0) * 180 / .pi }, + pitch: f.pitch.map { Double(truncating: $0) * 180 / .pi }, + captureQuality: q, landmarks: marks + )) + } + print(encodeJSON(results)) + exit(0) +} + +// ─── Human rectangles ──────────────────────────────────────────────────────── + +if isHumans { + let request = VNDetectHumanRectanglesRequest() + request.upperBodyOnly = false + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision human detection failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let results = ((request.results ?? [])).map { o -> HumanResult in + let b = box(o.boundingBox) + return HumanResult(x: b.0, y: b.1, w: b.2, h: b.3, confidence: o.confidence) + } + print(encodeJSON(results)) + exit(0) +} + +// ─── Body / hand / animal pose ─────────────────────────────────────────────── + +if isBodyPose { + let request = VNDetectHumanBodyPoseRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision body pose failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let results = ((request.results ?? [])).map { o in + PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) + } + print(encodeJSON(results)) + exit(0) +} + +if isHandPose { + let request = VNDetectHumanHandPoseRequest() + request.maximumHandCount = 4 + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision hand pose failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let results = ((request.results ?? [])).map { o -> PoseResult in + let ch: String + switch o.chirality { case .left: ch = "left"; case .right: ch = "right"; default: ch = "unknown" } + return PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: ch) + } + print(encodeJSON(results)) + exit(0) +} + +if isAnimalPose { + if #available(macOS 14.0, *) { + let request = VNDetectAnimalBodyPoseRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision animal pose failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let results = ((request.results ?? [])).map { o in + PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) + } + print(encodeJSON(results)) + exit(0) + } + fputs("ERROR: --animal-pose requires macOS 14 or newer\n", stderr) + exit(2) +} + +// ─── Animals (cat / dog) ───────────────────────────────────────────────────── + +if isAnimals { + let request = VNRecognizeAnimalsRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision animal recognition failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + let results = ((request.results ?? [])).map { o -> AnimalResult in + let b = box(o.boundingBox) + return AnimalResult( + labels: o.labels.map { ClassificationResult(identifier: $0.identifier, confidence: $0.confidence) }, + x: b.0, y: b.1, w: b.2, h: b.3, confidence: o.confidence + ) + } + print(encodeJSON(results)) + exit(0) +} + +// ─── Horizon ───────────────────────────────────────────────────────────────── + +if isHorizon { + let request = VNDetectHorizonRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision horizon detection failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNHorizonObservation else { + print("null") + exit(0) + } + print(encodeJSON(HorizonResult(angleDegrees: Double(o.angle) * 180 / .pi))) + exit(0) +} + +// ─── Contours ──────────────────────────────────────────────────────────────── + +if isContours { + let request = VNDetectContoursRequest() + request.detectsDarkOnLight = !args.contains("--light-on-dark") + if let r = roiRect { request.regionOfInterest = r } + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision contour detection failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNContoursObservation else { + print(encodeJSON(ContoursResult(totalContours: 0, topLevel: []))) + exit(0) + } + let maxPoints = optValue("--max-points").flatMap { Int($0) } ?? 0 + var top: [ContourResult] = [] + for (i, c) in o.topLevelContours.enumerated() { + let bb = unROI(c.normalizedPath.boundingBox) + let b = box(bb) + var pts: [[Double]]? = nil + if maxPoints > 0 { + let all = c.normalizedPoints + let stride = max(1, all.count / maxPoints) + pts = Swift.stride(from: 0, to: all.count, by: stride).map { + let q = unROI(CGRect(x: CGFloat(all[$0].x), y: CGFloat(all[$0].y), width: 0, height: 0)) + return [Double(q.origin.x), 1.0 - Double(q.origin.y)] + } + } + top.append(ContourResult(index: i, pointCount: c.pointCount, childCount: c.childContourCount, + x: b.0, y: b.1, w: b.2, h: b.3, points: pts)) + } + print(encodeJSON(ContoursResult(totalContours: o.contourCount, topLevel: top))) + exit(0) +} + +// ─── Saliency (attention | objectness) ─────────────────────────────────────── + +if isSaliency { + let kind = optValue("--saliency") ?? "attention" + let request: VNImageBasedRequest = kind == "objectness" + ? VNGenerateObjectnessBasedSaliencyImageRequest() + : VNGenerateAttentionBasedSaliencyImageRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision saliency failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNSaliencyImageObservation else { + print(encodeJSON(SaliencyResult(regions: [], heatmapPath: nil))) + exit(0) + } + let regions = (o.salientObjects ?? []).map { r -> HumanResult in + let b = box(r.boundingBox) + return HumanResult(x: b.0, y: b.1, w: b.2, h: b.3, confidence: r.confidence) + } + var heat: String? = nil + if let out = optValue("--out"), let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) { heat = out } + print(encodeJSON(SaliencyResult(regions: regions, heatmapPath: heat))) + exit(0) +} + +// ─── Foreground subject cutout / person mask ───────────────────────────────── + +if isForeground { + if #available(macOS 14.0, *) { + let out = requireOut() + let request = VNGenerateForegroundInstanceMaskRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision foreground mask failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNInstanceMaskObservation else { + print(encodeJSON(MaskResult(instances: 0, outPath: ""))) + exit(0) + } + let maskOnly = args.contains("--mask-only") + do { + let pb = maskOnly + ? try o.generateScaledMaskForImage(forInstances: o.allInstances, from: handler) + : try o.generateMaskedImage(ofInstances: o.allInstances, from: handler, croppedToInstancesExtent: args.contains("--tight")) + guard let cg = cgFromPixelBuffer(pb), writePNG(cg, to: out) else { throw NSError(domain: "vision-helper", code: 1) } + } catch { + fputs("ERROR: could not write foreground image: \(error.localizedDescription)\n", stderr) + exit(1) + } + print(encodeJSON(MaskResult(instances: o.allInstances.count, outPath: out))) + exit(0) + } + fputs("ERROR: --foreground-mask requires macOS 14 or newer\n", stderr) + exit(2) +} + +if isPersonMask { + let out = requireOut() + let request = VNGeneratePersonSegmentationRequest() + request.qualityLevel = .accurate + request.outputPixelFormat = kCVPixelFormatType_OneComponent8 + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision person segmentation failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNPixelBufferObservation, + let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) else { + fputs("ERROR: could not write person mask\n", stderr) + exit(1) + } + print(encodeJSON(MaskResult(instances: 1, outPath: out))) + exit(0) +} + +// ─── Aesthetics (macOS 15+) ────────────────────────────────────────────────── + +if isAesthetics { + if #available(macOS 15.0, *) { + let request = VNCalculateImageAestheticsScoresRequest() + do { try handler.perform([request]) } catch { + fputs("ERROR: Vision aesthetics failed: \(error.localizedDescription)\n", stderr) + exit(1) + } + guard let o = request.results?.first as? VNImageAestheticsScoresObservation else { + print("null") + exit(0) + } + print(encodeJSON(AestheticsResult(overallScore: o.overallScore, isUtility: o.isUtility))) + exit(0) + } + fputs("ERROR: --aesthetics requires macOS 15 or newer\n", stderr) + exit(2) +} + // ─── Faces ─────────────────────────────────────────────────────────────────── if isFaces { diff --git a/src/ui.ts b/src/ui.ts index 2ab3d1e..7aac5d9 100644 --- a/src/ui.ts +++ b/src/ui.ts @@ -7,7 +7,8 @@ import { execFile } from 'child_process'; import { promisify } from 'util'; import { existsSync, mkdirSync } from 'fs'; -import { open } from 'fs/promises'; +import { open, readFile } from 'fs/promises'; +import { createHash } from 'crypto'; import { tmpdir } from 'os'; import { resolve, dirname, join } from 'path'; import { fileURLToPath } from 'url'; @@ -73,6 +74,8 @@ export interface CaptureResult { frame: ScreenFrame; /** pixelWidth / frame.w — ≈2 on Retina */ scale: number; + /** SHA-256 of the PNG bytes — pin assertions to a specific capture, detect replaced files */ + sha256: string; /** ISO timestamp */ capturedAt: string; /** Human-readable description of what was captured */ @@ -217,11 +220,12 @@ export async function captureScreen(opts: CaptureOptions = {}): Promise 0 ? px.w / frame.w : 1, capturedAt: new Date().toISOString(), diff --git a/src/vision.ts b/src/vision.ts new file mode 100644 index 0000000..1d76a4c --- /dev/null +++ b/src/vision.ts @@ -0,0 +1,589 @@ +// Extended Vision API: everything beyond the classic OCR/faces/barcodes set. +// +// All geometry is normalized 0–1 with a TOP-LEFT origin unless stated otherwise. +// Functions that produce pixels (masks, crops, heatmaps) write PNG files and +// return paths — never image bytes. + +import { execFile } from 'child_process'; +import { resolve, dirname, join } from 'path'; +import { fileURLToPath } from 'url'; +import { mkdirSync } from 'fs'; +import { tmpdir } from 'os'; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN_PATH = resolve(__dirname, '../bin/vision-helper'); +const TIMEOUT_MS = 60_000; + +// ─── Shared types ──────────────────────────────────────────────────────────── + +/** Normalized rectangle, 0–1, top-left origin. */ +export interface NormalizedRect { + x: number; + y: number; + width: number; + height: number; +} + +interface RawBox { + x: number; + y: number; + w: number; + h: number; + confidence: number; +} + +function box(b: RawBox): NormalizedRect & { confidence: number } { + return { x: b.x, y: b.y, width: b.w, height: b.h, confidence: b.confidence }; +} + +/** Options shared by every text-recognition path (OCR, text regions, document structure). */ +export interface TextRecognitionOptions { + /** BCP-47 codes in priority order, e.g. `['pl-PL', 'en-US']`. See `supportedOcrLanguages()`. */ + languages?: string[]; + /** Let Vision pick the language per text run. Default false. */ + autoDetectLanguage?: boolean; + /** Apply language-model correction. Default true. Disable for IDs, codes, IBANs, hashes. */ + languageCorrection?: boolean; + /** Domain vocabulary that should win over the language model, e.g. product names. */ + customWords?: string[]; + /** `.fast` recognition level — noticeably quicker, lower accuracy. Default false. */ + fast?: boolean; + /** Only look inside this region (normalized, top-left origin). Results are still reported in full-image space. */ + regionOfInterest?: NormalizedRect; + /** Ignore text shorter than this fraction of image height (0–1). */ + minTextHeight?: number; +} + +/** Serialises shared text options into helper flags. */ +export function textOptionArgs(o: TextRecognitionOptions = {}): string[] { + const args: string[] = []; + if (o.languages?.length) args.push('--lang', o.languages.join(',')); + if (o.autoDetectLanguage) args.push('--auto-lang'); + if (o.languageCorrection === false) args.push('--no-correction'); + if (o.customWords?.length) args.push('--custom-words', o.customWords.join(',')); + if (o.fast) args.push('--fast'); + if (o.regionOfInterest) { + const r = o.regionOfInterest; + args.push('--roi', `${r.x},${r.y},${r.width},${r.height}`); + } + if (o.minTextHeight !== undefined) args.push('--min-text-height', String(o.minTextHeight)); + return args; +} + +async function run(args: string[], input?: string): Promise { + const stdout = await new Promise((resolvePromise, reject) => { + const child = execFile( + BIN_PATH, + args, + { timeout: TIMEOUT_MS, maxBuffer: 64 * 1024 * 1024 }, + (err, out) => { + if (err) { + reject(err); + return; + } + resolvePromise(out); + } + ); + if (input !== undefined) child.stdin?.end(input); + }); + // Some Vision models log to stdout; the JSON payload is always the last line. + const lines = stdout.trim().split('\n'); + return JSON.parse(lines[lines.length - 1]) as T; +} + +function outPath(out: string | undefined, prefix: string): string { + if (out) return resolve(out); + const dir = join(tmpdir(), 'macos-vision'); + mkdirSync(dir, { recursive: true }); + return join(dir, `${prefix}-${Date.now()}-${Math.floor(Math.random() * 1e6)}.png`); +} + +// ─── Capabilities ──────────────────────────────────────────────────────────── + +export interface VisionCapabilities { + /** Version of the bundled `vision-helper` protocol */ + helperVersion: string; + macosVersion: string; + /** BCP-47 codes Vision can OCR on this machine */ + ocrLanguages: string[]; + /** Feature → available on this macOS. Gate agent plans on this. */ + features: Record; +} + +/** What this machine can do. Cheap; cache it per process. */ +export function visionCapabilities(): Promise { + return run(['--capabilities']); +} + +/** BCP-47 codes supported by the accurate OCR model. */ +export function supportedOcrLanguages(): Promise { + return run(['--languages']); +} + +// ─── Image info ────────────────────────────────────────────────────────────── + +export interface ImageInfo { + width: number; + height: number; + hasAlpha: boolean; + bitsPerComponent: number; + colorSpace?: string; + dpi?: number; + /** EXIF orientation 1–8 when present */ + orientation?: number; + /** UTI, e.g. 'public.png', 'public.jpeg' */ + format?: string; +} + +/** Pixel dimensions and metadata without running any model. */ +export function imageInfo(imagePath: string): Promise { + return run(['--image-info', resolve(imagePath)]); +} + +// ─── Text regions (no recognition) ─────────────────────────────────────────── + +export type TextRegion = NormalizedRect & { confidence: number }; + +/** Where text is, without reading it. Much faster than OCR — use to pick regions of interest. */ +export async function detectTextRegions( + imagePath: string, + options: Pick = {} +): Promise { + const raw = await run(['--text-rects', ...textOptionArgs(options), resolve(imagePath)]); + return raw.map(box); +} + +// ─── Image similarity ──────────────────────────────────────────────────────── + +export interface ImageComparison { + /** Feature-print distance. 0 = identical; < ~0.3 visually the same scene; > ~0.8 different content. */ + distance: number; +} + +/** Compare two images semantically (Vision feature prints). Robust to small shifts and compression. */ +export function compareImages(imagePathA: string, imagePathB: string): Promise { + return run(['--compare', resolve(imagePathA), resolve(imagePathB)]); +} + +// ─── Entities in text (NSDataDetector) ─────────────────────────────────────── + +export interface TextEntity { + type: 'link' | 'email' | 'phone' | 'address' | 'date' | 'transit' | 'unknown'; + /** Matched substring */ + text: string; + /** UTF-16 offsets into the input */ + start: number; + end: number; + /** Normalised value: absolute URL, e-mail, phone, ISO-8601 date, "street, city, zip" */ + value?: string; + /** Structured parts (address components, event duration, transit info) */ + components?: Record; +} + +/** Links, e-mails, phones, addresses and dates in plain text. Pure Foundation — no model. */ +export function extractEntities(text: string): Promise { + return run(['--entities'], text); +} + +// ─── Document structure (macOS 26+) ────────────────────────────────────────── + +export interface DocLine { + text: string; + confidence: number; + bbox: DocBox; +} +export interface DocBox { + x: number; + y: number; + w: number; + h: number; +} +export interface DocText { + text: string; + alignment?: 'leading' | 'center' | 'trailing'; + bbox: DocBox; + lines: DocLine[]; +} +export interface DocCell { + text: string; + row: number; + col: number; + rowSpan: number; + colSpan: number; + bbox: DocBox; +} +export interface DocTable { + rowCount: number; + columnCount: number; + /** Cell texts by row; spanning cells repeat in every row they cover */ + rows: string[][]; + /** Unique cells with spans */ + cells: DocCell[]; + bbox: DocBox; +} +export interface DocListItem { + marker: string; + text: string; + bbox: DocBox; +} +export interface DocList { + items: DocListItem[]; + bbox: DocBox; +} +export interface DocDetectedData { + type: + | 'link' + | 'email' + | 'phone' + | 'address' + | 'date' + | 'money' + | 'flight' + | 'tracking' + | 'measurement' + | 'payment' + | 'unknown'; + text: string; + value?: string; + bbox: DocBox; +} +export interface DocBarcode { + type: string; + value: string; + bbox: DocBox; +} + +export interface DocumentStructure { + /** Detected title block, if any */ + title?: DocText; + /** Full transcript in reading order */ + text: string; + paragraphs: DocText[]; + tables: DocTable[]; + lists: DocList[]; + barcodes: DocBarcode[]; + /** Data detectors run on the transcript: links, e-mails, phones, money, dates… with positions */ + detectedData: DocDetectedData[]; +} + +export class UnsupportedOnThisMacOSError extends Error { + constructor(feature: string, minVersion: string) { + super(`${feature} requires macOS ${minVersion} or newer`); + this.name = 'UnsupportedOnThisMacOSError'; + } +} + +/** + * Native document understanding (macOS 26+): paragraphs, tables, lists, title, + * barcodes and detected data with positions — no heuristics, no LLM. + * Throws `UnsupportedOnThisMacOSError` on older systems; check `visionCapabilities().features.documentStructure`. + */ +export async function recognizeDocument( + imagePath: string, + options: TextRecognitionOptions = {} +): Promise { + try { + return await run([ + '--document-structure', + ...textOptionArgs(options), + resolve(imagePath), + ]); + } catch (err) { + if ((err as { code?: number }).code === 2) { + throw new UnsupportedOnThisMacOSError('recognizeDocument', '26'); + } + throw err; + } +} + +// ─── Quality signals ───────────────────────────────────────────────────────── + +export interface LensSmudge { + /** 0–1 likelihood the lens was dirty/smudged */ + confidence: number; + /** False when the smudge model is unavailable on this hardware — treat confidence as unknown */ + supported: boolean; +} + +/** Was the photo taken through a dirty lens? macOS 26+. Returns `supported:false` instead of guessing. */ +export async function detectLensSmudge(imagePath: string): Promise { + try { + return await run(['--smudge', resolve(imagePath)]); + } catch (err) { + if ((err as { code?: number }).code === 2) return { confidence: 0, supported: false }; + throw err; + } +} + +export interface AestheticsScore { + /** -1…1, higher is nicer */ + overallScore: number; + /** True for screenshots, receipts, documents — "utility" images rather than photos */ + isUtility: boolean; +} + +/** Photo aesthetics + utility flag (macOS 15+). Good for "is this a screenshot or a photo?". */ +export async function imageAesthetics(imagePath: string): Promise { + try { + return await run(['--aesthetics', resolve(imagePath)]); + } catch (err) { + if ((err as { code?: number }).code === 2) + throw new UnsupportedOnThisMacOSError('imageAesthetics', '15'); + throw err; + } +} + +export interface Horizon { + /** Tilt in degrees; positive = clockwise. */ + angleDegrees: number; +} + +/** Horizon tilt for photos; null when no horizon is detected. */ +export function detectHorizon(imagePath: string): Promise { + return run(['--horizon', resolve(imagePath)]); +} + +// ─── People, faces, poses, animals ─────────────────────────────────────────── + +export interface FaceLandmarks { + x: number; + y: number; + width: number; + height: number; + confidence: number; + /** Head rotation in degrees, when available */ + roll?: number; + yaw?: number; + pitch?: number; + /** 0–1 sharpness/exposure quality of the face crop */ + captureQuality?: number; + /** Region name → polyline of [x, y] normalized points */ + landmarks: Record; +} + +export async function detectFaceLandmarks(imagePath: string): Promise { + const raw = await run< + Array> + >(['--face-landmarks', resolve(imagePath)]); + return raw.map((f) => ({ + ...box(f), + roll: f.roll, + yaw: f.yaw, + pitch: f.pitch, + captureQuality: f.captureQuality, + landmarks: f.landmarks, + })); +} + +export type HumanBox = NormalizedRect & { confidence: number }; + +/** Full-body person boxes. */ +export async function detectHumans(imagePath: string): Promise { + const raw = await run(['--humans', resolve(imagePath)]); + return raw.map(box); +} + +export interface Keypoint { + x: number; + y: number; + confidence: number; +} + +export interface Pose { + /** Joint name (Vision key, e.g. 'left_wrist_joint') → point. Only joints with confidence > 0. */ + joints: Record; + confidence: number; + /** Hands only */ + chirality?: 'left' | 'right' | 'unknown'; +} + +export function detectBodyPose(imagePath: string): Promise { + return run(['--body-pose', resolve(imagePath)]); +} + +export function detectHandPose(imagePath: string): Promise { + return run(['--hand-pose', resolve(imagePath)]); +} + +/** macOS 14+. */ +export async function detectAnimalPose(imagePath: string): Promise { + try { + return await run(['--animal-pose', resolve(imagePath)]); + } catch (err) { + if ((err as { code?: number }).code === 2) + throw new UnsupportedOnThisMacOSError('detectAnimalPose', '14'); + throw err; + } +} + +export interface Animal { + /** e.g. [{ identifier: 'Cat', confidence: 0.98 }] */ + labels: Array<{ identifier: string; confidence: number }>; + x: number; + y: number; + width: number; + height: number; + confidence: number; +} + +/** Cats and dogs with boxes. */ +export async function detectAnimals(imagePath: string): Promise { + const raw = await run>([ + '--animals', + resolve(imagePath), + ]); + return raw.map((a) => ({ ...box(a), labels: a.labels })); +} + +// ─── Saliency, contours ────────────────────────────────────────────────────── + +export interface SaliencyOptions { + /** 'attention' = where a human would look; 'objectness' = where objects are. Default 'attention'. */ + mode?: 'attention' | 'objectness'; + /** Write the heatmap PNG here (optional). */ + heatmapPath?: string; +} + +export interface Saliency { + /** Up to 3 salient regions, sorted by the model */ + regions: Array; + heatmapPath?: string; +} + +export async function detectSaliency( + imagePath: string, + options: SaliencyOptions = {} +): Promise { + const args = ['--saliency', options.mode ?? 'attention']; + if (options.heatmapPath) args.push('--out', resolve(options.heatmapPath)); + args.push(resolve(imagePath)); + const raw = await run<{ regions: RawBox[]; heatmapPath?: string }>(args); + return { regions: raw.regions.map(box), heatmapPath: raw.heatmapPath ?? undefined }; +} + +export interface ContourOptions { + /** Include up to N evenly-sampled points per top-level contour. Default 0 (boxes only). */ + maxPoints?: number; + /** Detect light shapes on a dark background instead of dark-on-light. */ + lightOnDark?: boolean; + regionOfInterest?: NormalizedRect; +} + +export interface Contour { + index: number; + pointCount: number; + childCount: number; + x: number; + y: number; + width: number; + height: number; + points?: [number, number][]; +} + +export interface Contours { + totalContours: number; + topLevel: Contour[]; +} + +/** Edge/shape contours — useful for charts, diagrams, UI boundaries. */ +export async function detectContours( + imagePath: string, + options: ContourOptions = {} +): Promise { + const args = ['--contours']; + if (options.maxPoints) args.push('--max-points', String(options.maxPoints)); + if (options.lightOnDark) args.push('--light-on-dark'); + args.push(...textOptionArgs({ regionOfInterest: options.regionOfInterest }), resolve(imagePath)); + const raw = await run<{ + totalContours: number; + topLevel: Array & { w: number; h: number }>; + }>(args); + return { + totalContours: raw.totalContours, + topLevel: raw.topLevel.map((c) => ({ + index: c.index, + pointCount: c.pointCount, + childCount: c.childCount, + x: c.x, + y: c.y, + width: c.w, + height: c.h, + points: c.points, + })), + }; +} + +// ─── Pixel-producing operations (write PNG, return path) ───────────────────── + +export interface CropResult { + outPath: string; + width: number; + height: number; +} + +/** Crop a normalized region (top-left origin) to a PNG. Cheap "zoom in" for a second OCR pass. */ +export function cropImage( + imagePath: string, + region: NormalizedRect, + out?: string +): Promise { + return run([ + '--crop', + `${region.x},${region.y},${region.width},${region.height}`, + '--out', + outPath(out, 'crop'), + resolve(imagePath), + ]); +} + +/** Detect the document in a photo and write a perspective-corrected, deskewed PNG. */ +export function cropDocument(imagePath: string, out?: string): Promise { + return run([ + '--document-crop', + '--out', + outPath(out, 'document'), + resolve(imagePath), + ]); +} + +export interface MaskResult { + /** Number of foreground instances found (0 → nothing written) */ + instances: number; + outPath: string; +} + +export interface ForegroundOptions { + /** Write only the alpha mask instead of the masked subject. */ + maskOnly?: boolean; + /** Crop the output to the subject's extent. */ + tight?: boolean; + out?: string; +} + +/** Subject cutout (macOS 14+): transparent-background PNG of the main foreground object(s). */ +export async function extractForeground( + imagePath: string, + options: ForegroundOptions = {} +): Promise { + const args = ['--foreground-mask', '--out', outPath(options.out, 'foreground')]; + if (options.maskOnly) args.push('--mask-only'); + if (options.tight) args.push('--tight'); + args.push(resolve(imagePath)); + try { + return await run(args); + } catch (err) { + if ((err as { code?: number }).code === 2) + throw new UnsupportedOnThisMacOSError('extractForeground', '14'); + throw err; + } +} + +/** Person segmentation mask as an 8-bit grayscale PNG (white = person). */ +export function personMask(imagePath: string, out?: string): Promise { + return run([ + '--person-mask', + '--out', + outPath(out, 'person-mask'), + resolve(imagePath), + ]); +} diff --git a/test/vision.test.ts b/test/vision.test.ts new file mode 100644 index 0000000..1cf6dcc --- /dev/null +++ b/test/vision.test.ts @@ -0,0 +1,214 @@ +import { describe, it, expect } from 'vitest'; +import { existsSync } from 'fs'; +import { resolve, dirname } from 'path'; +import { fileURLToPath } from 'url'; +import { tmpdir } from 'os'; +import { + ocr, + visionCapabilities, + supportedOcrLanguages, + imageInfo, + detectTextRegions, + compareImages, + extractEntities, + recognizeDocument, + detectLensSmudge, + detectHumans, + detectBodyPose, + detectSaliency, + detectContours, + cropImage, + cropDocument, + extractForeground, + UnsupportedOnThisMacOSError, + listDisplays, + checkPermissions, +} from '../src/index.js'; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const SAMPLE_IMG = resolve(__dirname, 'fixtures/sample.png'); +const SAMPLE_PDF = resolve(__dirname, 'fixtures/sample.pdf'); +const T = 30_000; + +describe('visionCapabilities()', () => { + it('reports helper version, macOS version and feature flags', async () => { + const caps = await visionCapabilities(); + expect(caps.helperVersion).toBeTruthy(); + expect(caps.macosVersion).toMatch(/^\d+\.\d+/); + expect(caps.features.ocr).toBe(true); + expect(typeof caps.features.documentStructure).toBe('boolean'); + expect(caps.ocrLanguages.length).toBeGreaterThan(5); + }); + + it('supportedOcrLanguages() includes English', async () => { + expect(await supportedOcrLanguages()).toContain('en-US'); + }); +}); + +describe('ocr() — tuning options', () => { + it('regionOfInterest restricts results but reports full-image coordinates', async () => { + const top = await ocr(SAMPLE_IMG, { + format: 'blocks', + regionOfInterest: { x: 0, y: 0, width: 1, height: 0.1 }, + }); + expect(top.length).toBeGreaterThan(0); + for (const b of top) { + expect(b.y).toBeLessThan(0.12); + expect(b.height).toBeLessThan(0.1); + } + expect(top.map((b) => b.text).join(' ')).toContain('Henry VIII'); + }, T); + + it('languages + languageCorrection:false still reads the fixture', async () => { + const text = await ocr(SAMPLE_IMG, { languages: ['en-US'], languageCorrection: false }); + expect(text).toContain('Henry VIII'); + }, T); + + it('fast mode returns text', async () => { + const text = await ocr(SAMPLE_IMG, { fast: true }); + expect(text).toContain('Henry'); + }, T); + + it('cache:true returns identical result on second call', async () => { + const opts = { cache: true, customWords: ['Wikipedia'] } as const; + const a = await ocr(SAMPLE_IMG, opts); + const started = Date.now(); + const b = await ocr(SAMPLE_IMG, opts); + expect(b).toBe(a); + expect(Date.now() - started).toBeLessThan(200); + }, T); + + it('onProgress fires once per PDF page', async () => { + const calls: Array<[number, number]> = []; + await ocr(SAMPLE_PDF, { onProgress: (d, n) => calls.push([d, n]) }); + expect(calls.length).toBeGreaterThan(0); + expect(calls[calls.length - 1][0]).toBe(calls[calls.length - 1][1]); + }, 60_000); +}); + +describe('imageInfo() / detectTextRegions() / compareImages()', () => { + it('imageInfo reads dimensions without a model', async () => { + const info = await imageInfo(SAMPLE_IMG); + expect(info.width).toBe(1088); + expect(info.height).toBe(1344); + expect(info.format).toBe('public.png'); + }); + + it('detectTextRegions finds many text boxes', async () => { + const regions = await detectTextRegions(SAMPLE_IMG); + expect(regions.length).toBeGreaterThan(20); + for (const r of regions) { + expect(r.x).toBeGreaterThanOrEqual(-0.01); + expect(r.y).toBeGreaterThanOrEqual(-0.01); + expect(r.width).toBeGreaterThan(0); + } + }, T); + + it('compareImages: identical → 0, different → > 0.5', async () => { + expect((await compareImages(SAMPLE_IMG, SAMPLE_IMG)).distance).toBe(0); + const crop = await cropImage(SAMPLE_IMG, { x: 0.6, y: 0.1, width: 0.3, height: 0.3 }); + expect((await compareImages(SAMPLE_IMG, crop.outPath)).distance).toBeGreaterThan(0.3); + }, T); +}); + +describe('extractEntities()', () => { + it('finds email, phone, url and date', async () => { + const ents = await extractEntities( + 'Kontakt: jan@example.com, tel. +48 601 234 567, https://prorok.pl, spotkanie 1 września 2026 o 14:00' + ); + const types = ents.map((e) => e.type); + expect(types).toContain('email'); + expect(types).toContain('phone'); + expect(types).toContain('link'); + expect(types).toContain('date'); + expect(ents.find((e) => e.type === 'email')?.value).toBe('jan@example.com'); + }); +}); + +describe('recognizeDocument()', () => { + it('returns structure on macOS 26+, throws UnsupportedOnThisMacOSError otherwise', async () => { + const caps = await visionCapabilities(); + if (!caps.features.documentStructure) { + await expect(recognizeDocument(SAMPLE_IMG)).rejects.toBeInstanceOf(UnsupportedOnThisMacOSError); + return; + } + const doc = await recognizeDocument(SAMPLE_IMG, { languages: ['en-US'] }); + expect(doc.text).toContain('Henry VIII'); + expect(doc.paragraphs.length).toBeGreaterThan(5); + expect(doc.title?.text).toContain('Henry'); + expect(Array.isArray(doc.tables)).toBe(true); + }, 60_000); + + it('detectLensSmudge never throws on supported systems', async () => { + const caps = await visionCapabilities(); + const r = await detectLensSmudge(SAMPLE_IMG); + expect(typeof r.supported).toBe('boolean'); + if (!caps.features.lensSmudge) expect(r.supported).toBe(false); + }, T); +}); + +describe('people / saliency / contours', () => { + it('detectHumans finds the portrait on the fixture', async () => { + const humans = await detectHumans(SAMPLE_IMG); + expect(humans.length).toBeGreaterThanOrEqual(1); + expect(humans[0].x).toBeGreaterThan(0.5); + }, T); + + it('detectBodyPose returns named joints', async () => { + const poses = await detectBodyPose(SAMPLE_IMG); + expect(poses.length).toBeGreaterThanOrEqual(1); + expect(Object.keys(poses[0].joints).length).toBeGreaterThan(5); + }, T); + + it('detectSaliency returns regions and can write a heatmap', async () => { + const out = resolve(tmpdir(), `macos-vision-test-heat-${Date.now()}.png`); + const s = await detectSaliency(SAMPLE_IMG, { mode: 'objectness', heatmapPath: out }); + expect(s.regions.length).toBeGreaterThan(0); + expect(s.heatmapPath).toBe(out); + expect(existsSync(out)).toBe(true); + }, T); + + it('detectContours counts contours and samples points', async () => { + const c = await detectContours(SAMPLE_IMG, { maxPoints: 4 }); + expect(c.totalContours).toBeGreaterThan(10); + expect(c.topLevel[0].points?.length).toBeLessThanOrEqual(5); + }, T); +}); + +describe('pixel ops return paths, never bytes', () => { + it('cropImage writes a PNG of the requested size', async () => { + const r = await cropImage(SAMPLE_IMG, { x: 0, y: 0, width: 0.5, height: 0.5 }); + expect(existsSync(r.outPath)).toBe(true); + expect(r.width).toBe(544); + expect(r.height).toBe(672); + }); + + it('cropDocument writes a perspective-corrected PNG', async () => { + const r = await cropDocument(SAMPLE_IMG); + expect(existsSync(r.outPath)).toBe(true); + expect(r.width).toBeGreaterThan(100); + }, T); + + it('extractForeground writes a cutout (macOS 14+)', async () => { + const caps = await visionCapabilities(); + if (!caps.features.foregroundMask) return; + const r = await extractForeground(SAMPLE_IMG, { tight: true }); + expect(r.instances).toBeGreaterThanOrEqual(1); + expect(existsSync(r.outPath)).toBe(true); + }, T); +}); + +describe('ui-helper', () => { + it('listDisplays reports at least one display with a scale', async () => { + const d = await listDisplays(); + expect(d.length).toBeGreaterThanOrEqual(1); + expect(d.some((x) => x.isMain)).toBe(true); + expect(d[0].scale).toBeGreaterThanOrEqual(1); + }); + + it('checkPermissions returns booleans', async () => { + const p = await checkPermissions(); + expect(typeof p.screenRecording).toBe('boolean'); + expect(typeof p.accessibility).toBe('boolean'); + }); +}); From 2ce750cb03d0cd01f3228e37c053f88fa536acb0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 07:07:49 +0200 Subject: [PATCH 3/7] refactor: shared helper plumbing, unified Box wire format, dedupe Swift/TS boilerplate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - src/helper.ts: one execHelper/runHelper/runGated for all three binaries (timeouts, buffer, exit-code-2 → UnsupportedOnThisMacOSError), tmpOutPath, sha256/fileSha256 — replaces three divergent runners - vision-helper: fd 1 diverted to stderr process-wide, JSON written to the saved stdout (one payload, no per-mode dup2 hack); perform()/fail() replace 19 copy-pasted catch blocks; Box {x,y,width,height,confidence} unifies Face/TextRect/Human/DocRegion so TS needs no RawBox/DocBox mapping layer; supportedLanguages(), macOS15/26 flags, hoisted ISO8601 formatter - ocr cache key = content hash + canonical textOptionArgs (order-independent); onProgress moved to OcrOptions; ocrPdf typed as Omit - captureScreen reads the PNG once (IHDR + sha256 from one buffer) - CLI: top-level await early exit for --capabilities/--languages (one spawn), single memoized OCR pass shared by --ocr/--entities, CLASSIC_FLAGS table - visionCapabilities() memoized per process; ui-helper hoists --all check Co-Authored-By: Claude Fable 5 --- src/cli.ts | 57 +++-- src/helper.ts | 92 ++++++++ src/index.ts | 125 +++------- src/native/ui-helper.swift | 3 +- src/native/vision-helper.swift | 406 ++++++++++++--------------------- src/ui.ts | 61 ++--- src/vision.ts | 196 +++++----------- 7 files changed, 379 insertions(+), 561 deletions(-) create mode 100644 src/helper.ts diff --git a/src/cli.ts b/src/cli.ts index 285a287..3f0500b 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -24,7 +24,6 @@ import { imageAesthetics, imageInfo, visionCapabilities, - supportedOcrLanguages, } from './index.js'; import type { TextRecognitionOptions } from './index.js'; @@ -144,32 +143,26 @@ if (roiRaw) { } textOptions.regionOfInterest = { x, y, width, height }; } -const ocrBase = { ...pageRange, ...textOptions, ...(flags.has('--cache') ? { cache: true } : {}) }; +const ocrBase = { ...pageRange, ...textOptions, cache: flags.has('--cache') }; -const metaOnly = flags.has('--capabilities') || flags.has('--languages'); +// Commands that need no input file. +if (flags.has('--capabilities') || flags.has('--languages')) { + const caps = await visionCapabilities(); + if (flags.has('--capabilities')) console.log(JSON.stringify(caps, null, 2)); + if (flags.has('--languages')) console.log(JSON.stringify(caps.ocrLanguages, null, 2)); + process.exit(0); +} -if (metaOnly) { - (async () => { - if (flags.has('--capabilities')) - console.log(JSON.stringify(await visionCapabilities(), null, 2)); - if (flags.has('--languages')) - console.log(JSON.stringify(await supportedOcrLanguages(), null, 2)); - })().catch((err) => { - console.error(err instanceof Error ? err.message : String(err)); - process.exit(1); - }); -} else if (!fileArgs[0]) { +if (!fileArgs[0]) { console.error('Error: no image or PDF path provided.\n'); console.log(USAGE); process.exit(1); } -const inputPath = fileArgs[0] ? resolve(fileArgs[0]) : ''; +const inputPath = resolve(fileArgs[0]); // ─── Markdown pipeline ───────────────────────────────────────────────────────────── -if (metaOnly) { - // already handled above -} else if (flags.has('--markdown')) { +if (flags.has('--markdown')) { const toStdout = flags.has('--stdout'); const opts: { model?: string; ollamaUrl?: string } = {}; if (model) opts.model = model; @@ -216,9 +209,13 @@ if (metaOnly) { const runRects = runAll || flags.has('--rectangles'); const runDoc = runAll || flags.has('--document'); const runClassify = runAll || flags.has('--classify'); + // One OCR pass shared by --ocr and --entities. + let textPromise: Promise | undefined; + const getText = () => (textPromise ??= ocr(inputPath, ocrBase) as Promise); + const extended: Array<[string, () => Promise]> = [ ['--structure', () => recognizeDocument(inputPath, textOptions)], - ['--entities', async () => extractEntities((await ocr(inputPath, ocrBase)) as string)], + ['--entities', async () => extractEntities(await getText())], ['--text-regions', () => detectTextRegions(inputPath, textOptions)], ['--humans', () => detectHumans(inputPath)], ['--face-landmarks', () => detectFaceLandmarks(inputPath)], @@ -229,24 +226,24 @@ if (metaOnly) { const runExtended = extended.filter(([flag]) => flags.has(flag)); // Default: OCR text when no feature flag is given + const CLASSIC_FLAGS = [ + '--ocr', + '--blocks', + '--faces', + '--barcodes', + '--rectangles', + '--document', + '--classify', + ]; const anyFeatureFlag = - runAll || - flags.has('--ocr') || - flags.has('--blocks') || - flags.has('--faces') || - flags.has('--barcodes') || - flags.has('--rectangles') || - flags.has('--document') || - flags.has('--classify') || - runExtended.length > 0; + runAll || CLASSIC_FLAGS.some((f) => flags.has(f)) || runExtended.length > 0; const useDefault = !anyFeatureFlag; (async () => { try { if (useDefault || runOcr) { - const text = await ocr(inputPath, ocrBase); - console.log(text as string); + console.log(await getText()); } if (runBlocks) { diff --git a/src/helper.ts b/src/helper.ts new file mode 100644 index 0000000..f45b592 --- /dev/null +++ b/src/helper.ts @@ -0,0 +1,92 @@ +// Shared plumbing for the native helpers (vision-helper, pdf-helper, ui-helper). +// +// Wire contract: a helper writes exactly one payload to stdout (the Swift side +// diverts framework noise to stderr), exits 1 on failure and 2 when a feature is +// not available on this macOS. + +import { execFile } from 'child_process'; +import { createHash } from 'crypto'; +import { mkdirSync } from 'fs'; +import { readFile } from 'fs/promises'; +import { tmpdir } from 'os'; +import { dirname, join, resolve } from 'path'; +import { fileURLToPath } from 'url'; + +const BIN_DIR = resolve(dirname(fileURLToPath(import.meta.url)), '../bin'); + +export const VISION_BIN = join(BIN_DIR, 'vision-helper'); +export const PDF_BIN = join(BIN_DIR, 'pdf-helper'); +export const UI_BIN = join(BIN_DIR, 'ui-helper'); + +/** Helper exit status meaning "not supported on this macOS". */ +export const EXIT_UNSUPPORTED = 2; + +export class UnsupportedOnThisMacOSError extends Error { + constructor(feature: string, minVersion: string) { + super(`${feature} requires macOS ${minVersion} or newer`); + this.name = 'UnsupportedOnThisMacOSError'; + } +} + +export interface ExecOptions { + timeout?: number; + /** Written to the helper's stdin, then closed. */ + input?: string; +} + +/** Spawn a helper and return its stdout. */ +export function execHelper(bin: string, args: string[], opts: ExecOptions = {}): Promise { + return new Promise((resolvePromise, reject) => { + const child = execFile( + bin, + args, + { timeout: opts.timeout ?? 30_000, maxBuffer: 64 * 1024 * 1024 }, + (err, stdout) => (err ? reject(err) : resolvePromise(stdout)) + ); + if (opts.input !== undefined) child.stdin?.end(opts.input); + }); +} + +/** Spawn a helper and parse its JSON payload. */ +export async function runHelper( + bin: string, + args: string[], + opts: ExecOptions = {} +): Promise { + return JSON.parse(await execHelper(bin, args, opts)) as T; +} + +/** Like `runHelper`, but translates the "unsupported" exit status into a typed error. */ +export async function runGated( + bin: string, + args: string[], + feature: string, + minVersion: string, + opts: ExecOptions = {} +): Promise { + try { + return await runHelper(bin, args, opts); + } catch (err) { + if ((err as { code?: number }).code === EXIT_UNSUPPORTED) { + throw new UnsupportedOnThisMacOSError(feature, minVersion); + } + throw err; + } +} + +/** `override` resolved, or a fresh path in `$TMPDIR/macos-vision/`. */ +export function tmpOutPath(prefix: string, override?: string, ext = 'png'): string { + if (override) return resolve(override); + const dir = join(tmpdir(), 'macos-vision'); + mkdirSync(dir, { recursive: true }); + return join(dir, `${prefix}-${Date.now()}-${Math.floor(Math.random() * 1e6)}.${ext}`); +} + +export function sha256(bytes: Buffer | string): string { + return createHash('sha256').update(bytes).digest('hex'); +} + +/** SHA-256 of a file's bytes, hex. */ +export async function fileSha256(filePath: string): Promise { + return sha256(await readFile(filePath)); +} diff --git a/src/index.ts b/src/index.ts index 89069ff..7a80aba 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,27 +1,17 @@ -import { execFile } from 'child_process'; -import { promisify } from 'util'; -import { resolve, dirname, extname, dirname as pathDirname } from 'path'; -import { fileURLToPath } from 'url'; +import { resolve, dirname, extname } from 'path'; import { open, readFile, writeFile, mkdir } from 'fs/promises'; -import { createHash } from 'crypto'; import { homedir } from 'os'; +import { VISION_BIN, PDF_BIN, execHelper, runHelper, fileSha256, sha256 } from './helper.js'; import { textOptionArgs } from './vision.js'; import type { TextRecognitionOptions } from './vision.js'; -const execFileAsync = promisify(execFile); -const __dirname = dirname(fileURLToPath(import.meta.url)); -const BIN_PATH = resolve(__dirname, '../bin/vision-helper'); -const PDF_BIN_PATH = resolve(__dirname, '../bin/pdf-helper'); const BINARY_TIMEOUT_MS = 30_000; const PDF_RASTERIZE_TIMEOUT_MS = 120_000; const OCR_CACHE_DIR = resolve(homedir(), '.cache', 'macos-vision', 'ocr'); -async function run(flag: string, imagePath: string): Promise { - const { stdout } = await execFileAsync(BIN_PATH, [flag, resolve(imagePath)], { - timeout: BINARY_TIMEOUT_MS, - }); - return stdout; -} +const run = (args: string[]) => runHelper(VISION_BIN, args, { timeout: BINARY_TIMEOUT_MS }); + +export { fileSha256 }; // ─── PDF helpers ───────────────────────────────────────────────────── @@ -62,8 +52,6 @@ export interface PdfPageRangeOptions { startPage?: number; /** Maximum number of pages to process. Default: all pages from `startPage`. Ignored for non-PDF inputs. */ maxPages?: number; - /** Called after each page is OCR'd (PDF inputs only). `done` counts from 1. */ - onProgress?: (done: number, total: number) => void; } function buildPdfArgs(absPath: string, options: PdfPageRangeOptions): string[] { @@ -99,11 +87,8 @@ export async function rasterizePdf( ): Promise { const absPath = resolve(pdfPath); const args = buildPdfArgs(absPath, options); - const { stdout } = await execFileAsync(PDF_BIN_PATH, args, { - timeout: PDF_RASTERIZE_TIMEOUT_MS, - }); - const pages: PdfPage[] = JSON.parse(stdout); - const cacheDir = pages.length > 0 ? pathDirname(pages[0].path) : ''; + const pages = await runHelper(PDF_BIN, args, { timeout: PDF_RASTERIZE_TIMEOUT_MS }); + const cacheDir = pages.length > 0 ? dirname(pages[0].path) : ''; return { pages, cacheDir }; } @@ -114,10 +99,9 @@ export async function rasterizePdf( async function ocrPdf( pdfPath: string, format: 'text' | 'blocks', - options: OcrOptions = {} + options: Omit = {} ): Promise { - // eslint-disable-next-line @typescript-eslint/no-unused-vars - const { startPage, maxPages, onProgress, format: _format, ...textOptions } = options; + const { startPage, maxPages, onProgress, ...textOptions } = options; const { pages } = await rasterizePdf(pdfPath, { startPage, maxPages }); if (format === 'blocks') { const all: VisionBlock[] = []; @@ -138,25 +122,13 @@ async function ocrPdf( // ─── OCR result cache ──────────────────────────────────────────────────────── -/** SHA-256 of a file's bytes, hex. */ -export async function fileSha256(filePath: string): Promise { - return createHash('sha256') - .update(await readFile(filePath)) - .digest('hex'); -} - +/** Content hash + the canonical helper argv, so option order does not matter. */ async function cacheKey( absPath: string, format: string, opts: TextRecognitionOptions ): Promise { - const content = await fileSha256(absPath); - const optsKey = JSON.stringify({ - format, - ...opts, - regionOfInterest: opts.regionOfInterest ?? null, - }); - return createHash('sha256').update(content).update(optsKey).digest('hex'); + return sha256((await fileSha256(absPath)) + JSON.stringify([format, ...textOptionArgs(opts)])); } async function readCache(key: string): Promise { @@ -198,6 +170,8 @@ export interface VisionBlock { export interface OcrOptions extends PdfPageRangeOptions, TextRecognitionOptions { /** Return plain text (default) or structured blocks with coordinates */ format?: 'text' | 'blocks'; + /** Called after each page is OCR'd (PDF inputs only). `done` counts from 1. */ + onProgress?: (done: number, total: number) => void; /** * Cache results in `~/.cache/macos-vision/ocr/` keyed by file content hash + options. * Repeated OCR of the same bytes (e.g. an agent re-reading a screenshot) returns instantly. @@ -241,18 +215,9 @@ export async function ocr( const textArgs = textOptionArgs(textOptions); if (format === 'blocks') { - const { stdout } = await execFileAsync(BIN_PATH, ['--json', ...textArgs, absPath], { - timeout: BINARY_TIMEOUT_MS, - maxBuffer: 64 * 1024 * 1024, - }); - const raw: Array<{ - t: string; - x: number; - y: number; - w: number; - h: number; - confidence: number; - }> = JSON.parse(stdout); + const raw = await run< + Array<{ t: string; x: number; y: number; w: number; h: number; confidence: number }> + >(['--json', ...textArgs, absPath]); const blocks = raw.map((b) => ({ text: b.t, x: b.x, @@ -265,11 +230,9 @@ export async function ocr( return blocks; } - const { stdout } = await execFileAsync(BIN_PATH, [...textArgs, absPath], { - timeout: BINARY_TIMEOUT_MS, - maxBuffer: 64 * 1024 * 1024, - }); - const text = stdout.trim(); + const text = ( + await execHelper(VISION_BIN, [...textArgs, absPath], { timeout: BINARY_TIMEOUT_MS }) + ).trim(); if (key) await writeCache(key, text); return text; } @@ -289,11 +252,8 @@ export interface Face { confidence: number; } -export async function detectFaces(imagePath: string): Promise { - const raw: Array<{ x: number; y: number; w: number; h: number; confidence: number }> = JSON.parse( - await run('--faces', imagePath) - ); - return raw.map((f) => ({ x: f.x, y: f.y, width: f.w, height: f.h, confidence: f.confidence })); +export function detectFaces(imagePath: string): Promise { + return run(['--faces', resolve(imagePath)]); } // ─── Barcode / QR detection ────────────────────────────────────────────────── @@ -315,25 +275,8 @@ export interface Barcode { confidence: number; } -export async function detectBarcodes(imagePath: string): Promise { - const raw: Array<{ - type: string; - value: string; - x: number; - y: number; - w: number; - h: number; - confidence: number; - }> = JSON.parse(await run('--barcodes', imagePath)); - return raw.map((b) => ({ - type: b.type, - value: b.value, - x: b.x, - y: b.y, - width: b.w, - height: b.h, - confidence: b.confidence, - })); +export function detectBarcodes(imagePath: string): Promise { + return run(['--barcodes', resolve(imagePath)]); } // ─── Rectangle detection ─────────────────────────────────────────────────── @@ -351,15 +294,8 @@ export interface Rectangle { confidence: number; } -export async function detectRectangles(imagePath: string): Promise { - const raw: Array<{ - topLeft: [number, number]; - topRight: [number, number]; - bottomLeft: [number, number]; - bottomRight: [number, number]; - confidence: number; - }> = JSON.parse(await run('--rectangles', imagePath)); - return raw; +export function detectRectangles(imagePath: string): Promise { + return run(['--rectangles', resolve(imagePath)]); } // ─── Document detection ────────────────────────────────────────────────────── @@ -379,8 +315,8 @@ export interface DocumentBounds { /** Returns the detected document boundary, or null if no document found. */ export async function detectDocument(imagePath: string): Promise { - const raw: DocumentBounds[] = JSON.parse(await run('--document', imagePath)); - return raw.length > 0 ? raw[0] : null; + const raw = await run(['--document', resolve(imagePath)]); + return raw[0] ?? null; } // ─── Image classification ───────────────────────────────────────────────────── @@ -393,9 +329,8 @@ export interface Classification { } /** Returns top image classifications sorted by confidence (highest first). */ -export async function classify(imagePath: string): Promise { - const raw: Classification[] = JSON.parse(await run('--classify', imagePath)); - return raw; +export function classify(imagePath: string): Promise { + return run(['--classify', resolve(imagePath)]); } // ─── Layout inference ──────────────────────────────────────────────────────────── @@ -456,6 +391,7 @@ export { } from './vision.js'; export type { NormalizedRect, + Detection, TextRecognitionOptions, VisionCapabilities, ImageInfo, @@ -465,7 +401,6 @@ export type { DocumentStructure, DocText, DocLine, - DocBox, DocTable, DocCell, DocList, diff --git a/src/native/ui-helper.swift b/src/native/ui-helper.swift index 1e75792..98e462d 100644 --- a/src/native/ui-helper.swift +++ b/src/native/ui-helper.swift @@ -83,12 +83,13 @@ if args.contains("--windows") { print("[]") exit(0) } + let includeAll = args.contains("--all") var results: [WindowInfo] = [] for w in list { guard let boundsDict = w[kCGWindowBounds as String] as? [String: Double] else { continue } let layer = w[kCGWindowLayer as String] as? Int ?? 0 // Layer 0 = normal app windows; skip menu bar, dock, overlays unless --all. - if layer != 0 && !args.contains("--all") { continue } + if layer != 0 && !includeAll { continue } let width = boundsDict["Width"] ?? 0 let height = boundsDict["Height"] ?? 0 if width < 40 || height < 40 { continue } // status items, tooltips diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index af9937c..cbc3397 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -14,15 +14,21 @@ struct OCRResult: Codable { let confidence: Float } -struct FaceResult: Codable { - let x: Double; let y: Double; let w: Double; let h: Double +/// Normalized box, top-left origin. Wire keys match the public TS `NormalizedRect`. +struct Box: Codable { + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float + init(_ r: CGRect, _ confidence: Float) { + x = Double(r.origin.x); y = flipY(Double(r.origin.y), Double(r.size.height)) + width = Double(r.size.width); height = Double(r.size.height) + self.confidence = confidence + } } struct BarcodeResult: Codable { let type: String let value: String - let x: Double; let y: Double; let w: Double; let h: Double + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float } @@ -56,11 +62,6 @@ func encodeJSON(_ value: T) -> String { return str } -struct TextRectResult: Codable { - let x: Double; let y: Double; let w: Double; let h: Double - let confidence: Float -} - struct CompareResult: Codable { let distance: Double } @@ -87,7 +88,7 @@ struct Capabilities: Codable { } // Document structure (macOS 26+, RecognizeDocumentsRequest) -struct DocRegion: Codable { let x: Double; let y: Double; let w: Double; let h: Double } +struct DocRegion: Codable { let x: Double; let y: Double; let width: Double; let height: Double } struct DocLine: Codable { let text: String; let confidence: Float; let bbox: DocRegion } struct DocText: Codable { let text: String @@ -123,28 +124,27 @@ struct DocStructure: Codable { struct PointResult: Codable { let x: Double; let y: Double; let confidence: Float } struct FaceLandmarksResult: Codable { - let x: Double; let y: Double; let w: Double; let h: Double + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float let roll: Double?; let yaw: Double?; let pitch: Double? let captureQuality: Float? let landmarks: [String: [[Double]]] } -struct HumanResult: Codable { let x: Double; let y: Double; let w: Double; let h: Double; let confidence: Float } struct PoseResult: Codable { let joints: [String: PointResult]; let confidence: Float; let chirality: String? } struct AnimalResult: Codable { let labels: [ClassificationResult] - let x: Double; let y: Double; let w: Double; let h: Double + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float } struct HorizonResult: Codable { let angleDegrees: Double } struct ContourResult: Codable { let index: Int; let pointCount: Int; let childCount: Int - let x: Double; let y: Double; let w: Double; let h: Double + let x: Double; let y: Double; let width: Double; let height: Double let points: [[Double]]? } struct ContoursResult: Codable { let totalContours: Int; let topLevel: [ContourResult] } struct SaliencyResult: Codable { - let regions: [HumanResult] + let regions: [Box] let heatmapPath: String? } struct MaskResult: Codable { let instances: Int; let outPath: String } @@ -158,19 +158,46 @@ struct ImageInfoResult: Codable { let HELPER_VERSION = "2" +// VisionCore logs model-loading noise straight to fd 1 in some modes. Point fd 1 at +// stderr for the whole process and write our JSON to the original stdout instead, +// so callers always get exactly one clean payload. +let resultOut: FileHandle = { + let saved = dup(STDOUT_FILENO) + dup2(STDERR_FILENO, STDOUT_FILENO) + return FileHandle(fileDescriptor: saved, closeOnDealloc: false) +}() + +func emit(_ text: String) { + resultOut.write((text + "\n").data(using: .utf8)!) +} + func macOSVersionString() -> String { let v = ProcessInfo.processInfo.operatingSystemVersion return "\(v.majorVersion).\(v.minorVersion).\(v.patchVersion)" } -func hasAestheticsAPI() -> Bool { - if #available(macOS 15.0, *) { return true } - return false +let macOS15 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 15, minorVersion: 0, patchVersion: 0)) +let macOS26 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 26, minorVersion: 0, patchVersion: 0)) +let iso8601 = ISO8601DateFormatter() + +func supportedLanguages() -> [String] { + let req = VNRecognizeTextRequest() + req.recognitionLevel = .accurate + return (try? req.supportedRecognitionLanguages()) ?? [] } -func hasDocumentAPI() -> Bool { - if #available(macOS 26.0, *) { return true } - return false +/// Run Vision requests against the shared handler; any failure is fatal for a CLI. +func perform(_ requests: [VNRequest], _ what: String) { + do { + try handler.perform(requests) + } catch { + fail("Vision \(what) failed: \(error.localizedDescription)") + } +} + +func fail(_ message: String, code: Int32 = 1) -> Never { + fputs("ERROR: \(message)\n", stderr) + exit(code) } // ─── Argument parsing ───────────────────────────────────────────────────────── @@ -222,33 +249,27 @@ let fileArgs = args.enumerated() // ─── Modes that need no image ─────────────────────────────────────────────── if args.contains("--languages") { - let req = VNRecognizeTextRequest() - req.recognitionLevel = .accurate - let langs = (try? req.supportedRecognitionLanguages()) ?? [] - print(encodeJSON(langs)) + emit(encodeJSON(supportedLanguages())) exit(0) } if args.contains("--capabilities") { - let req = VNRecognizeTextRequest() - req.recognitionLevel = .accurate - let langs = (try? req.supportedRecognitionLanguages()) ?? [] let caps = Capabilities( helperVersion: HELPER_VERSION, macosVersion: macOSVersionString(), - ocrLanguages: langs, + ocrLanguages: supportedLanguages(), features: [ "ocr": true, "ocrOptions": true, "faces": true, "barcodes": true, "rectangles": true, "document": true, "classify": true, "textRects": true, "compare": true, "entities": true, - "documentStructure": hasDocumentAPI(), "lensSmudge": hasDocumentAPI(), + "documentStructure": macOS26, "lensSmudge": macOS26, "faceLandmarks": true, "humans": true, "bodyPose": true, "handPose": true, "animals": true, "animalPose": true, "horizon": true, "contours": true, "saliency": true, "foregroundMask": true, "personMask": true, - "aesthetics": hasAestheticsAPI(), "documentCrop": true, "crop": true, "imageInfo": true, + "aesthetics": macOS15, "documentCrop": true, "crop": true, "imageInfo": true, ] ) - print(encodeJSON(caps)) + emit(encodeJSON(caps)) exit(0) } @@ -258,12 +279,10 @@ if isEntities { let text = String(decoding: data, as: UTF8.self) let types: NSTextCheckingResult.CheckingType = [.link, .phoneNumber, .address, .date, .transitInformation] guard let detector = try? NSDataDetector(types: types.rawValue) else { - fputs("ERROR: NSDataDetector unavailable\n", stderr) - exit(1) + fail("NSDataDetector unavailable") } let ns = text as NSString var results: [EntityResult] = [] - let iso = ISO8601DateFormatter() for m in detector.matches(in: text, options: [], range: NSRange(location: 0, length: ns.length)) { let matched = ns.substring(with: m.range) var type = "unknown" @@ -284,7 +303,7 @@ if isEntities { value = c[.street].map { s in [s, c[.city], c[.zip]].compactMap { $0 }.joined(separator: ", ") } } case .date: - type = "date"; value = m.date.map { iso.string(from: $0) } + type = "date"; value = m.date.map { iso8601.string(from: $0) } if m.duration > 0 { comps = ["durationSeconds": String(Int(m.duration))] } case .transitInformation: type = "transit" @@ -294,14 +313,14 @@ if isEntities { results.append(EntityResult(type: type, text: matched, start: m.range.location, end: m.range.location + m.range.length, value: value, components: comps)) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } guard let imagePath = fileArgs.first else { - print("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify|--text-rects|--document-structure|--smudge|--compare ] ") - print(" vision-helper --languages | --capabilities | --entities < text.txt") - print("OCR options: --lang pl,en --auto-lang --no-correction --custom-words a,b --fast --roi x,y,w,h --min-text-height f") + emit("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify|--text-rects|--document-structure|--smudge|--compare ] ") + emit(" vision-helper --languages | --capabilities | --entities < text.txt") + emit("OCR options: --lang pl,en --auto-lang --no-correction --custom-words a,b --fast --roi x,y,w,h --min-text-height f") exit(0) } @@ -309,14 +328,12 @@ guard let imagePath = fileArgs.first else { if isCompare { guard fileArgs.count >= 2 else { - fputs("ERROR: --compare needs two image paths\n", stderr) - exit(1) + fail("--compare needs two image paths") } func featurePrint(_ path: String) -> VNFeaturePrintObservation? { guard let img = NSImage(contentsOf: URL(fileURLWithPath: path)), let cg = img.cgImage(forProposedRect: nil, context: nil, hints: nil) else { - fputs("ERROR: Cannot open file: \(path)\n", stderr) - exit(1) + fail("Cannot open file: \(path)") } let req = VNGenerateImageFeaturePrintRequest() let h = VNImageRequestHandler(cgImage: cg, options: [:]) @@ -324,22 +341,19 @@ if isCompare { return req.results?.first as? VNFeaturePrintObservation } guard let a = featurePrint(fileArgs[0]), let b = featurePrint(fileArgs[1]) else { - fputs("ERROR: Vision feature print failed\n", stderr) - exit(1) + fail("Vision feature print failed") } var distance: Float = 0 do { try a.computeDistance(&distance, to: b) } catch { - fputs("ERROR: Vision feature print distance failed: \(error.localizedDescription)\n", stderr) - exit(1) + fail("Vision feature print distance failed: \(error.localizedDescription)") } - print(encodeJSON(CompareResult(distance: Double(distance)))) + emit(encodeJSON(CompareResult(distance: Double(distance)))) exit(0) } guard let image = NSImage(contentsOf: URL(fileURLWithPath: imagePath)), let cgImage = image.cgImage(forProposedRect: nil, context: nil, hints: nil) else { - fputs("ERROR: Cannot open file: \(imagePath)\n", stderr) - exit(1) + fail("Cannot open file: \(imagePath)") } let handler = VNImageRequestHandler(cgImage: cgImage, options: [:]) @@ -362,8 +376,7 @@ if let roi = optValue("--roi") { if p.count == 4 { roiRect = CGRect(x: p[0], y: 1.0 - p[1] - p[3], width: p[2], height: p[3]) } else { - fputs("ERROR: --roi expects x,y,w,h (normalized 0-1)\n", stderr) - exit(1) + fail("--roi expects x,y,w,h (normalized 0-1)") } } @@ -408,41 +421,22 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is if let mh = ocrMinHeight { request.minimumTextHeight = mh } if let r = roiRect { request.regionOfInterest = r } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision OCR failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(isJsonMode ? encodeJSON(ocrResults) : rawText.trimmingCharacters(in: .whitespacesAndNewlines)) + perform([request], "OCR") + emit(isJsonMode ? encodeJSON(ocrResults) : rawText.trimmingCharacters(in: .whitespacesAndNewlines)) exit(0) } // ─── Text rectangles (fast "where is text" without recognition) ────────────── if isTextRects { - var results: [TextRectResult] = [] + var results: [Box] = [] let request = VNDetectTextRectanglesRequest { (req, _) in guard let obs = req.results as? [VNTextObservation] else { return } - for o in obs { - let box = unROI(o.boundingBox) - results.append(TextRectResult( - x: Double(box.origin.x), - y: flipY(Double(box.origin.y), Double(box.size.height)), - w: Double(box.size.width), - h: Double(box.size.height), - confidence: o.confidence - )) - } + results = obs.map { Box(unROI($0.boundingBox), $0.confidence) } } if let r = roiRect { request.regionOfInterest = r } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision text rectangle detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "text rectangle detection") + emit(encodeJSON(results)) exit(0) } @@ -452,7 +446,7 @@ if isTextRects { func region(_ r: NormalizedRegion) -> DocRegion { let b = r.boundingBox return DocRegion(x: Double(b.origin.x), y: flipY(Double(b.origin.y), Double(b.height)), - w: Double(b.width), h: Double(b.height)) + width: Double(b.width), height: Double(b.height)) } @available(macOS 26.0, *) @@ -481,7 +475,7 @@ func docData(_ d: DocumentObservation.Container.DataDetectorMatch, in text: Stri case .emailAddress(let e): type = "email"; value = e.emailAddress case .phoneNumber(let p): type = "phone"; value = p.phoneNumber case .postalAddress(let a): type = "address"; value = a.fullAddress - case .calendarEvent(let c): type = "date"; value = c.startDate.map { ISO8601DateFormatter().string(from: $0) } + case .calendarEvent(let c): type = "date"; value = c.startDate.map { iso8601.string(from: $0) } case .moneyAmount(let m): type = "money"; value = "\(m.amount) \(m.currency.identifier)" case .flightNumber(let f): type = "flight"; value = "\(f.airlineCode)\(f.flightNumber)" case .shipmentTrackingNumber(let s): type = "tracking"; value = s.trackingNumber @@ -564,8 +558,7 @@ func runDocumentStructure() -> DocStructure? { } sema.wait() if let e = failure { - fputs("ERROR: Vision document recognition failed: \(e.localizedDescription)\n", stderr) - exit(1) + fail("Vision document recognition failed: \(e.localizedDescription)") } return result } @@ -594,30 +587,25 @@ func runSmudge() -> SmudgeResult { close(savedStdout) let noise = (try? String(contentsOfFile: tmp, encoding: .utf8)) ?? "" unlink(tmp) - if let e = failure { - fputs("ERROR: Vision lens smudge detection failed: \(e.localizedDescription)\n", stderr) - exit(1) - } + if let e = failure { fail("Vision lens smudge detection failed: \(e.localizedDescription)") } let supported = !noise.contains("Unable to find") return SmudgeResult(confidence: conf, supported: supported) } if isStructure { if #available(macOS 26.0, *) { - if let s = runDocumentStructure() { print(encodeJSON(s)) } + if let s = runDocumentStructure() { emit(encodeJSON(s)) } exit(0) } - fputs("ERROR: --document-structure requires macOS 26 or newer\n", stderr) - exit(2) + fail("--document-structure requires macOS 26 or newer", code: 2) } if isSmudge { if #available(macOS 26.0, *) { - print(encodeJSON(runSmudge())) + emit(encodeJSON(runSmudge())) exit(0) } - fputs("ERROR: --smudge requires macOS 26 or newer\n", stderr) - exit(2) + fail("--smudge requires macOS 26 or newer", code: 2) } @@ -639,16 +627,11 @@ func cgFromPixelBuffer(_ pb: CVPixelBuffer) -> CGImage? { func requireOut() -> String { guard let out = optValue("--out") else { - fputs("ERROR: this mode requires --out \n", stderr) - exit(1) + fail("this mode requires --out ") } return out } -func box(_ r: CGRect) -> (Double, Double, Double, Double) { - (Double(r.origin.x), flipY(Double(r.origin.y), Double(r.size.height)), Double(r.size.width), Double(r.size.height)) -} - func jointsDict(_ points: [VNRecognizedPointKey: VNRecognizedPoint]) -> [String: PointResult] { var d: [String: PointResult] = [:] for (k, p) in points where p.confidence > 0 { @@ -675,7 +658,7 @@ if isImageInfo { colorSpace: cgImage.colorSpace?.name as String?, dpi: dpi, orientation: orientation, format: format ) - print(encodeJSON(info)) + emit(encodeJSON(info)) exit(0) } @@ -685,16 +668,14 @@ if isCrop { let out = requireOut() let p = optValue("--crop")?.split(separator: ",").compactMap { Double($0) } ?? [] guard p.count == 4 else { - fputs("ERROR: --crop expects x,y,w,h (normalized 0-1)\n", stderr) - exit(1) + fail("--crop expects x,y,w,h (normalized 0-1)") } let W = Double(cgImage.width), H = Double(cgImage.height) let rect = CGRect(x: p[0] * W, y: p[1] * H, width: p[2] * W, height: p[3] * H).integral guard let cropped = cgImage.cropping(to: rect), writePNG(cropped, to: out) else { - fputs("ERROR: crop failed\n", stderr) - exit(1) + fail("crop failed") } - print(encodeJSON(CropResult(outPath: out, width: cropped.width, height: cropped.height))) + emit(encodeJSON(CropResult(outPath: out, width: cropped.width, height: cropped.height))) exit(0) } @@ -705,8 +686,7 @@ if isDocCrop { let request = VNDetectDocumentSegmentationRequest() try? handler.perform([request]) guard let o = request.results?.first as? VNRectangleObservation else { - fputs("ERROR: no document detected\n", stderr) - exit(1) + fail("no document detected") } let ci = CIImage(cgImage: cgImage) let W = ci.extent.width, H = ci.extent.height @@ -720,10 +700,9 @@ if isDocCrop { guard let outCI = filter.outputImage, let outCG = ciContext.createCGImage(outCI, from: outCI.extent), writePNG(outCG, to: out) else { - fputs("ERROR: perspective correction failed\n", stderr) - exit(1) + fail("perspective correction failed") } - print(encodeJSON(CropResult(outPath: out, width: outCG.width, height: outCG.height))) + emit(encodeJSON(CropResult(outPath: out, width: outCG.width, height: outCG.height))) exit(0) } @@ -732,15 +711,12 @@ if isDocCrop { if isLandmarks { let lm = VNDetectFaceLandmarksRequest() let cq = VNDetectFaceCaptureQualityRequest() - do { try handler.perform([lm, cq]) } catch { - fputs("ERROR: Vision face landmarks failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([lm, cq], "face landmarks") let faces = (lm.results ?? []) let quals = (cq.results ?? []) var results: [FaceLandmarksResult] = [] for (i, f) in faces.enumerated() { - let b = box(f.boundingBox) + let b = Box(f.boundingBox, f.confidence) var marks: [String: [[Double]]] = [:] if let l = f.landmarks { let regions: [(String, VNFaceLandmarkRegion2D?)] = [ @@ -758,14 +734,14 @@ if isLandmarks { } let q: Float? = i < quals.count ? quals[i].faceCaptureQuality : nil results.append(FaceLandmarksResult( - x: b.0, y: b.1, w: b.2, h: b.3, confidence: f.confidence, + x: b.x, y: b.y, width: b.width, height: b.height, confidence: f.confidence, roll: f.roll.map { Double(truncating: $0) * 180 / .pi }, yaw: f.yaw.map { Double(truncating: $0) * 180 / .pi }, pitch: f.pitch.map { Double(truncating: $0) * 180 / .pi }, captureQuality: q, landmarks: marks )) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } @@ -774,15 +750,9 @@ if isLandmarks { if isHumans { let request = VNDetectHumanRectanglesRequest() request.upperBodyOnly = false - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision human detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - let results = ((request.results ?? [])).map { o -> HumanResult in - let b = box(o.boundingBox) - return HumanResult(x: b.0, y: b.1, w: b.2, h: b.3, confidence: o.confidence) - } - print(encodeJSON(results)) + perform([request], "human detection") + let results = (request.results ?? []).map { Box($0.boundingBox, $0.confidence) } + emit(encodeJSON(results)) exit(0) } @@ -790,66 +760,53 @@ if isHumans { if isBodyPose { let request = VNDetectHumanBodyPoseRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision body pose failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - let results = ((request.results ?? [])).map { o in + perform([request], "body pose") + let results = (request.results ?? []).map { o in PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } if isHandPose { let request = VNDetectHumanHandPoseRequest() request.maximumHandCount = 4 - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision hand pose failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - let results = ((request.results ?? [])).map { o -> PoseResult in + perform([request], "hand pose") + let results = (request.results ?? []).map { o -> PoseResult in let ch: String switch o.chirality { case .left: ch = "left"; case .right: ch = "right"; default: ch = "unknown" } return PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: ch) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } if isAnimalPose { if #available(macOS 14.0, *) { let request = VNDetectAnimalBodyPoseRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision animal pose failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - let results = ((request.results ?? [])).map { o in + perform([request], "animal pose") + let results = (request.results ?? []).map { o in PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } - fputs("ERROR: --animal-pose requires macOS 14 or newer\n", stderr) - exit(2) + fail("--animal-pose requires macOS 14 or newer", code: 2) } // ─── Animals (cat / dog) ───────────────────────────────────────────────────── if isAnimals { let request = VNRecognizeAnimalsRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision animal recognition failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - let results = ((request.results ?? [])).map { o -> AnimalResult in - let b = box(o.boundingBox) + perform([request], "animal recognition") + let results = (request.results ?? []).map { o -> AnimalResult in + let b = Box(o.boundingBox, o.confidence) return AnimalResult( labels: o.labels.map { ClassificationResult(identifier: $0.identifier, confidence: $0.confidence) }, - x: b.0, y: b.1, w: b.2, h: b.3, confidence: o.confidence + x: b.x, y: b.y, width: b.width, height: b.height, confidence: o.confidence ) } - print(encodeJSON(results)) + emit(encodeJSON(results)) exit(0) } @@ -857,15 +814,12 @@ if isAnimals { if isHorizon { let request = VNDetectHorizonRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision horizon detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "horizon detection") guard let o = request.results?.first as? VNHorizonObservation else { - print("null") + emit("null") exit(0) } - print(encodeJSON(HorizonResult(angleDegrees: Double(o.angle) * 180 / .pi))) + emit(encodeJSON(HorizonResult(angleDegrees: Double(o.angle) * 180 / .pi))) exit(0) } @@ -875,19 +829,15 @@ if isContours { let request = VNDetectContoursRequest() request.detectsDarkOnLight = !args.contains("--light-on-dark") if let r = roiRect { request.regionOfInterest = r } - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision contour detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "contour detection") guard let o = request.results?.first as? VNContoursObservation else { - print(encodeJSON(ContoursResult(totalContours: 0, topLevel: []))) + emit(encodeJSON(ContoursResult(totalContours: 0, topLevel: []))) exit(0) } let maxPoints = optValue("--max-points").flatMap { Int($0) } ?? 0 var top: [ContourResult] = [] for (i, c) in o.topLevelContours.enumerated() { - let bb = unROI(c.normalizedPath.boundingBox) - let b = box(bb) + let b = Box(unROI(c.normalizedPath.boundingBox), 1) var pts: [[Double]]? = nil if maxPoints > 0 { let all = c.normalizedPoints @@ -898,9 +848,9 @@ if isContours { } } top.append(ContourResult(index: i, pointCount: c.pointCount, childCount: c.childContourCount, - x: b.0, y: b.1, w: b.2, h: b.3, points: pts)) + x: b.x, y: b.y, width: b.width, height: b.height, points: pts)) } - print(encodeJSON(ContoursResult(totalContours: o.contourCount, topLevel: top))) + emit(encodeJSON(ContoursResult(totalContours: o.contourCount, topLevel: top))) exit(0) } @@ -911,21 +861,15 @@ if isSaliency { let request: VNImageBasedRequest = kind == "objectness" ? VNGenerateObjectnessBasedSaliencyImageRequest() : VNGenerateAttentionBasedSaliencyImageRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision saliency failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "saliency") guard let o = request.results?.first as? VNSaliencyImageObservation else { - print(encodeJSON(SaliencyResult(regions: [], heatmapPath: nil))) + emit(encodeJSON(SaliencyResult(regions: [], heatmapPath: nil))) exit(0) } - let regions = (o.salientObjects ?? []).map { r -> HumanResult in - let b = box(r.boundingBox) - return HumanResult(x: b.0, y: b.1, w: b.2, h: b.3, confidence: r.confidence) - } + let regions = (o.salientObjects ?? []).map { Box($0.boundingBox, $0.confidence) } var heat: String? = nil if let out = optValue("--out"), let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) { heat = out } - print(encodeJSON(SaliencyResult(regions: regions, heatmapPath: heat))) + emit(encodeJSON(SaliencyResult(regions: regions, heatmapPath: heat))) exit(0) } @@ -935,12 +879,9 @@ if isForeground { if #available(macOS 14.0, *) { let out = requireOut() let request = VNGenerateForegroundInstanceMaskRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision foreground mask failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "foreground mask") guard let o = request.results?.first as? VNInstanceMaskObservation else { - print(encodeJSON(MaskResult(instances: 0, outPath: ""))) + emit(encodeJSON(MaskResult(instances: 0, outPath: ""))) exit(0) } let maskOnly = args.contains("--mask-only") @@ -950,14 +891,12 @@ if isForeground { : try o.generateMaskedImage(ofInstances: o.allInstances, from: handler, croppedToInstancesExtent: args.contains("--tight")) guard let cg = cgFromPixelBuffer(pb), writePNG(cg, to: out) else { throw NSError(domain: "vision-helper", code: 1) } } catch { - fputs("ERROR: could not write foreground image: \(error.localizedDescription)\n", stderr) - exit(1) + fail("could not write foreground image: \(error.localizedDescription)") } - print(encodeJSON(MaskResult(instances: o.allInstances.count, outPath: out))) + emit(encodeJSON(MaskResult(instances: o.allInstances.count, outPath: out))) exit(0) } - fputs("ERROR: --foreground-mask requires macOS 14 or newer\n", stderr) - exit(2) + fail("--foreground-mask requires macOS 14 or newer", code: 2) } if isPersonMask { @@ -965,16 +904,12 @@ if isPersonMask { let request = VNGeneratePersonSegmentationRequest() request.qualityLevel = .accurate request.outputPixelFormat = kCVPixelFormatType_OneComponent8 - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision person segmentation failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "person segmentation") guard let o = request.results?.first as? VNPixelBufferObservation, let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) else { - fputs("ERROR: could not write person mask\n", stderr) - exit(1) + fail("could not write person mask") } - print(encodeJSON(MaskResult(instances: 1, outPath: out))) + emit(encodeJSON(MaskResult(instances: 1, outPath: out))) exit(0) } @@ -983,45 +918,27 @@ if isPersonMask { if isAesthetics { if #available(macOS 15.0, *) { let request = VNCalculateImageAestheticsScoresRequest() - do { try handler.perform([request]) } catch { - fputs("ERROR: Vision aesthetics failed: \(error.localizedDescription)\n", stderr) - exit(1) - } + perform([request], "aesthetics") guard let o = request.results?.first as? VNImageAestheticsScoresObservation else { - print("null") + emit("null") exit(0) } - print(encodeJSON(AestheticsResult(overallScore: o.overallScore, isUtility: o.isUtility))) + emit(encodeJSON(AestheticsResult(overallScore: o.overallScore, isUtility: o.isUtility))) exit(0) } - fputs("ERROR: --aesthetics requires macOS 15 or newer\n", stderr) - exit(2) + fail("--aesthetics requires macOS 15 or newer", code: 2) } // ─── Faces ─────────────────────────────────────────────────────────────────── if isFaces { - var results: [FaceResult] = [] + var results: [Box] = [] let request = VNDetectFaceRectanglesRequest { (req, _) in guard let obs = req.results as? [VNFaceObservation] else { return } - for o in obs { - let box = o.boundingBox - results.append(FaceResult( - x: Double(box.origin.x), - y: flipY(Double(box.origin.y), Double(box.size.height)), - w: Double(box.size.width), - h: Double(box.size.height), - confidence: o.confidence - )) - } - } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision face detection failed: \(error.localizedDescription)\n", stderr) - exit(1) + results = obs.map { Box($0.boundingBox, $0.confidence) } } - print(encodeJSON(results)) + perform([request], "face detection") + emit(encodeJSON(results)) exit(0) } @@ -1032,25 +949,15 @@ if isBarcodes { let request = VNDetectBarcodesRequest { (req, _) in guard let obs = req.results as? [VNBarcodeObservation] else { return } for o in obs { - let box = o.boundingBox + let b = Box(o.boundingBox, o.confidence) results.append(BarcodeResult( - type: o.symbology.rawValue, - value: o.payloadStringValue ?? "", - x: Double(box.origin.x), - y: flipY(Double(box.origin.y), Double(box.size.height)), - w: Double(box.size.width), - h: Double(box.size.height), - confidence: o.confidence + type: o.symbology.rawValue, value: o.payloadStringValue ?? "", + x: b.x, y: b.y, width: b.width, height: b.height, confidence: o.confidence )) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision barcode detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "barcode detection") + emit(encodeJSON(results)) exit(0) } @@ -1068,14 +975,9 @@ if isRectangles { )) } } - (request as VNDetectRectanglesRequest).maximumObservations = 0 - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision rectangle detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + request.maximumObservations = 0 + perform([request], "rectangle detection") + emit(encodeJSON(results)) exit(0) } @@ -1093,13 +995,8 @@ if isDocument { )) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision document detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "document detection") + emit(encodeJSON(results)) exit(0) } @@ -1114,12 +1011,7 @@ if isClassify { results.append(ClassificationResult(identifier: o.identifier, confidence: o.confidence)) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision classification failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "classification") + emit(encodeJSON(results)) exit(0) } diff --git a/src/ui.ts b/src/ui.ts index 7aac5d9..7d9631a 100644 --- a/src/ui.ts +++ b/src/ui.ts @@ -4,18 +4,9 @@ // Captures land on disk; only paths, geometry, and metadata flow back. // The library never synthesizes input — eyes, not hands. -import { execFile } from 'child_process'; -import { promisify } from 'util'; -import { existsSync, mkdirSync } from 'fs'; -import { open, readFile } from 'fs/promises'; -import { createHash } from 'crypto'; -import { tmpdir } from 'os'; -import { resolve, dirname, join } from 'path'; -import { fileURLToPath } from 'url'; - -const execFileAsync = promisify(execFile); -const __dirname = dirname(fileURLToPath(import.meta.url)); -const UI_BIN_PATH = resolve(__dirname, '../bin/ui-helper'); +import { readFile } from 'fs/promises'; +import { UI_BIN, runHelper, execHelper, tmpOutPath, sha256 } from './helper.js'; + const UI_HELPER_TIMEOUT_MS = 15_000; const SCREENCAPTURE_TIMEOUT_MS = 15_000; @@ -95,10 +86,8 @@ export interface CaptureOptions { // ─── ui-helper ─────────────────────────────────────────────────────────────── -async function runUi(...args: string[]): Promise { - const { stdout } = await execFileAsync(UI_BIN_PATH, args, { timeout: UI_HELPER_TIMEOUT_MS }); - return JSON.parse(stdout) as T; -} +const runUi = (...args: string[]) => + runHelper(UI_BIN, args, { timeout: UI_HELPER_TIMEOUT_MS }); /** On-screen windows, front-to-back. Pass `true` to include menu bar, dock and overlays. */ export function listWindows(includeAll = false): Promise { @@ -117,23 +106,9 @@ export function checkPermissions(): Promise { // ─── Capture ───────────────────────────────────────────────────────────────── -function capturePath(outPath?: string): string { - if (outPath) return resolve(outPath); - const dir = join(tmpdir(), 'macos-vision'); - mkdirSync(dir, { recursive: true }); - return join(dir, `capture-${Date.now()}.png`); -} - -/** Width/height straight from the PNG IHDR header — no subprocess. */ -async function pngPixelSize(path: string): Promise<{ w: number; h: number }> { - const fh = await open(path, 'r'); - try { - const buf = Buffer.alloc(8); - await fh.read(buf, 0, 8, 16); // IHDR starts at byte 8; width/height at 16/20 - return { w: buf.readUInt32BE(0), h: buf.readUInt32BE(4) }; - } finally { - await fh.close(); - } +/** Width/height straight from the PNG IHDR header (bytes 16–23). */ +function pngPixelSize(png: Buffer): { w: number; h: number } { + return { w: png.readUInt32BE(16), h: png.readUInt32BE(20) }; } async function resolveWindow(opts: CaptureOptions): Promise { @@ -178,19 +153,18 @@ export async function captureScreen(opts: CaptureOptions = {}): Promise 0 ? px.w / frame.w : 1, capturedAt: new Date().toISOString(), diff --git a/src/vision.ts b/src/vision.ts index 1d76a4c..1b175f7 100644 --- a/src/vision.ts +++ b/src/vision.ts @@ -4,15 +4,16 @@ // Functions that produce pixels (masks, crops, heatmaps) write PNG files and // return paths — never image bytes. -import { execFile } from 'child_process'; -import { resolve, dirname, join } from 'path'; -import { fileURLToPath } from 'url'; -import { mkdirSync } from 'fs'; -import { tmpdir } from 'os'; +import { resolve } from 'path'; +import { + VISION_BIN, + runHelper, + runGated, + tmpOutPath, + UnsupportedOnThisMacOSError, +} from './helper.js'; -const __dirname = dirname(fileURLToPath(import.meta.url)); -const BIN_PATH = resolve(__dirname, '../bin/vision-helper'); -const TIMEOUT_MS = 60_000; +export { UnsupportedOnThisMacOSError }; // ─── Shared types ──────────────────────────────────────────────────────────── @@ -24,17 +25,8 @@ export interface NormalizedRect { height: number; } -interface RawBox { - x: number; - y: number; - w: number; - h: number; - confidence: number; -} - -function box(b: RawBox): NormalizedRect & { confidence: number } { - return { x: b.x, y: b.y, width: b.w, height: b.h, confidence: b.confidence }; -} +/** A detection: where it is and how sure Vision is. */ +export type Detection = NormalizedRect & { confidence: number }; /** Options shared by every text-recognition path (OCR, text regions, document structure). */ export interface TextRecognitionOptions { @@ -54,7 +46,7 @@ export interface TextRecognitionOptions { minTextHeight?: number; } -/** Serialises shared text options into helper flags. */ +/** Serialises shared text options into helper flags. Canonical — also used as the OCR cache key. */ export function textOptionArgs(o: TextRecognitionOptions = {}): string[] { const args: string[] = []; if (o.languages?.length) args.push('--lang', o.languages.join(',')); @@ -70,33 +62,10 @@ export function textOptionArgs(o: TextRecognitionOptions = {}): string[] { return args; } -async function run(args: string[], input?: string): Promise { - const stdout = await new Promise((resolvePromise, reject) => { - const child = execFile( - BIN_PATH, - args, - { timeout: TIMEOUT_MS, maxBuffer: 64 * 1024 * 1024 }, - (err, out) => { - if (err) { - reject(err); - return; - } - resolvePromise(out); - } - ); - if (input !== undefined) child.stdin?.end(input); - }); - // Some Vision models log to stdout; the JSON payload is always the last line. - const lines = stdout.trim().split('\n'); - return JSON.parse(lines[lines.length - 1]) as T; -} - -function outPath(out: string | undefined, prefix: string): string { - if (out) return resolve(out); - const dir = join(tmpdir(), 'macos-vision'); - mkdirSync(dir, { recursive: true }); - return join(dir, `${prefix}-${Date.now()}-${Math.floor(Math.random() * 1e6)}.png`); -} +const run = (args: string[], input?: string) => + runHelper(VISION_BIN, args, { timeout: 60_000, input }); +const gated = (args: string[], feature: string, minVersion: string) => + runGated(VISION_BIN, args, feature, minVersion, { timeout: 60_000 }); // ─── Capabilities ──────────────────────────────────────────────────────────── @@ -110,9 +79,15 @@ export interface VisionCapabilities { features: Record; } -/** What this machine can do. Cheap; cache it per process. */ +let capabilitiesPromise: Promise | undefined; + +/** What this machine can do. Constant for the process lifetime, so memoized. */ export function visionCapabilities(): Promise { - return run(['--capabilities']); + capabilitiesPromise ??= run(['--capabilities']).catch((err) => { + capabilitiesPromise = undefined; + throw err; + }); + return capabilitiesPromise; } /** BCP-47 codes supported by the accurate OCR model. */ @@ -142,15 +117,14 @@ export function imageInfo(imagePath: string): Promise { // ─── Text regions (no recognition) ─────────────────────────────────────────── -export type TextRegion = NormalizedRect & { confidence: number }; +export type TextRegion = Detection; /** Where text is, without reading it. Much faster than OCR — use to pick regions of interest. */ -export async function detectTextRegions( +export function detectTextRegions( imagePath: string, options: Pick = {} ): Promise { - const raw = await run(['--text-rects', ...textOptionArgs(options), resolve(imagePath)]); - return raw.map(box); + return run(['--text-rects', ...textOptionArgs(options), resolve(imagePath)]); } // ─── Image similarity ──────────────────────────────────────────────────────── @@ -190,18 +164,12 @@ export function extractEntities(text: string): Promise { export interface DocLine { text: string; confidence: number; - bbox: DocBox; -} -export interface DocBox { - x: number; - y: number; - w: number; - h: number; + bbox: NormalizedRect; } export interface DocText { text: string; alignment?: 'leading' | 'center' | 'trailing'; - bbox: DocBox; + bbox: NormalizedRect; lines: DocLine[]; } export interface DocCell { @@ -210,7 +178,7 @@ export interface DocCell { col: number; rowSpan: number; colSpan: number; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocTable { rowCount: number; @@ -219,16 +187,16 @@ export interface DocTable { rows: string[][]; /** Unique cells with spans */ cells: DocCell[]; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocListItem { marker: string; text: string; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocList { items: DocListItem[]; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocDetectedData { type: @@ -245,12 +213,12 @@ export interface DocDetectedData { | 'unknown'; text: string; value?: string; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocBarcode { type: string; value: string; - bbox: DocBox; + bbox: NormalizedRect; } export interface DocumentStructure { @@ -266,34 +234,20 @@ export interface DocumentStructure { detectedData: DocDetectedData[]; } -export class UnsupportedOnThisMacOSError extends Error { - constructor(feature: string, minVersion: string) { - super(`${feature} requires macOS ${minVersion} or newer`); - this.name = 'UnsupportedOnThisMacOSError'; - } -} - /** * Native document understanding (macOS 26+): paragraphs, tables, lists, title, * barcodes and detected data with positions — no heuristics, no LLM. * Throws `UnsupportedOnThisMacOSError` on older systems; check `visionCapabilities().features.documentStructure`. */ -export async function recognizeDocument( +export function recognizeDocument( imagePath: string, options: TextRecognitionOptions = {} ): Promise { - try { - return await run([ - '--document-structure', - ...textOptionArgs(options), - resolve(imagePath), - ]); - } catch (err) { - if ((err as { code?: number }).code === 2) { - throw new UnsupportedOnThisMacOSError('recognizeDocument', '26'); - } - throw err; - } + return gated( + ['--document-structure', ...textOptionArgs(options), resolve(imagePath)], + 'recognizeDocument', + '26' + ); } // ─── Quality signals ───────────────────────────────────────────────────────── @@ -308,9 +262,9 @@ export interface LensSmudge { /** Was the photo taken through a dirty lens? macOS 26+. Returns `supported:false` instead of guessing. */ export async function detectLensSmudge(imagePath: string): Promise { try { - return await run(['--smudge', resolve(imagePath)]); + return await gated(['--smudge', resolve(imagePath)], 'detectLensSmudge', '26'); } catch (err) { - if ((err as { code?: number }).code === 2) return { confidence: 0, supported: false }; + if (err instanceof UnsupportedOnThisMacOSError) return { confidence: 0, supported: false }; throw err; } } @@ -345,12 +299,7 @@ export function detectHorizon(imagePath: string): Promise { // ─── People, faces, poses, animals ─────────────────────────────────────────── -export interface FaceLandmarks { - x: number; - y: number; - width: number; - height: number; - confidence: number; +export interface FaceLandmarks extends Detection { /** Head rotation in degrees, when available */ roll?: number; yaw?: number; @@ -361,26 +310,15 @@ export interface FaceLandmarks { landmarks: Record; } -export async function detectFaceLandmarks(imagePath: string): Promise { - const raw = await run< - Array> - >(['--face-landmarks', resolve(imagePath)]); - return raw.map((f) => ({ - ...box(f), - roll: f.roll, - yaw: f.yaw, - pitch: f.pitch, - captureQuality: f.captureQuality, - landmarks: f.landmarks, - })); +export function detectFaceLandmarks(imagePath: string): Promise { + return run(['--face-landmarks', resolve(imagePath)]); } -export type HumanBox = NormalizedRect & { confidence: number }; +export type HumanBox = Detection; /** Full-body person boxes. */ -export async function detectHumans(imagePath: string): Promise { - const raw = await run(['--humans', resolve(imagePath)]); - return raw.map(box); +export function detectHumans(imagePath: string): Promise { + return run(['--humans', resolve(imagePath)]); } export interface Keypoint { @@ -416,23 +354,14 @@ export async function detectAnimalPose(imagePath: string): Promise { } } -export interface Animal { +export interface Animal extends Detection { /** e.g. [{ identifier: 'Cat', confidence: 0.98 }] */ labels: Array<{ identifier: string; confidence: number }>; - x: number; - y: number; - width: number; - height: number; - confidence: number; } /** Cats and dogs with boxes. */ -export async function detectAnimals(imagePath: string): Promise { - const raw = await run>([ - '--animals', - resolve(imagePath), - ]); - return raw.map((a) => ({ ...box(a), labels: a.labels })); +export function detectAnimals(imagePath: string): Promise { + return run(['--animals', resolve(imagePath)]); } // ─── Saliency, contours ────────────────────────────────────────────────────── @@ -446,19 +375,18 @@ export interface SaliencyOptions { export interface Saliency { /** Up to 3 salient regions, sorted by the model */ - regions: Array; + regions: Detection[]; heatmapPath?: string; } -export async function detectSaliency( +export function detectSaliency( imagePath: string, options: SaliencyOptions = {} ): Promise { const args = ['--saliency', options.mode ?? 'attention']; if (options.heatmapPath) args.push('--out', resolve(options.heatmapPath)); args.push(resolve(imagePath)); - const raw = await run<{ regions: RawBox[]; heatmapPath?: string }>(args); - return { regions: raw.regions.map(box), heatmapPath: raw.heatmapPath ?? undefined }; + return run(args); } export interface ContourOptions { @@ -469,14 +397,10 @@ export interface ContourOptions { regionOfInterest?: NormalizedRect; } -export interface Contour { +export interface Contour extends NormalizedRect { index: number; pointCount: number; childCount: number; - x: number; - y: number; - width: number; - height: number; points?: [number, number][]; } @@ -531,7 +455,7 @@ export function cropImage( '--crop', `${region.x},${region.y},${region.width},${region.height}`, '--out', - outPath(out, 'crop'), + tmpOutPath('crop', out), resolve(imagePath), ]); } @@ -541,7 +465,7 @@ export function cropDocument(imagePath: string, out?: string): Promise([ '--document-crop', '--out', - outPath(out, 'document'), + tmpOutPath('document', out), resolve(imagePath), ]); } @@ -565,7 +489,7 @@ export async function extractForeground( imagePath: string, options: ForegroundOptions = {} ): Promise { - const args = ['--foreground-mask', '--out', outPath(options.out, 'foreground')]; + const args = ['--foreground-mask', '--out', tmpOutPath('foreground', options.out)]; if (options.maskOnly) args.push('--mask-only'); if (options.tight) args.push('--tight'); args.push(resolve(imagePath)); @@ -583,7 +507,7 @@ export function personMask(imagePath: string, out?: string): Promise return run([ '--person-mask', '--out', - outPath(out, 'person-mask'), + tmpOutPath('person-mask', out), resolve(imagePath), ]); } From 8d5abc3d7e4b3292d279075e20efe8c834437dc1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 07:34:42 +0200 Subject: [PATCH 4/7] fix: restore contour width/height, gate modern Vision on SDK, clarify helper errors Found by an end-to-end pass over the built package: - detectContours still mapped the pre-refactor c.w/c.h wire names, so every contour came back with width/height undefined. Returns the helper shape now, with a regression test asserting width/height on every rect-shaped result. - vision-helper referenced symbols introduced in macOS 14/15/26 SDKs unconditionally, so the swiftc fallback in install-native.js could not build on an older Mac (#available guards runtime, not the SDK). Modern paths are now behind -DSDK_14/15/26, which both build scripts derive from "xcrun --sdk macosx --show-sdk-version"; a build without them still compiles, reports the feature as false in --capabilities and exits 2, which maps to UnsupportedOnThisMacOSError. Verified by compiling all four SDK tiers. - Helper failures leaked Node's "Command failed: /path/to/vision-helper --flags" plus a stack trace. execHelper now surfaces the helper's own ERROR line while preserving the exit status (so gating still works) and stderr (so captureScreen can still quote screencapture). Tests: 82 unit (6 new: wire-shape, capability gating, error reporting). Co-Authored-By: Claude Fable 5 --- .changeset/ui-helper-and-extended-vision.md | 4 +- README.md | 2 + scripts/build-native-cross.js | 17 +- scripts/install-native.js | 17 +- src/cli.ts | 2 +- src/helper.ts | 22 +- src/native/vision-helper.swift | 41 ++- src/vision.ts | 23 +- test/vision.test.ts | 305 ++++++++++++++------ 9 files changed, 314 insertions(+), 119 deletions(-) diff --git a/.changeset/ui-helper-and-extended-vision.md b/.changeset/ui-helper-and-extended-vision.md index b8f32ea..4c575c3 100644 --- a/.changeset/ui-helper-and-extended-vision.md +++ b/.changeset/ui-helper-and-extended-vision.md @@ -10,4 +10,6 @@ feat: ui-helper (windows / displays / permissions / captureScreen) and extended - `extractEntities` (links, e-mails, phones, addresses, dates), `detectTextRegions`, `compareImages`, `imageInfo`, `visionCapabilities`, `supportedOcrLanguages`. - People & scenes: `detectFaceLandmarks`, `detectHumans`, `detectBodyPose`, `detectHandPose`, `detectAnimals`, `detectAnimalPose`, `detectHorizon`, `detectSaliency`, `detectContours`, `imageAesthetics`, `detectLensSmudge`. - Pixel ops that return paths: `cropImage`, `cropDocument` (perspective-corrected), `extractForeground`, `personMask`. -- Native helpers now target macOS 13 (newer features gate on `#available`). Node >= 20. +- Native helpers now target macOS 13. Modern features are gated twice — at compile time on the SDK (`-DSDK_14/15/26`, set automatically by the build scripts) and at runtime on `#available` — so the `swiftc` fallback also builds on older macOS, with unavailable features reported as `false` by `visionCapabilities()` and raising `UnsupportedOnThisMacOSError`. +- Helper failures now surface their own message (`Cannot open file: …`) instead of Node's `Command failed: /path/to/vision-helper …`. +- Node >= 20. diff --git a/README.md b/README.md index 0b0781e..ca46bd6 100644 --- a/README.md +++ b/README.md @@ -167,6 +167,8 @@ const caps = await visionCapabilities(); // { helperVersion, macosVersion, ocrLanguages: ['en-US','pl-PL',…], features: { documentStructure, foregroundMask, … } } ``` +A feature is reported `true` only when both this macOS **and** the SDK the helper was built against provide it. Anything reported `false` raises `UnsupportedOnThisMacOSError` rather than failing obscurely, so an agent can branch on `caps.features` before planning work. + ### OCR tuning `ocr()` now accepts Vision's recognition knobs. Results for `regionOfInterest` are still reported in full-image coordinates. diff --git a/scripts/build-native-cross.js b/scripts/build-native-cross.js index e75420a..fc21658 100644 --- a/scripts/build-native-cross.js +++ b/scripts/build-native-cross.js @@ -13,6 +13,21 @@ const TARGETS = [ ]; const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; +// Symbols added in newer SDKs are absent when building against an older one, so +// the helper gates them on -DSDK_nn. Detect what this machine's SDK provides. +function sdkDefines() { + try { + const raw = execSync('xcrun --sdk macosx --show-sdk-version', { encoding: 'utf8' }).trim(); + const major = parseInt(raw.split('.')[0], 10); + if (!Number.isFinite(major)) return []; + return [14, 15, 26].filter((n) => major >= n).map((n) => `-DSDK_${n}`); + } catch { + return []; + } +} + +const DEFINES = sdkDefines().join(' '); + for (const { arch, swift } of TARGETS) { const outDir = path.join(root, 'bin', `darwin-${arch}`); mkdirSync(outDir, { recursive: true }); @@ -20,7 +35,7 @@ for (const { arch, swift } of TARGETS) { for (const name of HELPERS) { const src = path.join(root, 'src', 'native', `${name}.swift`); const out = path.join(outDir, name); - execSync(`swiftc -O -target ${swift} "${src}" -o "${out}"`, { stdio: 'inherit' }); + execSync(`swiftc -O -target ${swift} ${DEFINES} "${src}" -o "${out}"`, { stdio: 'inherit' }); } const tarball = `bin-darwin-${arch}.tar.gz`; diff --git a/scripts/install-native.js b/scripts/install-native.js index 0ab610c..3bc37e0 100644 --- a/scripts/install-native.js +++ b/scripts/install-native.js @@ -19,6 +19,20 @@ const pkg = JSON.parse(readFileSync(path.join(root, 'package.json'), 'utf8')); const binDir = path.join(root, 'bin'); const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; +// Symbols added in newer SDKs are absent when building against an older one, so +// the helper gates them on -DSDK_nn. Detect what this machine's SDK provides. +function sdkDefines() { + try { + const raw = execSync('xcrun --sdk macosx --show-sdk-version', { encoding: 'utf8' }).trim(); + const major = parseInt(raw.split('.')[0], 10); + if (!Number.isFinite(major)) return []; + return [14, 15, 26].filter((n) => major >= n).map((n) => `-DSDK_${n}`); + } catch { + return []; + } +} + + // 0. Skip if all binaries already exist (cached install) if (HELPERS.every((h) => existsSync(path.join(binDir, h)))) { process.exit(0); @@ -85,11 +99,12 @@ function fallbackToSwiftc() { process.exit(1); } mkdirSync(binDir, { recursive: true }); + const defines = sdkDefines().join(' '); for (const h of HELPERS) { const src = path.join(root, 'src', 'native', `${h}.swift`); const out = path.join(binDir, h); try { - execSync(`swiftc -O "${src}" -o "${out}"`, { stdio: 'inherit' }); + execSync(`swiftc -O ${defines} "${src}" -o "${out}"`, { stdio: 'inherit' }); console.log(`✅ macos-vision: compiled ${h} locally`); } catch { console.error(`❌ macos-vision: ${h} compilation failed.`); diff --git a/src/cli.ts b/src/cli.ts index 3f0500b..067523f 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -283,7 +283,7 @@ if (flags.has('--markdown')) { console.log(JSON.stringify(await fn(), null, 2)); } } catch (error) { - console.error('Error:', error); + console.error(`Error: ${error instanceof Error ? error.message : String(error)}`); process.exit(1); } })(); diff --git a/src/helper.ts b/src/helper.ts index f45b592..7483edc 100644 --- a/src/helper.ts +++ b/src/helper.ts @@ -28,6 +28,26 @@ export class UnsupportedOnThisMacOSError extends Error { } } +/** + * Helpers report failures as `ERROR: ` on stderr. Surface that + * line instead of Node's `Command failed: /long/path/to/helper --flags …`, + * keeping `code` so callers can still recognise the "unsupported" status. + */ +function helperError(err: unknown, stderr: string): Error { + const raw = err as { message?: string; code?: number }; + const reported = stderr + .split('\n') + .map((line) => line.trim()) + .find((line) => line.startsWith('ERROR: ')); + const error = new Error( + reported ? reported.slice('ERROR: '.length) : (raw.message ?? String(err)) + ) as Error & { code?: number; stderr?: string }; + if (raw.code !== undefined) error.code = raw.code; + // Kept for callers that inspect a non-helper tool's output (e.g. screencapture). + error.stderr = stderr; + return error; +} + export interface ExecOptions { timeout?: number; /** Written to the helper's stdin, then closed. */ @@ -41,7 +61,7 @@ export function execHelper(bin: string, args: string[], opts: ExecOptions = {}): bin, args, { timeout: opts.timeout ?? 30_000, maxBuffer: 64 * 1024 * 1024 }, - (err, stdout) => (err ? reject(err) : resolvePromise(stdout)) + (err, stdout, stderr) => (err ? reject(helperError(err, stderr)) : resolvePromise(stdout)) ); if (opts.input !== undefined) child.stdin?.end(opts.input); }); diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index cbc3397..c1885f5 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -176,8 +176,29 @@ func macOSVersionString() -> String { return "\(v.majorVersion).\(v.minorVersion).\(v.patchVersion)" } +let macOS14 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 14, minorVersion: 0, patchVersion: 0)) let macOS15 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 15, minorVersion: 0, patchVersion: 0)) let macOS26 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 26, minorVersion: 0, patchVersion: 0)) + +// A symbol introduced in a newer SDK is simply absent when the helper is built +// against an older one — which is exactly what the swiftc fallback does on an +// older Mac. So each modern feature is gated twice: at compile time on the SDK +// (-DSDK_nn, set by the build scripts) and at runtime on #available. +#if SDK_14 +let sdk14 = true +#else +let sdk14 = false +#endif +#if SDK_15 +let sdk15 = true +#else +let sdk15 = false +#endif +#if SDK_26 +let sdk26 = true +#else +let sdk26 = false +#endif let iso8601 = ISO8601DateFormatter() func supportedLanguages() -> [String] { @@ -262,11 +283,11 @@ if args.contains("--capabilities") { "ocr": true, "ocrOptions": true, "faces": true, "barcodes": true, "rectangles": true, "document": true, "classify": true, "textRects": true, "compare": true, "entities": true, - "documentStructure": macOS26, "lensSmudge": macOS26, + "documentStructure": macOS26 && sdk26, "lensSmudge": macOS26 && sdk26, "faceLandmarks": true, "humans": true, "bodyPose": true, "handPose": true, - "animals": true, "animalPose": true, "horizon": true, "contours": true, - "saliency": true, "foregroundMask": true, "personMask": true, - "aesthetics": macOS15, "documentCrop": true, "crop": true, "imageInfo": true, + "animals": true, "animalPose": macOS14 && sdk14, "horizon": true, "contours": true, + "saliency": true, "foregroundMask": macOS14 && sdk14, "personMask": true, + "aesthetics": macOS15 && sdk15, "documentCrop": true, "crop": true, "imageInfo": true, ] ) emit(encodeJSON(caps)) @@ -442,6 +463,7 @@ if isTextRects { // ─── macOS 26+: document structure & lens smudge (new Vision Swift API) ────── +#if SDK_26 @available(macOS 26.0, *) func region(_ r: NormalizedRegion) -> DocRegion { let b = r.boundingBox @@ -591,20 +613,25 @@ func runSmudge() -> SmudgeResult { let supported = !noise.contains("Unable to find") return SmudgeResult(confidence: conf, supported: supported) } +#endif if isStructure { +#if SDK_26 if #available(macOS 26.0, *) { if let s = runDocumentStructure() { emit(encodeJSON(s)) } exit(0) } +#endif fail("--document-structure requires macOS 26 or newer", code: 2) } if isSmudge { +#if SDK_26 if #available(macOS 26.0, *) { emit(encodeJSON(runSmudge())) exit(0) } +#endif fail("--smudge requires macOS 26 or newer", code: 2) } @@ -782,6 +809,7 @@ if isHandPose { } if isAnimalPose { +#if SDK_14 if #available(macOS 14.0, *) { let request = VNDetectAnimalBodyPoseRequest() perform([request], "animal pose") @@ -791,6 +819,7 @@ if isAnimalPose { emit(encodeJSON(results)) exit(0) } +#endif fail("--animal-pose requires macOS 14 or newer", code: 2) } @@ -876,6 +905,7 @@ if isSaliency { // ─── Foreground subject cutout / person mask ───────────────────────────────── if isForeground { +#if SDK_14 if #available(macOS 14.0, *) { let out = requireOut() let request = VNGenerateForegroundInstanceMaskRequest() @@ -896,6 +926,7 @@ if isForeground { emit(encodeJSON(MaskResult(instances: o.allInstances.count, outPath: out))) exit(0) } +#endif fail("--foreground-mask requires macOS 14 or newer", code: 2) } @@ -916,6 +947,7 @@ if isPersonMask { // ─── Aesthetics (macOS 15+) ────────────────────────────────────────────────── if isAesthetics { +#if SDK_15 if #available(macOS 15.0, *) { let request = VNCalculateImageAestheticsScoresRequest() perform([request], "aesthetics") @@ -926,6 +958,7 @@ if isAesthetics { emit(encodeJSON(AestheticsResult(overallScore: o.overallScore, isUtility: o.isUtility))) exit(0) } +#endif fail("--aesthetics requires macOS 15 or newer", code: 2) } diff --git a/src/vision.ts b/src/vision.ts index 1b175f7..f00744c 100644 --- a/src/vision.ts +++ b/src/vision.ts @@ -410,31 +410,12 @@ export interface Contours { } /** Edge/shape contours — useful for charts, diagrams, UI boundaries. */ -export async function detectContours( - imagePath: string, - options: ContourOptions = {} -): Promise { +export function detectContours(imagePath: string, options: ContourOptions = {}): Promise { const args = ['--contours']; if (options.maxPoints) args.push('--max-points', String(options.maxPoints)); if (options.lightOnDark) args.push('--light-on-dark'); args.push(...textOptionArgs({ regionOfInterest: options.regionOfInterest }), resolve(imagePath)); - const raw = await run<{ - totalContours: number; - topLevel: Array & { w: number; h: number }>; - }>(args); - return { - totalContours: raw.totalContours, - topLevel: raw.topLevel.map((c) => ({ - index: c.index, - pointCount: c.pointCount, - childCount: c.childCount, - x: c.x, - y: c.y, - width: c.w, - height: c.h, - points: c.points, - })), - }; + return run(args); } // ─── Pixel-producing operations (write PNG, return path) ───────────────────── diff --git a/test/vision.test.ts b/test/vision.test.ts index 1cf6dcc..6fefbff 100644 --- a/test/vision.test.ts +++ b/test/vision.test.ts @@ -20,9 +20,12 @@ import { cropImage, cropDocument, extractForeground, + imageAesthetics, + detectAnimalPose, UnsupportedOnThisMacOSError, listDisplays, checkPermissions, + captureScreen, } from '../src/index.js'; const __dirname = dirname(fileURLToPath(import.meta.url)); @@ -46,37 +49,53 @@ describe('visionCapabilities()', () => { }); describe('ocr() — tuning options', () => { - it('regionOfInterest restricts results but reports full-image coordinates', async () => { - const top = await ocr(SAMPLE_IMG, { - format: 'blocks', - regionOfInterest: { x: 0, y: 0, width: 1, height: 0.1 }, - }); - expect(top.length).toBeGreaterThan(0); - for (const b of top) { - expect(b.y).toBeLessThan(0.12); - expect(b.height).toBeLessThan(0.1); - } - expect(top.map((b) => b.text).join(' ')).toContain('Henry VIII'); - }, T); - - it('languages + languageCorrection:false still reads the fixture', async () => { - const text = await ocr(SAMPLE_IMG, { languages: ['en-US'], languageCorrection: false }); - expect(text).toContain('Henry VIII'); - }, T); - - it('fast mode returns text', async () => { - const text = await ocr(SAMPLE_IMG, { fast: true }); - expect(text).toContain('Henry'); - }, T); - - it('cache:true returns identical result on second call', async () => { - const opts = { cache: true, customWords: ['Wikipedia'] } as const; - const a = await ocr(SAMPLE_IMG, opts); - const started = Date.now(); - const b = await ocr(SAMPLE_IMG, opts); - expect(b).toBe(a); - expect(Date.now() - started).toBeLessThan(200); - }, T); + it( + 'regionOfInterest restricts results but reports full-image coordinates', + async () => { + const top = await ocr(SAMPLE_IMG, { + format: 'blocks', + regionOfInterest: { x: 0, y: 0, width: 1, height: 0.1 }, + }); + expect(top.length).toBeGreaterThan(0); + for (const b of top) { + expect(b.y).toBeLessThan(0.12); + expect(b.height).toBeLessThan(0.1); + } + expect(top.map((b) => b.text).join(' ')).toContain('Henry VIII'); + }, + T + ); + + it( + 'languages + languageCorrection:false still reads the fixture', + async () => { + const text = await ocr(SAMPLE_IMG, { languages: ['en-US'], languageCorrection: false }); + expect(text).toContain('Henry VIII'); + }, + T + ); + + it( + 'fast mode returns text', + async () => { + const text = await ocr(SAMPLE_IMG, { fast: true }); + expect(text).toContain('Henry'); + }, + T + ); + + it( + 'cache:true returns identical result on second call', + async () => { + const opts = { cache: true, customWords: ['Wikipedia'] } as const; + const a = await ocr(SAMPLE_IMG, opts); + const started = Date.now(); + const b = await ocr(SAMPLE_IMG, opts); + expect(b).toBe(a); + expect(Date.now() - started).toBeLessThan(200); + }, + T + ); it('onProgress fires once per PDF page', async () => { const calls: Array<[number, number]> = []; @@ -94,21 +113,29 @@ describe('imageInfo() / detectTextRegions() / compareImages()', () => { expect(info.format).toBe('public.png'); }); - it('detectTextRegions finds many text boxes', async () => { - const regions = await detectTextRegions(SAMPLE_IMG); - expect(regions.length).toBeGreaterThan(20); - for (const r of regions) { - expect(r.x).toBeGreaterThanOrEqual(-0.01); - expect(r.y).toBeGreaterThanOrEqual(-0.01); - expect(r.width).toBeGreaterThan(0); - } - }, T); + it( + 'detectTextRegions finds many text boxes', + async () => { + const regions = await detectTextRegions(SAMPLE_IMG); + expect(regions.length).toBeGreaterThan(20); + for (const r of regions) { + expect(r.x).toBeGreaterThanOrEqual(-0.01); + expect(r.y).toBeGreaterThanOrEqual(-0.01); + expect(r.width).toBeGreaterThan(0); + } + }, + T + ); - it('compareImages: identical → 0, different → > 0.5', async () => { - expect((await compareImages(SAMPLE_IMG, SAMPLE_IMG)).distance).toBe(0); - const crop = await cropImage(SAMPLE_IMG, { x: 0.6, y: 0.1, width: 0.3, height: 0.3 }); - expect((await compareImages(SAMPLE_IMG, crop.outPath)).distance).toBeGreaterThan(0.3); - }, T); + it( + 'compareImages: identical → 0, different → > 0.5', + async () => { + expect((await compareImages(SAMPLE_IMG, SAMPLE_IMG)).distance).toBe(0); + const crop = await cropImage(SAMPLE_IMG, { x: 0.6, y: 0.1, width: 0.3, height: 0.3 }); + expect((await compareImages(SAMPLE_IMG, crop.outPath)).distance).toBeGreaterThan(0.3); + }, + T + ); }); describe('extractEntities()', () => { @@ -129,7 +156,9 @@ describe('recognizeDocument()', () => { it('returns structure on macOS 26+, throws UnsupportedOnThisMacOSError otherwise', async () => { const caps = await visionCapabilities(); if (!caps.features.documentStructure) { - await expect(recognizeDocument(SAMPLE_IMG)).rejects.toBeInstanceOf(UnsupportedOnThisMacOSError); + await expect(recognizeDocument(SAMPLE_IMG)).rejects.toBeInstanceOf( + UnsupportedOnThisMacOSError + ); return; } const doc = await recognizeDocument(SAMPLE_IMG, { languages: ['en-US'] }); @@ -139,40 +168,88 @@ describe('recognizeDocument()', () => { expect(Array.isArray(doc.tables)).toBe(true); }, 60_000); - it('detectLensSmudge never throws on supported systems', async () => { - const caps = await visionCapabilities(); - const r = await detectLensSmudge(SAMPLE_IMG); - expect(typeof r.supported).toBe('boolean'); - if (!caps.features.lensSmudge) expect(r.supported).toBe(false); - }, T); + it( + 'detectLensSmudge never throws on supported systems', + async () => { + const caps = await visionCapabilities(); + const r = await detectLensSmudge(SAMPLE_IMG); + expect(typeof r.supported).toBe('boolean'); + if (!caps.features.lensSmudge) expect(r.supported).toBe(false); + }, + T + ); }); describe('people / saliency / contours', () => { - it('detectHumans finds the portrait on the fixture', async () => { - const humans = await detectHumans(SAMPLE_IMG); - expect(humans.length).toBeGreaterThanOrEqual(1); - expect(humans[0].x).toBeGreaterThan(0.5); - }, T); - - it('detectBodyPose returns named joints', async () => { - const poses = await detectBodyPose(SAMPLE_IMG); - expect(poses.length).toBeGreaterThanOrEqual(1); - expect(Object.keys(poses[0].joints).length).toBeGreaterThan(5); - }, T); - - it('detectSaliency returns regions and can write a heatmap', async () => { - const out = resolve(tmpdir(), `macos-vision-test-heat-${Date.now()}.png`); - const s = await detectSaliency(SAMPLE_IMG, { mode: 'objectness', heatmapPath: out }); - expect(s.regions.length).toBeGreaterThan(0); - expect(s.heatmapPath).toBe(out); - expect(existsSync(out)).toBe(true); - }, T); - - it('detectContours counts contours and samples points', async () => { - const c = await detectContours(SAMPLE_IMG, { maxPoints: 4 }); - expect(c.totalContours).toBeGreaterThan(10); - expect(c.topLevel[0].points?.length).toBeLessThanOrEqual(5); - }, T); + it( + 'detectHumans finds the portrait on the fixture', + async () => { + const humans = await detectHumans(SAMPLE_IMG); + expect(humans.length).toBeGreaterThanOrEqual(1); + expect(humans[0].x).toBeGreaterThan(0.5); + }, + T + ); + + it( + 'detectBodyPose returns named joints', + async () => { + const poses = await detectBodyPose(SAMPLE_IMG); + expect(poses.length).toBeGreaterThanOrEqual(1); + expect(Object.keys(poses[0].joints).length).toBeGreaterThan(5); + }, + T + ); + + it( + 'detectSaliency returns regions and can write a heatmap', + async () => { + const out = resolve(tmpdir(), `macos-vision-test-heat-${Date.now()}.png`); + const s = await detectSaliency(SAMPLE_IMG, { mode: 'objectness', heatmapPath: out }); + expect(s.regions.length).toBeGreaterThan(0); + expect(s.heatmapPath).toBe(out); + expect(existsSync(out)).toBe(true); + }, + T + ); + + it( + 'detectContours counts contours and samples points', + async () => { + const c = await detectContours(SAMPLE_IMG, { maxPoints: 4 }); + expect(c.totalContours).toBeGreaterThan(10); + expect(c.topLevel[0].points?.length).toBeLessThanOrEqual(5); + }, + T + ); + + // Guards the helper↔TS wire contract: every rect-shaped result uses width/height. + it('all rect-shaped results expose numeric width/height', async () => { + const [regions, humans, contours, saliency, blocks] = await Promise.all([ + detectTextRegions(SAMPLE_IMG), + detectHumans(SAMPLE_IMG), + detectContours(SAMPLE_IMG, { maxPoints: 2 }), + detectSaliency(SAMPLE_IMG, { mode: 'objectness' }), + ocr(SAMPLE_IMG, { format: 'blocks' }), + ]); + const rects = [...regions, ...humans, ...contours.topLevel, ...saliency.regions, ...blocks]; + expect(rects.length).toBeGreaterThan(0); + for (const r of rects) { + expect(typeof r.width).toBe('number'); + expect(typeof r.height).toBe('number'); + expect(Number.isFinite(r.width)).toBe(true); + } + }, 60_000); + + it('recognizeDocument bboxes use width/height too', async () => { + const caps = await visionCapabilities(); + if (!caps.features.documentStructure) return; + const doc = await recognizeDocument(SAMPLE_IMG, { languages: ['en-US'] }); + for (const p of doc.paragraphs.slice(0, 5)) { + expect(typeof p.bbox.width).toBe('number'); + expect(typeof p.bbox.height).toBe('number'); + } + }, 60_000); }); describe('pixel ops return paths, never bytes', () => { @@ -183,19 +260,69 @@ describe('pixel ops return paths, never bytes', () => { expect(r.height).toBe(672); }); - it('cropDocument writes a perspective-corrected PNG', async () => { - const r = await cropDocument(SAMPLE_IMG); - expect(existsSync(r.outPath)).toBe(true); - expect(r.width).toBeGreaterThan(100); - }, T); + it( + 'cropDocument writes a perspective-corrected PNG', + async () => { + const r = await cropDocument(SAMPLE_IMG); + expect(existsSync(r.outPath)).toBe(true); + expect(r.width).toBeGreaterThan(100); + }, + T + ); - it('extractForeground writes a cutout (macOS 14+)', async () => { - const caps = await visionCapabilities(); - if (!caps.features.foregroundMask) return; - const r = await extractForeground(SAMPLE_IMG, { tight: true }); - expect(r.instances).toBeGreaterThanOrEqual(1); - expect(existsSync(r.outPath)).toBe(true); - }, T); + it( + 'extractForeground writes a cutout (macOS 14+)', + async () => { + const caps = await visionCapabilities(); + if (!caps.features.foregroundMask) return; + const r = await extractForeground(SAMPLE_IMG, { tight: true }); + expect(r.instances).toBeGreaterThanOrEqual(1); + expect(existsSync(r.outPath)).toBe(true); + }, + T + ); +}); + +describe('helper error reporting', () => { + // A helper failure must surface its own message, not Node's + // "Command failed: /long/path/to/vision-helper --flags …". + it('surfaces the helper ERROR line, not the spawn command', async () => { + await expect(imageInfo('/nonexistent-xyz.png')).rejects.toThrow(/^Cannot open file:/); + }); + + it('keeps the exit status so gating still works', async () => { + await imageInfo('/nonexistent-xyz.png').catch((err) => { + expect((err as { code?: number }).code).toBe(1); + }); + }); + + it('captureScreen surfaces screencapture stderr', async () => { + await expect(captureScreen({ rect: { x: 0, y: 0, w: -5, h: -5 } })).rejects.toThrow( + /does not intersect any displays/ + ); + }); +}); + +describe('capability gating', () => { + // The helper reports a feature only when both its SDK and this macOS provide it; + // anything reported unavailable must fail with the typed error, not a raw crash. + it.each([ + ['documentStructure', () => recognizeDocument(SAMPLE_IMG)], + ['aesthetics', () => imageAesthetics(SAMPLE_IMG)], + ['foregroundMask', () => extractForeground(SAMPLE_IMG)], + ['animalPose', () => detectAnimalPose(SAMPLE_IMG)], + ])( + 'caps.%s agrees with the call', + async (feature, call) => { + const caps = await visionCapabilities(); + if (caps.features[feature]) { + await expect(call()).resolves.toBeDefined(); + } else { + await expect(call()).rejects.toBeInstanceOf(UnsupportedOnThisMacOSError); + } + }, + 60_000 + ); }); describe('ui-helper', () => { From cd4f38d715ad2ffdf9bb0b65507406d6df1970c2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 12:21:49 +0200 Subject: [PATCH 5/7] fix: restore macOS 12 / Node 18 support, stabilise test suite MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review of feat/ui-helper found three blockers before release: - The branch dropped macOS 12 (native target 12→13) and Node 18 under a `minor` changeset, so a `^1.5.0` consumer on macOS 12 would auto-upgrade into binaries that cannot launch. The platform drop traced to a single un-gated `automaticallyDetectsLanguage` call (macOS 13+); wrapping it in `if #available` — the pattern the file already uses for 14/15/26 — lets all three helpers build at macos12 with every gated feature still enabled. The Node bump was unrelated to this branch (`fetch`/`AbortSignal.timeout` in ollama.ts predate it and ran fine on 18). - The suite was red: two pre-existing detectDocument tests timed out at the default 5s when the new model-bound tests ran alongside them, though they pass in isolation and the helper itself takes 0.12s. Added vitest.config.ts (30s timeout, no file parallelism, worktree copies excluded): 82/82 pass. - Release builds now refuse an SDK too old to compile in every gated feature, instead of silently shipping binaries that report them unavailable. Also: imageAesthetics uses runGated() like its siblings instead of hand-checking the exit code; the OCR cache key includes helper and macOS version so entries do not survive an upgrade that changes results; corrected the comment above unROI, which claimed the opposite of what the code does (verified empirically that Vision reports ROI-relative coordinates). Co-Authored-By: Claude Opus 5 --- .changeset/ui-helper-and-extended-vision.md | 4 ++-- README.md | 2 +- package.json | 2 +- scripts/build-native-cross.js | 18 ++++++++++++++++-- src/index.ts | 13 ++++++++++--- src/native/vision-helper.swift | 12 ++++++++---- src/vision.ts | 14 ++++++-------- vitest.config.ts | 17 +++++++++++++++++ 8 files changed, 61 insertions(+), 21 deletions(-) create mode 100644 vitest.config.ts diff --git a/.changeset/ui-helper-and-extended-vision.md b/.changeset/ui-helper-and-extended-vision.md index 4c575c3..8778b44 100644 --- a/.changeset/ui-helper-and-extended-vision.md +++ b/.changeset/ui-helper-and-extended-vision.md @@ -10,6 +10,6 @@ feat: ui-helper (windows / displays / permissions / captureScreen) and extended - `extractEntities` (links, e-mails, phones, addresses, dates), `detectTextRegions`, `compareImages`, `imageInfo`, `visionCapabilities`, `supportedOcrLanguages`. - People & scenes: `detectFaceLandmarks`, `detectHumans`, `detectBodyPose`, `detectHandPose`, `detectAnimals`, `detectAnimalPose`, `detectHorizon`, `detectSaliency`, `detectContours`, `imageAesthetics`, `detectLensSmudge`. - Pixel ops that return paths: `cropImage`, `cropDocument` (perspective-corrected), `extractForeground`, `personMask`. -- Native helpers now target macOS 13. Modern features are gated twice — at compile time on the SDK (`-DSDK_14/15/26`, set automatically by the build scripts) and at runtime on `#available` — so the `swiftc` fallback also builds on older macOS, with unavailable features reported as `false` by `visionCapabilities()` and raising `UnsupportedOnThisMacOSError`. +- Native helpers still target macOS 12. Modern features are gated twice — at compile time on the SDK (`-DSDK_14/15/26`, set automatically by the build scripts) and at runtime on `#available` — so the `swiftc` fallback also builds on older macOS, with unavailable features reported as `false` by `visionCapabilities()` and raising `UnsupportedOnThisMacOSError`. Release builds refuse to run on an SDK too old to compile in every gated feature. - Helper failures now surface their own message (`Cannot open file: …`) instead of Node's `Command failed: /path/to/vision-helper …`. -- Node >= 20. +- Opt-in OCR cache is keyed on helper and macOS version, so results never survive an upgrade that changes them. diff --git a/README.md b/README.md index ca46bd6..8b98983 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ Uses macOS's built-in [Vision framework](https://developer.apple.com/documentati ## Requirements -- macOS 13+ (Apple Silicon or Intel) — some features need newer macOS (see `visionCapabilities()`) +- macOS 12+ (Apple Silicon or Intel) — some features need newer macOS (see `visionCapabilities()`) - Node.js 20+ - [Ollama](https://ollama.com) running locally — only if you use the Markdown pipeline - Xcode Command Line Tools (`xcode-select --install`) — **only** needed as an offline fallback when prebuilt binaries cannot be downloaded diff --git a/package.json b/package.json index f8a4993..8ae267e 100644 --- a/package.json +++ b/package.json @@ -87,7 +87,7 @@ "darwin" ], "engines": { - "node": ">=20.0.0" + "node": ">=18.0.0" }, "devDependencies": { "@changesets/cli": "^2.31.0", diff --git a/scripts/build-native-cross.js b/scripts/build-native-cross.js index fc21658..a5d9829 100644 --- a/scripts/build-native-cross.js +++ b/scripts/build-native-cross.js @@ -8,8 +8,8 @@ const __dirname = path.dirname(fileURLToPath(import.meta.url)); const root = path.resolve(__dirname, '..'); const TARGETS = [ - { arch: 'arm64', swift: 'arm64-apple-macos13' }, - { arch: 'x64', swift: 'x86_64-apple-macos13' }, + { arch: 'arm64', swift: 'arm64-apple-macos12' }, + { arch: 'x64', swift: 'x86_64-apple-macos12' }, ]; const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; @@ -28,6 +28,20 @@ function sdkDefines() { const DEFINES = sdkDefines().join(' '); +// Release artifacts must carry every gated feature. Building on an older SDK +// would silently ship binaries where documentStructure/aesthetics report false +// on machines that actually support them — fail loudly instead. +const REQUIRED_DEFINES = ['-DSDK_14', '-DSDK_15', '-DSDK_26']; +const missing = REQUIRED_DEFINES.filter((d) => !DEFINES.includes(d)); +if (missing.length) { + console.error( + `❌ macos-vision: SDK too old for a release build — missing ${missing.join(', ')}.\n` + + ` Features gated behind those SDKs would be compiled out of the published binaries.\n` + + ` Install a newer Xcode (or set SKIP_SDK_CHECK=1 for a local, non-release build).` + ); + if (!process.env.SKIP_SDK_CHECK) process.exit(1); +} + for (const { arch, swift } of TARGETS) { const outDir = path.join(root, 'bin', `darwin-${arch}`); mkdirSync(outDir, { recursive: true }); diff --git a/src/index.ts b/src/index.ts index 7a80aba..349ce09 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2,7 +2,7 @@ import { resolve, dirname, extname } from 'path'; import { open, readFile, writeFile, mkdir } from 'fs/promises'; import { homedir } from 'os'; import { VISION_BIN, PDF_BIN, execHelper, runHelper, fileSha256, sha256 } from './helper.js'; -import { textOptionArgs } from './vision.js'; +import { textOptionArgs, visionCapabilities } from './vision.js'; import type { TextRecognitionOptions } from './vision.js'; const BINARY_TIMEOUT_MS = 30_000; @@ -122,13 +122,20 @@ async function ocrPdf( // ─── OCR result cache ──────────────────────────────────────────────────────── -/** Content hash + the canonical helper argv, so option order does not matter. */ +/** + * Content hash + the canonical helper argv, so option order does not matter. + * The helper and macOS versions are part of the key: both change what Vision + * returns for identical input, so entries must not survive an upgrade. + */ async function cacheKey( absPath: string, format: string, opts: TextRecognitionOptions ): Promise { - return sha256((await fileSha256(absPath)) + JSON.stringify([format, ...textOptionArgs(opts)])); + const [hash, caps] = await Promise.all([fileSha256(absPath), visionCapabilities()]); + return sha256( + hash + JSON.stringify([caps.helperVersion, caps.macosVersion, format, ...textOptionArgs(opts)]) + ); } async function readCache(key: string): Promise { diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index c1885f5..56ed4cf 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -388,9 +388,11 @@ let ocrNoCorrection = args.contains("--no-correction") let ocrFast = args.contains("--fast") let ocrMinHeight: Float? = optValue("--min-text-height").flatMap { Float($0) } -// Region of interest: normalized x,y,w,h with TOP-LEFT origin (same space as our output). -// Vision wants bottom-left origin, so flip. Results stay relative to the full image -// because the handler reports observations in full-image coordinates. +// Region of interest: normalized x,y,w,h with TOP-LEFT origin (same space as our +// output). Vision wants bottom-left origin, so flip. Vision then reports +// observations RELATIVE TO THE ROI, so every bounding box must go back through +// unROI() to land in full-image space — verified empirically: with and without an +// ROI the same text yields identical coordinates only when unROI is applied. var roiRect: CGRect? = nil if let roi = optValue("--roi") { let p = roi.split(separator: ",").compactMap { Double($0) } @@ -436,7 +438,9 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is } request.recognitionLevel = ocrFast ? .fast : .accurate if !ocrLanguages.isEmpty { request.recognitionLanguages = ocrLanguages } - if ocrAutoLang { request.automaticallyDetectsLanguage = true } + if ocrAutoLang { + if #available(macOS 13.0, *) { request.automaticallyDetectsLanguage = true } + } request.usesLanguageCorrection = !ocrNoCorrection if !ocrCustomWords.isEmpty { request.customWords = ocrCustomWords } if let mh = ocrMinHeight { request.minimumTextHeight = mh } diff --git a/src/vision.ts b/src/vision.ts index f00744c..5d3cb3b 100644 --- a/src/vision.ts +++ b/src/vision.ts @@ -277,14 +277,12 @@ export interface AestheticsScore { } /** Photo aesthetics + utility flag (macOS 15+). Good for "is this a screenshot or a photo?". */ -export async function imageAesthetics(imagePath: string): Promise { - try { - return await run(['--aesthetics', resolve(imagePath)]); - } catch (err) { - if ((err as { code?: number }).code === 2) - throw new UnsupportedOnThisMacOSError('imageAesthetics', '15'); - throw err; - } +export function imageAesthetics(imagePath: string): Promise { + return gated( + ['--aesthetics', resolve(imagePath)], + 'imageAesthetics', + '15' + ); } export interface Horizon { diff --git a/vitest.config.ts b/vitest.config.ts new file mode 100644 index 0000000..b186783 --- /dev/null +++ b/vitest.config.ts @@ -0,0 +1,17 @@ +import { defineConfig } from 'vitest/config'; + +export default defineConfig({ + test: { + // Every test spawns a native helper that runs a Vision model. Under the + // default 5s timeout these pass in isolation but flake when the suite runs + // in parallel and saturates the machine. + testTimeout: 30_000, + hookTimeout: 30_000, + // Vision requests are already CPU/ANE-bound; running many files at once + // buys nothing and is what pushed the suite over its timeouts. + fileParallelism: false, + // The repo's own worktrees contain copies of these tests — running them + // twice doubles an already slow, model-bound suite. + exclude: ['**/node_modules/**', '**/dist/**', '**/.claude/worktrees/**'], + }, +}); From 1f71bccd7e6400885bd9f22bfa87391b3d5ad2b3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 20:34:25 +0200 Subject: [PATCH 6/7] feat: report screenLocked and diagnose locked-screen capture failures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Keeps this package in step with the consumer that will eventually drop its own copy of ui-helper. On a locked Mac, window and region capture fail outright and a full-screen capture returns only the lock screen — verified against a genuinely locked session. checkPermissions now reports screenLocked so a caller can check before starting, and captureScreen consults it on failure to say plainly that retrying cannot succeed until someone unlocks. Co-Authored-By: Claude Opus 5 --- .changeset/ui-helper-and-extended-vision.md | 2 +- README.md | 2 +- src/native/ui-helper.swift | 14 ++++++++++++-- src/ui.ts | 17 +++++++++++++++-- 4 files changed, 29 insertions(+), 6 deletions(-) diff --git a/.changeset/ui-helper-and-extended-vision.md b/.changeset/ui-helper-and-extended-vision.md index 8778b44..39523b0 100644 --- a/.changeset/ui-helper-and-extended-vision.md +++ b/.changeset/ui-helper-and-extended-vision.md @@ -4,7 +4,7 @@ feat: ui-helper (windows / displays / permissions / captureScreen) and extended Vision API -- New native `ui-helper` (third prebuilt binary): `listWindows`, `listDisplays`, `checkPermissions`, `captureScreen` (returns path + geometry + sha256, never bytes). +- New native `ui-helper` (third prebuilt binary): `listWindows`, `listDisplays`, `checkPermissions`, `captureScreen` (returns path + geometry + sha256, never bytes). `checkPermissions` also reports `screenLocked`; capture failures consult it so a locked Mac is diagnosed outright rather than guessed at. - OCR tuning: `languages`, `autoDetectLanguage`, `languageCorrection`, `customWords`, `fast`, `regionOfInterest`, `minTextHeight`; opt-in content-hash `cache`; `onProgress` for PDFs. - `recognizeDocument` (macOS 26+): native paragraphs, tables, lists, title, barcodes and detected data with positions. - `extractEntities` (links, e-mails, phones, addresses, dates), `detectTextRegions`, `compareImages`, `imageInfo`, `visionCapabilities`, `supportedOcrLanguages`. diff --git a/README.md b/README.md index 8b98983..f55d00c 100644 --- a/README.md +++ b/README.md @@ -243,7 +243,7 @@ Read-only introspection of the desktop plus PNG captures, meant as the "eyes" of ```js import { listWindows, listDisplays, checkPermissions, captureScreen } from 'macos-vision'; -const perms = await checkPermissions(); // { screenRecording, accessibility } +const perms = await checkPermissions(); // { screenRecording, accessibility, screenLocked } const displays = await listDisplays(); // bounds in screen points + backing scale const windows = await listWindows(); // on-screen app windows, front-to-back diff --git a/src/native/ui-helper.swift b/src/native/ui-helper.swift index 98e462d..e23f984 100644 --- a/src/native/ui-helper.swift +++ b/src/native/ui-helper.swift @@ -30,6 +30,15 @@ struct DisplayInfo: Codable { struct PermissionsInfo: Codable { let screenRecording: Bool let accessibility: Bool + /// True while the login session is locked. Verified behaviour in that state: + /// window and region capture fail outright, and a full-screen capture returns + /// only the lock screen — so no capture is useful until the user unlocks. + let screenLocked: Bool +} + +func isScreenLocked() -> Bool { + guard let session = CGSessionCopyCurrentDictionary() as? [String: Any] else { return false } + return session["CGSSessionScreenIsLocked"] as? Bool ?? false } func encodeJSON(_ value: T) -> String { @@ -45,7 +54,8 @@ let args = CommandLine.arguments if args.contains("--permissions") { let info = PermissionsInfo( screenRecording: CGPreflightScreenCaptureAccess(), - accessibility: AXIsProcessTrusted() + accessibility: AXIsProcessTrusted(), + screenLocked: isScreenLocked() ) print(encodeJSON(info)) exit(0) @@ -78,12 +88,12 @@ if args.contains("--displays") { } if args.contains("--windows") { + let includeAll = args.contains("--all") let options: CGWindowListOption = [.optionOnScreenOnly, .excludeDesktopElements] guard let list = CGWindowListCopyWindowInfo(options, kCGNullWindowID) as? [[String: Any]] else { print("[]") exit(0) } - let includeAll = args.contains("--all") var results: [WindowInfo] = [] for w in list { guard let boundsDict = w[kCGWindowBounds as String] as? [String: Double] else { continue } diff --git a/src/ui.ts b/src/ui.ts index 7d9631a..5142b3b 100644 --- a/src/ui.ts +++ b/src/ui.ts @@ -45,6 +45,8 @@ export interface DisplayInfo { export interface PermissionsInfo { screenRecording: boolean; accessibility: boolean; + /** True while the login session is locked — no capture is useful until unlocked. */ + screenLocked: boolean; } /** Rectangle in global screen points, top-left origin (CGEvent click space). */ @@ -186,9 +188,20 @@ export async function captureScreen(opts: CaptureOptions = {}): Promise p.screenLocked) + .catch(() => false); throw new Error( - `screencapture failed for ${targetDesc}${stderr ? ` (${stderr})` : ''}. ` + - 'Common causes: the screen is locked or asleep, or the window was closed.' + locked + ? `screencapture failed for ${targetDesc}: the screen is locked. ` + + 'Window and region capture do not work on a locked Mac, and a full-screen ' + + 'capture would only show the lock screen. Ask the user to unlock, then retry — ' + + 'retrying while locked cannot succeed.' + : `screencapture failed for ${targetDesc}${stderr ? ` (${stderr})` : ''}. ` + + 'The window may have closed, or the display may be asleep.' ); } let bytes: Buffer; From f91d15cf72083c60d43d6a241d5e714bb49e8a7d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adrian=20Wo=C5=82czuk?= Date: Sat, 22 Aug 2026 20:39:56 +0200 Subject: [PATCH 7/7] fix: report an unusable lens-smudge model as unsupported, not an error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit detectLensSmudge documents that it returns `supported: false` rather than guessing, and the helper already detected hardware without the model by sniffing VisionCore's stdout notice. CI runners fail differently: the request throws Foundation._GenericObjCError.nilError, which took the fail() path and turned a "this Mac cannot do it" into a hard error — turning the suite red on GitHub while passing locally. The image has already loaded by that point, so a throw here can only mean the model is unusable on this machine. Report it the same way as the notice path. Co-Authored-By: Claude Opus 5 --- src/native/vision-helper.swift | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index 56ed4cf..9e097ad 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -613,7 +613,12 @@ func runSmudge() -> SmudgeResult { close(savedStdout) let noise = (try? String(contentsOfFile: tmp, encoding: .utf8)) ?? "" unlink(tmp) - if let e = failure { fail("Vision lens smudge detection failed: \(e.localizedDescription)") } + // A throw here means the smudge model is not usable on this machine — the + // image already loaded, and CI runners without the model raise + // Foundation._GenericObjCError.nilError rather than logging the notice below. + // Report it as unsupported, which is this request's documented contract, + // instead of failing the whole call. + if failure != nil { return SmudgeResult(confidence: 0, supported: false) } let supported = !noise.contains("Unable to find") return SmudgeResult(confidence: conf, supported: supported) }