diff --git a/.changeset/ui-helper-and-extended-vision.md b/.changeset/ui-helper-and-extended-vision.md new file mode 100644 index 0000000..39523b0 --- /dev/null +++ b/.changeset/ui-helper-and-extended-vision.md @@ -0,0 +1,15 @@ +--- +"macos-vision": minor +--- + +feat: ui-helper (windows / displays / permissions / captureScreen) and extended Vision API + +- New native `ui-helper` (third prebuilt binary): `listWindows`, `listDisplays`, `checkPermissions`, `captureScreen` (returns path + geometry + sha256, never bytes). `checkPermissions` also reports `screenLocked`; capture failures consult it so a locked Mac is diagnosed outright rather than guessed at. +- OCR tuning: `languages`, `autoDetectLanguage`, `languageCorrection`, `customWords`, `fast`, `regionOfInterest`, `minTextHeight`; opt-in content-hash `cache`; `onProgress` for PDFs. +- `recognizeDocument` (macOS 26+): native paragraphs, tables, lists, title, barcodes and detected data with positions. +- `extractEntities` (links, e-mails, phones, addresses, dates), `detectTextRegions`, `compareImages`, `imageInfo`, `visionCapabilities`, `supportedOcrLanguages`. +- People & scenes: `detectFaceLandmarks`, `detectHumans`, `detectBodyPose`, `detectHandPose`, `detectAnimals`, `detectAnimalPose`, `detectHorizon`, `detectSaliency`, `detectContours`, `imageAesthetics`, `detectLensSmudge`. +- Pixel ops that return paths: `cropImage`, `cropDocument` (perspective-corrected), `extractForeground`, `personMask`. +- Native helpers still target macOS 12. Modern features are gated twice — at compile time on the SDK (`-DSDK_14/15/26`, set automatically by the build scripts) and at runtime on `#available` — so the `swiftc` fallback also builds on older macOS, with unavailable features reported as `false` by `visionCapabilities()` and raising `UnsupportedOnThisMacOSError`. Release builds refuse to run on an SDK too old to compile in every gated feature. +- Helper failures now surface their own message (`Cannot open file: …`) instead of Node's `Command failed: /path/to/vision-helper …`. +- Opt-in OCR cache is keyed on helper and macOS version, so results never survive an upgrade that changes them. diff --git a/README.md b/README.md index 08f48ff..f55d00c 100644 --- a/README.md +++ b/README.md @@ -6,8 +6,8 @@ Uses macOS's built-in [Vision framework](https://developer.apple.com/documentati ## Requirements -- macOS 12+ (Apple Silicon or Intel) -- Node.js 18+ +- macOS 12+ (Apple Silicon or Intel) — some features need newer macOS (see `visionCapabilities()`) +- Node.js 20+ - [Ollama](https://ollama.com) running locally — only if you use the Markdown pipeline - Xcode Command Line Tools (`xcode-select --install`) — **only** needed as an offline fallback when prebuilt binaries cannot be downloaded @@ -17,7 +17,7 @@ Uses macOS's built-in [Vision framework](https://developer.apple.com/documentati npm install macos-vision ``` -The native Swift binaries (`vision-helper`, `pdf-helper`) are downloaded as prebuilt artifacts from the matching GitHub Release (signed by SHA-256). If the download fails (no network, custom registry, unpublished version), the postinstall falls back to compiling locally with `swiftc` — that's the only path that needs Xcode Command Line Tools. Set `MACOS_VISION_SKIP_DOWNLOAD=1` to force local compilation. +The native Swift binaries (`vision-helper`, `pdf-helper`, `ui-helper`) are downloaded as prebuilt artifacts from the matching GitHub Release (signed by SHA-256). If the download fails (no network, custom registry, unpublished version), the postinstall falls back to compiling locally with `swiftc` — that's the only path that needs Xcode Command Line Tools. Set `MACOS_VISION_SKIP_DOWNLOAD=1` to force local compilation. ## What you get @@ -28,6 +28,12 @@ The native Swift binaries (`vision-helper`, `pdf-helper`) are downloaded as preb | Image classification | Apple Vision | offline | | Layout inference (lines, paragraphs, reading order) | heuristic in TypeScript | offline | | PDF rasterization | PDFKit (`pdf-helper`) | offline | +| Screen capture + window / display / permission introspection | `screencapture` + CoreGraphics (`ui-helper`) | offline | +| Document structure — paragraphs, tables, lists, detected data (macOS 26+) | Apple Vision `RecognizeDocumentsRequest` | offline | +| Entities in text — links, e-mails, phones, addresses, dates | Foundation `NSDataDetector` | offline | +| Image similarity, saliency, contours, text regions | Apple Vision | offline | +| People — face landmarks, body / hand pose, person masks, subject cutout | Apple Vision | offline | +| Crop / deskew / perspective-correct documents | CoreImage | offline | | **Image / PDF → Markdown** | Apple Vision OCR + local LLM via Ollama | local LLM call | --- @@ -144,6 +150,117 @@ for (const block of layout) { --- +## API — Extended Vision + +Everything below follows the same conventions: normalized 0–1 coordinates with a **top-left origin**, results as JSON, and pixel-producing operations **write PNG files and return paths** — never image bytes. Check `visionCapabilities()` first: features gate on the macOS version. + +```js +import { + visionCapabilities, supportedOcrLanguages, ocr, + recognizeDocument, extractEntities, detectTextRegions, compareImages, imageInfo, + detectFaceLandmarks, detectHumans, detectBodyPose, detectHandPose, detectAnimals, + detectSaliency, detectContours, detectHorizon, imageAesthetics, detectLensSmudge, + cropImage, cropDocument, extractForeground, personMask, +} from 'macos-vision'; + +const caps = await visionCapabilities(); +// { helperVersion, macosVersion, ocrLanguages: ['en-US','pl-PL',…], features: { documentStructure, foregroundMask, … } } +``` + +A feature is reported `true` only when both this macOS **and** the SDK the helper was built against provide it. Anything reported `false` raises `UnsupportedOnThisMacOSError` rather than failing obscurely, so an agent can branch on `caps.features` before planning work. + +### OCR tuning + +`ocr()` now accepts Vision's recognition knobs. Results for `regionOfInterest` are still reported in full-image coordinates. + +```js +const blocks = await ocr('invoice.png', { + format: 'blocks', + languages: ['pl-PL', 'en-US'], // priority order; see supportedOcrLanguages() + languageCorrection: false, // keep IBANs, IDs and hashes verbatim + customWords: ['Prorok', 'FV/2026/08'], + regionOfInterest: { x: 0, y: 0, width: 1, height: 0.25 }, + fast: false, // true → quicker, less accurate + cache: true, // ~/.cache/macos-vision/ocr, keyed by file sha256 + options +}); +await ocr('long.pdf', { onProgress: (done, total) => console.log(`${done}/${total}`) }); +``` + +### Document structure (macOS 26+) + +Native layout understanding — no heuristics, no LLM. Throws `UnsupportedOnThisMacOSError` on older systems. + +```js +const doc = await recognizeDocument('invoice.png', { languages: ['pl-PL'] }); +doc.title?.text; // 'Faktura VAT' +doc.paragraphs[0].lines; // [{ text, confidence, bbox }] +doc.tables[0].rows; // string[][] — cell texts by row +doc.tables[0].cells; // [{ text, row, col, rowSpan, colSpan, bbox }] +doc.lists[0].items; // [{ marker, text, bbox }] +doc.detectedData; // [{ type: 'money' | 'date' | 'email' | 'phone' | 'link' | …, text, value, bbox }] +``` + +### Text utilities + +```js +await extractEntities(text); // links / e-mails / phones / addresses / dates with offsets — any macOS +await detectTextRegions('shot.png'); // where text is, without reading it (fast ROI picker) +await compareImages('before.png', 'after.png'); // { distance } — 0 identical, > ~0.8 different content +await imageInfo('photo.jpg'); // { width, height, dpi, format, orientation, … } +``` + +### People, scenes, quality + +```js +await detectFaceLandmarks('photo.jpg'); // bbox + roll/yaw/pitch + captureQuality + landmark polylines +await detectHumans('photo.jpg'); // full-body boxes +await detectBodyPose('photo.jpg'); // { joints: { left_wrist_joint: { x, y, confidence }, … } } +await detectHandPose('photo.jpg'); // + chirality +await detectAnimals('photo.jpg'); // cats & dogs with labels +await detectAnimalPose('photo.jpg'); // macOS 14+ +await detectSaliency('photo.jpg', { mode: 'attention' | 'objectness', heatmapPath: 'heat.png' }); +await detectContours('chart.png', { maxPoints: 32 }); +await detectHorizon('landscape.jpg'); // { angleDegrees } | null +await imageAesthetics('photo.jpg'); // { overallScore, isUtility } — macOS 15+; isUtility = screenshot/receipt-like +await detectLensSmudge('photo.jpg'); // { confidence, supported } — macOS 26+, supported:false when the model is absent +``` + +### Pixel operations (return paths) + +```js +await cropImage('shot.png', { x: 0.5, y: 0, width: 0.5, height: 0.3 }); // zoom for a second OCR pass +await cropDocument('receipt-photo.jpg'); // detect + perspective-correct + deskew +await extractForeground('product.jpg', { tight: true }); // subject cutout with alpha (macOS 14+) +await personMask('photo.jpg'); // 8-bit mask, white = person +``` + +--- + +## API — UI (screen capture, windows, permissions) + +Read-only introspection of the desktop plus PNG captures, meant as the "eyes" of UI-testing agents. Requires **Screen Recording** permission for the host process (System Settings → Privacy & Security → Screen Recording). + +```js +import { listWindows, listDisplays, checkPermissions, captureScreen } from 'macos-vision'; + +const perms = await checkPermissions(); // { screenRecording, accessibility, screenLocked } +const displays = await listDisplays(); // bounds in screen points + backing scale +const windows = await listWindows(); // on-screen app windows, front-to-back + +// Capture the frontmost Safari window (app name: exact or case-insensitive prefix) +const shot = await captureScreen({ app: 'Safari' }); +// { path, pixelWidth, pixelHeight, sha256, frame: { x, y, w, h }, scale, capturedAt, target } + +// Other targets: { windowId }, { rect: { x, y, w, h } }, { displayId } (default: main display) +const region = await captureScreen({ rect: { x: 0, y: 0, w: 800, h: 600 }, outPath: './region.png' }); +``` + +All coordinates are **global screen points with a top-left origin** — the same space `CGEvent` clicks use — so `frame` + an OCR block's normalized bbox maps straight to a click point. + +**Privacy invariant:** these functions return paths, geometry, and text — never image bytes. Captures are written to disk (`$TMPDIR/macos-vision/` unless `outPath` is given) and the caller owns cleanup. The library never synthesizes input: *eyes, not hands*. + +--- + ## API — Markdown pipeline (VisionScribe) `VisionScribe` converts an image or PDF to Markdown by combining Apple Vision OCR with a local LLM (via Ollama). The LLM never sees the image — it only formats text that Vision already extracted. This keeps image processing local and reduces the risk of vision-model hallucinations, but Markdown reconstruction is still best-effort and depends on the local model and document complexity. @@ -268,6 +385,15 @@ The `VisionScribe` API, the system prompt, and the chunking strategy are unchang | `options.format` | `'text' \| 'blocks'` | `'text'` | Plain text or structured blocks with coordinates | | `options.startPage` | `number` | `1` | PDFs only — first page to OCR, 1-based. Ignored for images. | | `options.maxPages` | `number` | all | PDFs only — maximum number of pages to OCR. Ignored for images. | +| `options.onProgress` | `(done, total) => void` | — | PDFs only — called after each page. | +| `options.languages` | `string[]` | Vision default | BCP-47 codes in priority order. | +| `options.autoDetectLanguage` | `boolean` | `false` | Let Vision pick the language per run. | +| `options.languageCorrection` | `boolean` | `true` | Disable for codes, IDs, IBANs. | +| `options.customWords` | `string[]` | — | Vocabulary that overrides the language model. | +| `options.fast` | `boolean` | `false` | `.fast` recognition level. | +| `options.regionOfInterest` | `{ x, y, width, height }` | whole image | Normalized, top-left origin; output stays in full-image space. | +| `options.minTextHeight` | `number` | — | Ignore text shorter than this fraction of image height. | +| `options.cache` | `boolean` | `false` | Cache by content hash + options in `~/.cache/macos-vision/ocr`. | Returns `Promise` or `Promise`. diff --git a/scripts/build-native-cross.js b/scripts/build-native-cross.js index f5aab7f..a5d9829 100644 --- a/scripts/build-native-cross.js +++ b/scripts/build-native-cross.js @@ -11,7 +11,36 @@ const TARGETS = [ { arch: 'arm64', swift: 'arm64-apple-macos12' }, { arch: 'x64', swift: 'x86_64-apple-macos12' }, ]; -const HELPERS = ['vision-helper', 'pdf-helper']; +const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; + +// Symbols added in newer SDKs are absent when building against an older one, so +// the helper gates them on -DSDK_nn. Detect what this machine's SDK provides. +function sdkDefines() { + try { + const raw = execSync('xcrun --sdk macosx --show-sdk-version', { encoding: 'utf8' }).trim(); + const major = parseInt(raw.split('.')[0], 10); + if (!Number.isFinite(major)) return []; + return [14, 15, 26].filter((n) => major >= n).map((n) => `-DSDK_${n}`); + } catch { + return []; + } +} + +const DEFINES = sdkDefines().join(' '); + +// Release artifacts must carry every gated feature. Building on an older SDK +// would silently ship binaries where documentStructure/aesthetics report false +// on machines that actually support them — fail loudly instead. +const REQUIRED_DEFINES = ['-DSDK_14', '-DSDK_15', '-DSDK_26']; +const missing = REQUIRED_DEFINES.filter((d) => !DEFINES.includes(d)); +if (missing.length) { + console.error( + `❌ macos-vision: SDK too old for a release build — missing ${missing.join(', ')}.\n` + + ` Features gated behind those SDKs would be compiled out of the published binaries.\n` + + ` Install a newer Xcode (or set SKIP_SDK_CHECK=1 for a local, non-release build).` + ); + if (!process.env.SKIP_SDK_CHECK) process.exit(1); +} for (const { arch, swift } of TARGETS) { const outDir = path.join(root, 'bin', `darwin-${arch}`); @@ -20,7 +49,7 @@ for (const { arch, swift } of TARGETS) { for (const name of HELPERS) { const src = path.join(root, 'src', 'native', `${name}.swift`); const out = path.join(outDir, name); - execSync(`swiftc -O -target ${swift} "${src}" -o "${out}"`, { stdio: 'inherit' }); + execSync(`swiftc -O -target ${swift} ${DEFINES} "${src}" -o "${out}"`, { stdio: 'inherit' }); } const tarball = `bin-darwin-${arch}.tar.gz`; diff --git a/scripts/install-native.js b/scripts/install-native.js index 59c8633..3bc37e0 100644 --- a/scripts/install-native.js +++ b/scripts/install-native.js @@ -17,7 +17,21 @@ const root = path.resolve(__dirname, '..'); const pkg = JSON.parse(readFileSync(path.join(root, 'package.json'), 'utf8')); const binDir = path.join(root, 'bin'); -const HELPERS = ['vision-helper', 'pdf-helper']; +const HELPERS = ['vision-helper', 'pdf-helper', 'ui-helper']; + +// Symbols added in newer SDKs are absent when building against an older one, so +// the helper gates them on -DSDK_nn. Detect what this machine's SDK provides. +function sdkDefines() { + try { + const raw = execSync('xcrun --sdk macosx --show-sdk-version', { encoding: 'utf8' }).trim(); + const major = parseInt(raw.split('.')[0], 10); + if (!Number.isFinite(major)) return []; + return [14, 15, 26].filter((n) => major >= n).map((n) => `-DSDK_${n}`); + } catch { + return []; + } +} + // 0. Skip if all binaries already exist (cached install) if (HELPERS.every((h) => existsSync(path.join(binDir, h)))) { @@ -85,11 +99,12 @@ function fallbackToSwiftc() { process.exit(1); } mkdirSync(binDir, { recursive: true }); + const defines = sdkDefines().join(' '); for (const h of HELPERS) { const src = path.join(root, 'src', 'native', `${h}.swift`); const out = path.join(binDir, h); try { - execSync(`swiftc -O "${src}" -o "${out}"`, { stdio: 'inherit' }); + execSync(`swiftc -O ${defines} "${src}" -o "${out}"`, { stdio: 'inherit' }); console.log(`✅ macos-vision: compiled ${h} locally`); } catch { console.error(`❌ macos-vision: ${h} compilation failed.`); diff --git a/src/cli.ts b/src/cli.ts index a8f15b5..067523f 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -15,7 +15,17 @@ import { DocumentBounds, classify, Classification, + recognizeDocument, + extractEntities, + detectTextRegions, + detectHumans, + detectFaceLandmarks, + detectSaliency, + imageAesthetics, + imageInfo, + visionCapabilities, } from './index.js'; +import type { TextRecognitionOptions } from './index.js'; const USAGE = ` Usage: macos-vision [options] @@ -30,6 +40,27 @@ Vision options: --classify Image classification --all Run all of the above +OCR tuning: + --lang Recognition languages, comma-separated BCP-47 (e.g. pl-PL,en-US) + --auto-lang Let Vision detect the language automatically + --no-correction Disable language-model correction (IDs, codes, IBANs) + --custom-words Domain vocabulary that should win over the language model + --fast Fast recognition level (quicker, less accurate) + --roi x,y,w,h Only read inside this normalized region (top-left origin) + --cache Cache OCR results by file hash in ~/.cache/macos-vision/ocr + +Extended analysis: + --structure Document structure: paragraphs, tables, lists, data (macOS 26+) + --entities Links / e-mails / phones / addresses / dates found in OCR text + --text-regions Where text is, without reading it + --humans Person bounding boxes + --face-landmarks Face landmarks, head pose, capture quality + --saliency Attention-based salient regions + --aesthetics Aesthetics score + utility flag (macOS 15+) + --info Pixel dimensions and image metadata + --capabilities What this machine supports (no input file needed) + --languages Supported OCR languages (no input file needed) + PDF page range (PDFs only; ignored for images): --start-page First page to process, 1-based (default: 1) --max-pages Maximum number of pages to process (default: all) @@ -75,6 +106,9 @@ const ollamaUrl = takeOpt('--ollama-url', argv); const outPath = takeOpt('-o', argv) ?? takeOpt('--output', argv); const startPageRaw = takeOpt('--start-page', argv); const maxPagesRaw = takeOpt('--max-pages', argv); +const langRaw = takeOpt('--lang', argv); +const customWordsRaw = takeOpt('--custom-words', argv); +const roiRaw = takeOpt('--roi', argv); function parsePageOpt(name: string, raw: string | undefined): number | undefined { if (raw === undefined) return undefined; @@ -95,6 +129,30 @@ if (maxPages !== undefined) pageRange.maxPages = maxPages; const flags = new Set(argv.filter((a) => a.startsWith('--'))); const fileArgs = argv.filter((a) => !a.startsWith('-')); +const textOptions: TextRecognitionOptions = {}; +if (langRaw) textOptions.languages = langRaw.split(','); +if (customWordsRaw) textOptions.customWords = customWordsRaw.split(','); +if (flags.has('--auto-lang')) textOptions.autoDetectLanguage = true; +if (flags.has('--no-correction')) textOptions.languageCorrection = false; +if (flags.has('--fast')) textOptions.fast = true; +if (roiRaw) { + const [x, y, width, height] = roiRaw.split(',').map(Number); + if ([x, y, width, height].some((n) => !Number.isFinite(n))) { + console.error(`Error: --roi expects x,y,w,h (got "${roiRaw}")`); + process.exit(1); + } + textOptions.regionOfInterest = { x, y, width, height }; +} +const ocrBase = { ...pageRange, ...textOptions, cache: flags.has('--cache') }; + +// Commands that need no input file. +if (flags.has('--capabilities') || flags.has('--languages')) { + const caps = await visionCapabilities(); + if (flags.has('--capabilities')) console.log(JSON.stringify(caps, null, 2)); + if (flags.has('--languages')) console.log(JSON.stringify(caps.ocrLanguages, null, 2)); + process.exit(0); +} + if (!fileArgs[0]) { console.error('Error: no image or PDF path provided.\n'); console.log(USAGE); @@ -151,31 +209,47 @@ if (flags.has('--markdown')) { const runRects = runAll || flags.has('--rectangles'); const runDoc = runAll || flags.has('--document'); const runClassify = runAll || flags.has('--classify'); + // One OCR pass shared by --ocr and --entities. + let textPromise: Promise | undefined; + const getText = () => (textPromise ??= ocr(inputPath, ocrBase) as Promise); + + const extended: Array<[string, () => Promise]> = [ + ['--structure', () => recognizeDocument(inputPath, textOptions)], + ['--entities', async () => extractEntities(await getText())], + ['--text-regions', () => detectTextRegions(inputPath, textOptions)], + ['--humans', () => detectHumans(inputPath)], + ['--face-landmarks', () => detectFaceLandmarks(inputPath)], + ['--saliency', () => detectSaliency(inputPath)], + ['--aesthetics', () => imageAesthetics(inputPath)], + ['--info', () => imageInfo(inputPath)], + ]; + const runExtended = extended.filter(([flag]) => flags.has(flag)); // Default: OCR text when no feature flag is given + const CLASSIC_FLAGS = [ + '--ocr', + '--blocks', + '--faces', + '--barcodes', + '--rectangles', + '--document', + '--classify', + ]; const anyFeatureFlag = - runAll || - flags.has('--ocr') || - flags.has('--blocks') || - flags.has('--faces') || - flags.has('--barcodes') || - flags.has('--rectangles') || - flags.has('--document') || - flags.has('--classify'); + runAll || CLASSIC_FLAGS.some((f) => flags.has(f)) || runExtended.length > 0; const useDefault = !anyFeatureFlag; (async () => { try { if (useDefault || runOcr) { - const text = await ocr(inputPath, pageRange); - console.log(text as string); + console.log(await getText()); } if (runBlocks) { const blocks = (await ocr(inputPath, { + ...ocrBase, format: 'blocks', - ...pageRange, })) as VisionBlock[]; console.log(JSON.stringify(blocks, null, 2)); } @@ -204,8 +278,12 @@ if (flags.has('--markdown')) { const labels = (await classify(inputPath)) as Classification[]; console.log(JSON.stringify(labels, null, 2)); } + + for (const [, fn] of runExtended) { + console.log(JSON.stringify(await fn(), null, 2)); + } } catch (error) { - console.error('Error:', error); + console.error(`Error: ${error instanceof Error ? error.message : String(error)}`); process.exit(1); } })(); diff --git a/src/helper.ts b/src/helper.ts new file mode 100644 index 0000000..7483edc --- /dev/null +++ b/src/helper.ts @@ -0,0 +1,112 @@ +// Shared plumbing for the native helpers (vision-helper, pdf-helper, ui-helper). +// +// Wire contract: a helper writes exactly one payload to stdout (the Swift side +// diverts framework noise to stderr), exits 1 on failure and 2 when a feature is +// not available on this macOS. + +import { execFile } from 'child_process'; +import { createHash } from 'crypto'; +import { mkdirSync } from 'fs'; +import { readFile } from 'fs/promises'; +import { tmpdir } from 'os'; +import { dirname, join, resolve } from 'path'; +import { fileURLToPath } from 'url'; + +const BIN_DIR = resolve(dirname(fileURLToPath(import.meta.url)), '../bin'); + +export const VISION_BIN = join(BIN_DIR, 'vision-helper'); +export const PDF_BIN = join(BIN_DIR, 'pdf-helper'); +export const UI_BIN = join(BIN_DIR, 'ui-helper'); + +/** Helper exit status meaning "not supported on this macOS". */ +export const EXIT_UNSUPPORTED = 2; + +export class UnsupportedOnThisMacOSError extends Error { + constructor(feature: string, minVersion: string) { + super(`${feature} requires macOS ${minVersion} or newer`); + this.name = 'UnsupportedOnThisMacOSError'; + } +} + +/** + * Helpers report failures as `ERROR: ` on stderr. Surface that + * line instead of Node's `Command failed: /long/path/to/helper --flags …`, + * keeping `code` so callers can still recognise the "unsupported" status. + */ +function helperError(err: unknown, stderr: string): Error { + const raw = err as { message?: string; code?: number }; + const reported = stderr + .split('\n') + .map((line) => line.trim()) + .find((line) => line.startsWith('ERROR: ')); + const error = new Error( + reported ? reported.slice('ERROR: '.length) : (raw.message ?? String(err)) + ) as Error & { code?: number; stderr?: string }; + if (raw.code !== undefined) error.code = raw.code; + // Kept for callers that inspect a non-helper tool's output (e.g. screencapture). + error.stderr = stderr; + return error; +} + +export interface ExecOptions { + timeout?: number; + /** Written to the helper's stdin, then closed. */ + input?: string; +} + +/** Spawn a helper and return its stdout. */ +export function execHelper(bin: string, args: string[], opts: ExecOptions = {}): Promise { + return new Promise((resolvePromise, reject) => { + const child = execFile( + bin, + args, + { timeout: opts.timeout ?? 30_000, maxBuffer: 64 * 1024 * 1024 }, + (err, stdout, stderr) => (err ? reject(helperError(err, stderr)) : resolvePromise(stdout)) + ); + if (opts.input !== undefined) child.stdin?.end(opts.input); + }); +} + +/** Spawn a helper and parse its JSON payload. */ +export async function runHelper( + bin: string, + args: string[], + opts: ExecOptions = {} +): Promise { + return JSON.parse(await execHelper(bin, args, opts)) as T; +} + +/** Like `runHelper`, but translates the "unsupported" exit status into a typed error. */ +export async function runGated( + bin: string, + args: string[], + feature: string, + minVersion: string, + opts: ExecOptions = {} +): Promise { + try { + return await runHelper(bin, args, opts); + } catch (err) { + if ((err as { code?: number }).code === EXIT_UNSUPPORTED) { + throw new UnsupportedOnThisMacOSError(feature, minVersion); + } + throw err; + } +} + +/** `override` resolved, or a fresh path in `$TMPDIR/macos-vision/`. */ +export function tmpOutPath(prefix: string, override?: string, ext = 'png'): string { + if (override) return resolve(override); + const dir = join(tmpdir(), 'macos-vision'); + mkdirSync(dir, { recursive: true }); + return join(dir, `${prefix}-${Date.now()}-${Math.floor(Math.random() * 1e6)}.${ext}`); +} + +export function sha256(bytes: Buffer | string): string { + return createHash('sha256').update(bytes).digest('hex'); +} + +/** SHA-256 of a file's bytes, hex. */ +export async function fileSha256(filePath: string): Promise { + return sha256(await readFile(filePath)); +} diff --git a/src/index.ts b/src/index.ts index ea18b8f..349ce09 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,22 +1,17 @@ -import { execFile } from 'child_process'; -import { promisify } from 'util'; -import { resolve, dirname, extname, dirname as pathDirname } from 'path'; -import { fileURLToPath } from 'url'; -import { open } from 'fs/promises'; - -const execFileAsync = promisify(execFile); -const __dirname = dirname(fileURLToPath(import.meta.url)); -const BIN_PATH = resolve(__dirname, '../bin/vision-helper'); -const PDF_BIN_PATH = resolve(__dirname, '../bin/pdf-helper'); +import { resolve, dirname, extname } from 'path'; +import { open, readFile, writeFile, mkdir } from 'fs/promises'; +import { homedir } from 'os'; +import { VISION_BIN, PDF_BIN, execHelper, runHelper, fileSha256, sha256 } from './helper.js'; +import { textOptionArgs, visionCapabilities } from './vision.js'; +import type { TextRecognitionOptions } from './vision.js'; + const BINARY_TIMEOUT_MS = 30_000; const PDF_RASTERIZE_TIMEOUT_MS = 120_000; +const OCR_CACHE_DIR = resolve(homedir(), '.cache', 'macos-vision', 'ocr'); -async function run(flag: string, imagePath: string): Promise { - const { stdout } = await execFileAsync(BIN_PATH, [flag, resolve(imagePath)], { - timeout: BINARY_TIMEOUT_MS, - }); - return stdout; -} +const run = (args: string[]) => runHelper(VISION_BIN, args, { timeout: BINARY_TIMEOUT_MS }); + +export { fileSha256 }; // ─── PDF helpers ───────────────────────────────────────────────────── @@ -92,11 +87,8 @@ export async function rasterizePdf( ): Promise { const absPath = resolve(pdfPath); const args = buildPdfArgs(absPath, options); - const { stdout } = await execFileAsync(PDF_BIN_PATH, args, { - timeout: PDF_RASTERIZE_TIMEOUT_MS, - }); - const pages: PdfPage[] = JSON.parse(stdout); - const cacheDir = pages.length > 0 ? pathDirname(pages[0].path) : ''; + const pages = await runHelper(PDF_BIN, args, { timeout: PDF_RASTERIZE_TIMEOUT_MS }); + const cacheDir = pages.length > 0 ? dirname(pages[0].path) : ''; return { pages, cacheDir }; } @@ -107,24 +99,62 @@ export async function rasterizePdf( async function ocrPdf( pdfPath: string, format: 'text' | 'blocks', - range: PdfPageRangeOptions = {} + options: Omit = {} ): Promise { - const { pages } = await rasterizePdf(pdfPath, range); + const { startPage, maxPages, onProgress, ...textOptions } = options; + const { pages } = await rasterizePdf(pdfPath, { startPage, maxPages }); if (format === 'blocks') { const all: VisionBlock[] = []; - for (const { page, path: pagePath } of pages) { - const blocks = (await ocr(pagePath, { format: 'blocks' })) as VisionBlock[]; + for (const [i, { page, path: pagePath }] of pages.entries()) { + const blocks = (await ocr(pagePath, { ...textOptions, format: 'blocks' })) as VisionBlock[]; all.push(...blocks.map((b) => ({ ...b, page }))); + onProgress?.(i + 1, pages.length); } return all; } const texts: string[] = []; - for (const { path: pagePath } of pages) { - texts.push((await ocr(pagePath)) as string); + for (const [i, { path: pagePath }] of pages.entries()) { + texts.push((await ocr(pagePath, { ...textOptions, format: 'text' })) as string); + onProgress?.(i + 1, pages.length); } return texts.join('\n\n--- Page Break ---\n\n'); } +// ─── OCR result cache ──────────────────────────────────────────────────────── + +/** + * Content hash + the canonical helper argv, so option order does not matter. + * The helper and macOS versions are part of the key: both change what Vision + * returns for identical input, so entries must not survive an upgrade. + */ +async function cacheKey( + absPath: string, + format: string, + opts: TextRecognitionOptions +): Promise { + const [hash, caps] = await Promise.all([fileSha256(absPath), visionCapabilities()]); + return sha256( + hash + JSON.stringify([caps.helperVersion, caps.macosVersion, format, ...textOptionArgs(opts)]) + ); +} + +async function readCache(key: string): Promise { + try { + return JSON.parse(await readFile(resolve(OCR_CACHE_DIR, `${key}.json`), 'utf8')) as T; + } catch { + return undefined; + } +} + +async function writeCache(key: string, value: unknown): Promise { + try { + await mkdir(OCR_CACHE_DIR, { recursive: true }); + await writeFile(resolve(OCR_CACHE_DIR, `${key}.json`), JSON.stringify(value)); + } catch { + // cache is best-effort + } +} + // ─── OCR ───────────────────────────────────────────────────────────────────── export interface VisionBlock { @@ -144,9 +174,17 @@ export interface VisionBlock { page?: number; } -export interface OcrOptions extends PdfPageRangeOptions { +export interface OcrOptions extends PdfPageRangeOptions, TextRecognitionOptions { /** Return plain text (default) or structured blocks with coordinates */ format?: 'text' | 'blocks'; + /** Called after each page is OCR'd (PDF inputs only). `done` counts from 1. */ + onProgress?: (done: number, total: number) => void; + /** + * Cache results in `~/.cache/macos-vision/ocr/` keyed by file content hash + options. + * Repeated OCR of the same bytes (e.g. an agent re-reading a screenshot) returns instantly. + * Default false. + */ + cache?: boolean; } export async function ocr( @@ -162,27 +200,32 @@ export async function ocr( options: OcrOptions = {} ): Promise { const absPath = resolve(imagePath); - const { format = 'text', startPage, maxPages } = options; + const { + format = 'text', + cache = false, + startPage, + maxPages, + onProgress, + ...textOptions + } = options; // ── PDF fast-path: rasterize via pdf-helper, then OCR each page ────── if (await isPdf(absPath)) { - return ocrPdf(absPath, format, { startPage, maxPages }); + return ocrPdf(absPath, format, { ...textOptions, cache, startPage, maxPages, onProgress }); } - // ── Existing image path (unchanged) ───────────────────────────────── + const key = cache ? await cacheKey(absPath, format, textOptions) : undefined; + if (key) { + const hit = await readCache(key); + if (hit !== undefined) return hit; + } + const textArgs = textOptionArgs(textOptions); + if (format === 'blocks') { - const { stdout } = await execFileAsync(BIN_PATH, ['--json', absPath], { - timeout: BINARY_TIMEOUT_MS, - }); - const raw: Array<{ - t: string; - x: number; - y: number; - w: number; - h: number; - confidence: number; - }> = JSON.parse(stdout); - return raw.map((b) => ({ + const raw = await run< + Array<{ t: string; x: number; y: number; w: number; h: number; confidence: number }> + >(['--json', ...textArgs, absPath]); + const blocks = raw.map((b) => ({ text: b.t, x: b.x, y: b.y, @@ -190,10 +233,15 @@ export async function ocr( height: b.h, confidence: b.confidence, })); + if (key) await writeCache(key, blocks); + return blocks; } - const { stdout } = await execFileAsync(BIN_PATH, [absPath], { timeout: BINARY_TIMEOUT_MS }); - return stdout.trim(); + const text = ( + await execHelper(VISION_BIN, [...textArgs, absPath], { timeout: BINARY_TIMEOUT_MS }) + ).trim(); + if (key) await writeCache(key, text); + return text; } // ─── Face detection ────────────────────────────────────────────────────── @@ -211,11 +259,8 @@ export interface Face { confidence: number; } -export async function detectFaces(imagePath: string): Promise { - const raw: Array<{ x: number; y: number; w: number; h: number; confidence: number }> = JSON.parse( - await run('--faces', imagePath) - ); - return raw.map((f) => ({ x: f.x, y: f.y, width: f.w, height: f.h, confidence: f.confidence })); +export function detectFaces(imagePath: string): Promise { + return run(['--faces', resolve(imagePath)]); } // ─── Barcode / QR detection ────────────────────────────────────────────────── @@ -237,25 +282,8 @@ export interface Barcode { confidence: number; } -export async function detectBarcodes(imagePath: string): Promise { - const raw: Array<{ - type: string; - value: string; - x: number; - y: number; - w: number; - h: number; - confidence: number; - }> = JSON.parse(await run('--barcodes', imagePath)); - return raw.map((b) => ({ - type: b.type, - value: b.value, - x: b.x, - y: b.y, - width: b.w, - height: b.h, - confidence: b.confidence, - })); +export function detectBarcodes(imagePath: string): Promise { + return run(['--barcodes', resolve(imagePath)]); } // ─── Rectangle detection ─────────────────────────────────────────────────── @@ -273,15 +301,8 @@ export interface Rectangle { confidence: number; } -export async function detectRectangles(imagePath: string): Promise { - const raw: Array<{ - topLeft: [number, number]; - topRight: [number, number]; - bottomLeft: [number, number]; - bottomRight: [number, number]; - confidence: number; - }> = JSON.parse(await run('--rectangles', imagePath)); - return raw; +export function detectRectangles(imagePath: string): Promise { + return run(['--rectangles', resolve(imagePath)]); } // ─── Document detection ────────────────────────────────────────────────────── @@ -301,8 +322,8 @@ export interface DocumentBounds { /** Returns the detected document boundary, or null if no document found. */ export async function detectDocument(imagePath: string): Promise { - const raw: DocumentBounds[] = JSON.parse(await run('--document', imagePath)); - return raw.length > 0 ? raw[0] : null; + const raw = await run(['--document', resolve(imagePath)]); + return raw[0] ?? null; } // ─── Image classification ───────────────────────────────────────────────────── @@ -315,9 +336,8 @@ export interface Classification { } /** Returns top image classifications sorted by confidence (highest first). */ -export async function classify(imagePath: string): Promise { - const raw: Classification[] = JSON.parse(await run('--classify', imagePath)); - return raw; +export function classify(imagePath: string): Promise { + return run(['--classify', resolve(imagePath)]); } // ─── Layout inference ──────────────────────────────────────────────────────────── @@ -338,3 +358,76 @@ export { inferLayout, sortBlocksByReadingOrder } from './layout.js'; // ─── Markdown pipeline (VisionScribe) ────────────────────────────────────────── export { VisionScribe, OllamaUnavailableError } from './markdown/index.js'; export type { VisionScribeOptions, ParagraphGroup } from './markdown/index.js'; + +// ─── UI: screen capture, windows, displays, permissions ──────────────────────── +export { listWindows, listDisplays, checkPermissions, captureScreen } from './ui.js'; +export type { + WindowInfo, + DisplayInfo, + PermissionsInfo, + ScreenFrame, + CaptureResult, + CaptureOptions, +} from './ui.js'; + +// ─── Extended Vision API ─────────────────────────────────────────────────────── +export { + visionCapabilities, + supportedOcrLanguages, + imageInfo, + detectTextRegions, + compareImages, + extractEntities, + recognizeDocument, + detectLensSmudge, + imageAesthetics, + detectHorizon, + detectFaceLandmarks, + detectHumans, + detectBodyPose, + detectHandPose, + detectAnimalPose, + detectAnimals, + detectSaliency, + detectContours, + cropImage, + cropDocument, + extractForeground, + personMask, + UnsupportedOnThisMacOSError, +} from './vision.js'; +export type { + NormalizedRect, + Detection, + TextRecognitionOptions, + VisionCapabilities, + ImageInfo, + TextRegion, + ImageComparison, + TextEntity, + DocumentStructure, + DocText, + DocLine, + DocTable, + DocCell, + DocList, + DocListItem, + DocDetectedData, + DocBarcode, + LensSmudge, + AestheticsScore, + Horizon, + FaceLandmarks, + HumanBox, + Keypoint, + Pose, + Animal, + SaliencyOptions, + Saliency, + ContourOptions, + Contour, + Contours, + CropResult, + MaskResult, + ForegroundOptions, +} from './vision.js'; diff --git a/src/native/ui-helper.swift b/src/native/ui-helper.swift new file mode 100644 index 0000000..e23f984 --- /dev/null +++ b/src/native/ui-helper.swift @@ -0,0 +1,122 @@ +import ApplicationServices +import CoreGraphics +import Foundation + +// ui-helper — read-only screen/window/permission introspection for macos-vision-mcp. +// Emits JSON on stdout. Never captures pixels itself (capture goes through +// /usr/sbin/screencapture); never synthesizes input. Eyes, not hands. + +// ─── Result structs ────────────────────────────────────────────────────────── + +struct WindowInfo: Codable { + let windowId: Int + let app: String + let pid: Int + let title: String + // Global screen points, top-left origin (matches CGEvent click coordinates). + let x: Double; let y: Double; let w: Double; let h: Double + let layer: Int + let isOnScreen: Bool +} + +struct DisplayInfo: Codable { + let displayId: Int + let isMain: Bool + // Global screen points, top-left origin. + let x: Double; let y: Double; let w: Double; let h: Double + let scale: Double +} + +struct PermissionsInfo: Codable { + let screenRecording: Bool + let accessibility: Bool + /// True while the login session is locked. Verified behaviour in that state: + /// window and region capture fail outright, and a full-screen capture returns + /// only the lock screen — so no capture is useful until the user unlocks. + let screenLocked: Bool +} + +func isScreenLocked() -> Bool { + guard let session = CGSessionCopyCurrentDictionary() as? [String: Any] else { return false } + return session["CGSSessionScreenIsLocked"] as? Bool ?? false +} + +func encodeJSON(_ value: T) -> String { + guard let data = try? JSONEncoder().encode(value), + let str = String(data: data, encoding: .utf8) else { return "[]" } + return str +} + +// ─── Modes ─────────────────────────────────────────────────────────────────── + +let args = CommandLine.arguments + +if args.contains("--permissions") { + let info = PermissionsInfo( + screenRecording: CGPreflightScreenCaptureAccess(), + accessibility: AXIsProcessTrusted(), + screenLocked: isScreenLocked() + ) + print(encodeJSON(info)) + exit(0) +} + +if args.contains("--displays") { + var count: UInt32 = 0 + var ids = [CGDirectDisplayID](repeating: 0, count: 16) + // Online (not Active) list: an asleep display is still online, and captures/ + // window queries keep working while it sleeps — report it. + CGGetOnlineDisplayList(16, &ids, &count) + var results: [DisplayInfo] = [] + for i in 0.. 0 { + scale = Double(mode.pixelWidth) / Double(mode.width) + } + results.append(DisplayInfo( + displayId: Int(id), + isMain: CGDisplayIsMain(id) != 0, + x: bounds.origin.x, y: bounds.origin.y, + w: bounds.size.width, h: bounds.size.height, + scale: scale + )) + } + print(encodeJSON(results)) + exit(0) +} + +if args.contains("--windows") { + let includeAll = args.contains("--all") + let options: CGWindowListOption = [.optionOnScreenOnly, .excludeDesktopElements] + guard let list = CGWindowListCopyWindowInfo(options, kCGNullWindowID) as? [[String: Any]] else { + print("[]") + exit(0) + } + var results: [WindowInfo] = [] + for w in list { + guard let boundsDict = w[kCGWindowBounds as String] as? [String: Double] else { continue } + let layer = w[kCGWindowLayer as String] as? Int ?? 0 + // Layer 0 = normal app windows; skip menu bar, dock, overlays unless --all. + if layer != 0 && !includeAll { continue } + let width = boundsDict["Width"] ?? 0 + let height = boundsDict["Height"] ?? 0 + if width < 40 || height < 40 { continue } // status items, tooltips + results.append(WindowInfo( + windowId: w[kCGWindowNumber as String] as? Int ?? 0, + app: w[kCGWindowOwnerName as String] as? String ?? "", + pid: w[kCGWindowOwnerPID as String] as? Int ?? 0, + title: w[kCGWindowName as String] as? String ?? "", + x: boundsDict["X"] ?? 0, y: boundsDict["Y"] ?? 0, + w: width, h: height, + layer: layer, + isOnScreen: (w[kCGWindowIsOnscreen as String] as? Bool) ?? true + )) + } + print(encodeJSON(results)) + exit(0) +} + +print("Usage: ui-helper [--windows [--all] | --displays | --permissions]") +exit(1) diff --git a/src/native/vision-helper.swift b/src/native/vision-helper.swift index 3b43128..9e097ad 100644 --- a/src/native/vision-helper.swift +++ b/src/native/vision-helper.swift @@ -1,6 +1,10 @@ import Vision import AppKit import Foundation +import CoreGraphics +import CoreImage +import ImageIO +import UniformTypeIdentifiers // ─── Result structs ────────────────────────────────────────────────────────── @@ -10,15 +14,21 @@ struct OCRResult: Codable { let confidence: Float } -struct FaceResult: Codable { - let x: Double; let y: Double; let w: Double; let h: Double +/// Normalized box, top-left origin. Wire keys match the public TS `NormalizedRect`. +struct Box: Codable { + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float + init(_ r: CGRect, _ confidence: Float) { + x = Double(r.origin.x); y = flipY(Double(r.origin.y), Double(r.size.height)) + width = Double(r.size.width); height = Double(r.size.height) + self.confidence = confidence + } } struct BarcodeResult: Codable { let type: String let value: String - let x: Double; let y: Double; let w: Double; let h: Double + let x: Double; let y: Double; let width: Double; let height: Double let confidence: Float } @@ -52,6 +62,165 @@ func encodeJSON(_ value: T) -> String { return str } +struct CompareResult: Codable { + let distance: Double +} + +struct SmudgeResult: Codable { + let confidence: Float + let supported: Bool +} + +struct EntityResult: Codable { + let type: String + let text: String + let start: Int + let end: Int + let value: String? + let components: [String: String]? +} + +struct Capabilities: Codable { + let helperVersion: String + let macosVersion: String + let ocrLanguages: [String] + let features: [String: Bool] +} + +// Document structure (macOS 26+, RecognizeDocumentsRequest) +struct DocRegion: Codable { let x: Double; let y: Double; let width: Double; let height: Double } +struct DocLine: Codable { let text: String; let confidence: Float; let bbox: DocRegion } +struct DocText: Codable { + let text: String + let alignment: String? + let bbox: DocRegion + let lines: [DocLine] +} +struct DocCell: Codable { + let text: String + let row: Int; let col: Int + let rowSpan: Int; let colSpan: Int + let bbox: DocRegion +} +struct DocTable: Codable { + let rowCount: Int; let columnCount: Int + let rows: [[String]] + let cells: [DocCell] + let bbox: DocRegion +} +struct DocListItem: Codable { let marker: String; let text: String; let bbox: DocRegion } +struct DocList: Codable { let items: [DocListItem]; let bbox: DocRegion } +struct DocData: Codable { let type: String; let text: String; let value: String?; let bbox: DocRegion } +struct DocBarcode: Codable { let type: String; let value: String; let bbox: DocRegion } +struct DocStructure: Codable { + let title: DocText? + let text: String + let paragraphs: [DocText] + let tables: [DocTable] + let lists: [DocList] + let barcodes: [DocBarcode] + let detectedData: [DocData] +} + +struct PointResult: Codable { let x: Double; let y: Double; let confidence: Float } +struct FaceLandmarksResult: Codable { + let x: Double; let y: Double; let width: Double; let height: Double + let confidence: Float + let roll: Double?; let yaw: Double?; let pitch: Double? + let captureQuality: Float? + let landmarks: [String: [[Double]]] +} +struct PoseResult: Codable { let joints: [String: PointResult]; let confidence: Float; let chirality: String? } +struct AnimalResult: Codable { + let labels: [ClassificationResult] + let x: Double; let y: Double; let width: Double; let height: Double + let confidence: Float +} +struct HorizonResult: Codable { let angleDegrees: Double } +struct ContourResult: Codable { + let index: Int; let pointCount: Int; let childCount: Int + let x: Double; let y: Double; let width: Double; let height: Double + let points: [[Double]]? +} +struct ContoursResult: Codable { let totalContours: Int; let topLevel: [ContourResult] } +struct SaliencyResult: Codable { + let regions: [Box] + let heatmapPath: String? +} +struct MaskResult: Codable { let instances: Int; let outPath: String } +struct AestheticsResult: Codable { let overallScore: Float; let isUtility: Bool } +struct CropResult: Codable { let outPath: String; let width: Int; let height: Int } +struct ImageInfoResult: Codable { + let width: Int; let height: Int + let hasAlpha: Bool; let bitsPerComponent: Int + let colorSpace: String?; let dpi: Double?; let orientation: Int?; let format: String? +} + +let HELPER_VERSION = "2" + +// VisionCore logs model-loading noise straight to fd 1 in some modes. Point fd 1 at +// stderr for the whole process and write our JSON to the original stdout instead, +// so callers always get exactly one clean payload. +let resultOut: FileHandle = { + let saved = dup(STDOUT_FILENO) + dup2(STDERR_FILENO, STDOUT_FILENO) + return FileHandle(fileDescriptor: saved, closeOnDealloc: false) +}() + +func emit(_ text: String) { + resultOut.write((text + "\n").data(using: .utf8)!) +} + +func macOSVersionString() -> String { + let v = ProcessInfo.processInfo.operatingSystemVersion + return "\(v.majorVersion).\(v.minorVersion).\(v.patchVersion)" +} + +let macOS14 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 14, minorVersion: 0, patchVersion: 0)) +let macOS15 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 15, minorVersion: 0, patchVersion: 0)) +let macOS26 = ProcessInfo.processInfo.isOperatingSystemAtLeast(OperatingSystemVersion(majorVersion: 26, minorVersion: 0, patchVersion: 0)) + +// A symbol introduced in a newer SDK is simply absent when the helper is built +// against an older one — which is exactly what the swiftc fallback does on an +// older Mac. So each modern feature is gated twice: at compile time on the SDK +// (-DSDK_nn, set by the build scripts) and at runtime on #available. +#if SDK_14 +let sdk14 = true +#else +let sdk14 = false +#endif +#if SDK_15 +let sdk15 = true +#else +let sdk15 = false +#endif +#if SDK_26 +let sdk26 = true +#else +let sdk26 = false +#endif +let iso8601 = ISO8601DateFormatter() + +func supportedLanguages() -> [String] { + let req = VNRecognizeTextRequest() + req.recognitionLevel = .accurate + return (try? req.supportedRecognitionLanguages()) ?? [] +} + +/// Run Vision requests against the shared handler; any failure is fatal for a CLI. +func perform(_ requests: [VNRequest], _ what: String) { + do { + try handler.perform(requests) + } catch { + fail("Vision \(what) failed: \(error.localizedDescription)") + } +} + +func fail(_ message: String, code: Int32 = 1) -> Never { + fputs("ERROR: \(message)\n", stderr) + exit(code) +} + // ─── Argument parsing ───────────────────────────────────────────────────────── let args = CommandLine.arguments @@ -61,25 +230,190 @@ let isBarcodes = args.contains("--barcodes") let isRectangles = args.contains("--rectangles") let isDocument = args.contains("--document") let isClassify = args.contains("--classify") +let isTextRects = args.contains("--text-rects") +let isCompare = args.contains("--compare") +let isSmudge = args.contains("--smudge") +let isStructure = args.contains("--document-structure") +let isEntities = args.contains("--entities") +let isLandmarks = args.contains("--face-landmarks") +let isHumans = args.contains("--humans") +let isBodyPose = args.contains("--body-pose") +let isHandPose = args.contains("--hand-pose") +let isAnimals = args.contains("--animals") +let isAnimalPose = args.contains("--animal-pose") +let isHorizon = args.contains("--horizon") +let isContours = args.contains("--contours") +let isSaliency = args.contains("--saliency") +let isForeground = args.contains("--foreground-mask") +let isPersonMask = args.contains("--person-mask") +let isAesthetics = args.contains("--aesthetics") +let isDocCrop = args.contains("--document-crop") +let isCrop = args.contains("--crop") +let isImageInfo = args.contains("--image-info") +let anyNewMode = isLandmarks || isHumans || isBodyPose || isHandPose || isAnimals || isAnimalPose || isHorizon + || isContours || isSaliency || isForeground || isPersonMask || isAesthetics || isDocCrop || isCrop || isImageInfo + +/// Value following a `--flag`, or nil. +func optValue(_ flag: String) -> String? { + guard let i = args.firstIndex(of: flag), i + 1 < args.count else { return nil } + return args[i + 1] +} + +// Flags that consume the next argument (so it is not mistaken for a file path). +let valueFlags: Set = ["--lang", "--custom-words", "--roi", "--min-text-height", "--out", "--saliency", "--crop", "--max-points"] +var consumed = Set() +for (i, a) in args.enumerated() where valueFlags.contains(a) && i + 1 < args.count { consumed.insert(i + 1) } +let fileArgs = args.enumerated() + .filter { i, a in i > 0 && !a.hasPrefix("--") && !consumed.contains(i) } + .map { $0.element } + +// ─── Modes that need no image ─────────────────────────────────────────────── + +if args.contains("--languages") { + emit(encodeJSON(supportedLanguages())) + exit(0) +} -let fileArgs = args.filter { !$0.hasPrefix("--") && !$0.contains("vision-helper") } +if args.contains("--capabilities") { + let caps = Capabilities( + helperVersion: HELPER_VERSION, + macosVersion: macOSVersionString(), + ocrLanguages: supportedLanguages(), + features: [ + "ocr": true, "ocrOptions": true, "faces": true, "barcodes": true, + "rectangles": true, "document": true, "classify": true, + "textRects": true, "compare": true, "entities": true, + "documentStructure": macOS26 && sdk26, "lensSmudge": macOS26 && sdk26, + "faceLandmarks": true, "humans": true, "bodyPose": true, "handPose": true, + "animals": true, "animalPose": macOS14 && sdk14, "horizon": true, "contours": true, + "saliency": true, "foregroundMask": macOS14 && sdk14, "personMask": true, + "aesthetics": macOS15 && sdk15, "documentCrop": true, "crop": true, "imageInfo": true, + ] + ) + emit(encodeJSON(caps)) + exit(0) +} + +// --entities: NSDataDetector over UTF-8 text read from stdin. No image involved. +if isEntities { + let data = FileHandle.standardInput.readDataToEndOfFile() + let text = String(decoding: data, as: UTF8.self) + let types: NSTextCheckingResult.CheckingType = [.link, .phoneNumber, .address, .date, .transitInformation] + guard let detector = try? NSDataDetector(types: types.rawValue) else { + fail("NSDataDetector unavailable") + } + let ns = text as NSString + var results: [EntityResult] = [] + for m in detector.matches(in: text, options: [], range: NSRange(location: 0, length: ns.length)) { + let matched = ns.substring(with: m.range) + var type = "unknown" + var value: String? = nil + var comps: [String: String]? = nil + switch m.resultType { + case .link: + type = "link"; value = m.url?.absoluteString + if let u = m.url, u.scheme == "mailto" { type = "email"; value = String(u.absoluteString.dropFirst(7)) } + case .phoneNumber: + type = "phone"; value = m.phoneNumber + case .address: + type = "address" + if let c = m.addressComponents { + var d: [String: String] = [:] + for (k, v) in c { d[k.rawValue] = v } + comps = d + value = c[.street].map { s in [s, c[.city], c[.zip]].compactMap { $0 }.joined(separator: ", ") } + } + case .date: + type = "date"; value = m.date.map { iso8601.string(from: $0) } + if m.duration > 0 { comps = ["durationSeconds": String(Int(m.duration))] } + case .transitInformation: + type = "transit" + if let c = m.components { var d: [String: String] = [:]; for (k, v) in c { d[k.rawValue] = v }; comps = d } + default: break + } + results.append(EntityResult(type: type, text: matched, start: m.range.location, + end: m.range.location + m.range.length, value: value, components: comps)) + } + emit(encodeJSON(results)) + exit(0) +} guard let imagePath = fileArgs.first else { - print("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify] ") + emit("Usage: vision-helper [--json|--faces|--barcodes|--rectangles|--document|--classify|--text-rects|--document-structure|--smudge|--compare ] ") + emit(" vision-helper --languages | --capabilities | --entities < text.txt") + emit("OCR options: --lang pl,en --auto-lang --no-correction --custom-words a,b --fast --roi x,y,w,h --min-text-height f") + exit(0) +} + +// ─── Image compare (feature print distance) ────────────────────────────────── + +if isCompare { + guard fileArgs.count >= 2 else { + fail("--compare needs two image paths") + } + func featurePrint(_ path: String) -> VNFeaturePrintObservation? { + guard let img = NSImage(contentsOf: URL(fileURLWithPath: path)), + let cg = img.cgImage(forProposedRect: nil, context: nil, hints: nil) else { + fail("Cannot open file: \(path)") + } + let req = VNGenerateImageFeaturePrintRequest() + let h = VNImageRequestHandler(cgImage: cg, options: [:]) + try? h.perform([req]) + return req.results?.first as? VNFeaturePrintObservation + } + guard let a = featurePrint(fileArgs[0]), let b = featurePrint(fileArgs[1]) else { + fail("Vision feature print failed") + } + var distance: Float = 0 + do { try a.computeDistance(&distance, to: b) } catch { + fail("Vision feature print distance failed: \(error.localizedDescription)") + } + emit(encodeJSON(CompareResult(distance: Double(distance)))) exit(0) } guard let image = NSImage(contentsOf: URL(fileURLWithPath: imagePath)), let cgImage = image.cgImage(forProposedRect: nil, context: nil, hints: nil) else { - fputs("ERROR: Cannot open file: \(imagePath)\n", stderr) - exit(1) + fail("Cannot open file: \(imagePath)") } let handler = VNImageRequestHandler(cgImage: cgImage, options: [:]) +// ─── Shared OCR options ────────────────────────────────────────────────────── + +let ocrLanguages: [String] = optValue("--lang")?.split(separator: ",").map { String($0).trimmingCharacters(in: .whitespaces) }.filter { !$0.isEmpty } ?? [] +let ocrCustomWords: [String] = optValue("--custom-words")?.split(separator: ",").map { String($0) }.filter { !$0.isEmpty } ?? [] +let ocrAutoLang = args.contains("--auto-lang") +let ocrNoCorrection = args.contains("--no-correction") +let ocrFast = args.contains("--fast") +let ocrMinHeight: Float? = optValue("--min-text-height").flatMap { Float($0) } + +// Region of interest: normalized x,y,w,h with TOP-LEFT origin (same space as our +// output). Vision wants bottom-left origin, so flip. Vision then reports +// observations RELATIVE TO THE ROI, so every bounding box must go back through +// unROI() to land in full-image space — verified empirically: with and without an +// ROI the same text yields identical coordinates only when unROI is applied. +var roiRect: CGRect? = nil +if let roi = optValue("--roi") { + let p = roi.split(separator: ",").compactMap { Double($0) } + if p.count == 4 { + roiRect = CGRect(x: p[0], y: 1.0 - p[1] - p[3], width: p[2], height: p[3]) + } else { + fail("--roi expects x,y,w,h (normalized 0-1)") + } +} + +/// Vision reports observations relative to `regionOfInterest`; map back to full-image space. +func unROI(_ r: CGRect) -> CGRect { + guard let roi = roiRect else { return r } + return CGRect(x: roi.origin.x + r.origin.x * roi.width, + y: roi.origin.y + r.origin.y * roi.height, + width: r.width * roi.width, height: r.height * roi.height) +} + // ─── OCR (default + --json) ─────────────────────────────────────────────────── -if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !isClassify) { +if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !isClassify && !isTextRects && !isSmudge && !isStructure && !anyNewMode) { var ocrResults: [OCRResult] = [] var rawText = "" @@ -87,7 +421,7 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is guard let obs = req.results as? [VNRecognizedTextObservation] else { return } for o in obs { guard let c = o.topCandidates(1).first else { continue } - let box = o.boundingBox + let box = unROI(o.boundingBox) if isJsonMode { ocrResults.append(OCRResult( t: c.string, @@ -102,42 +436,551 @@ if isJsonMode || (!isFaces && !isBarcodes && !isRectangles && !isDocument && !is } } } - request.recognitionLevel = .accurate + request.recognitionLevel = ocrFast ? .fast : .accurate + if !ocrLanguages.isEmpty { request.recognitionLanguages = ocrLanguages } + if ocrAutoLang { + if #available(macOS 13.0, *) { request.automaticallyDetectsLanguage = true } + } + request.usesLanguageCorrection = !ocrNoCorrection + if !ocrCustomWords.isEmpty { request.customWords = ocrCustomWords } + if let mh = ocrMinHeight { request.minimumTextHeight = mh } + if let r = roiRect { request.regionOfInterest = r } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision OCR failed: \(error.localizedDescription)\n", stderr) - exit(1) + perform([request], "OCR") + emit(isJsonMode ? encodeJSON(ocrResults) : rawText.trimmingCharacters(in: .whitespacesAndNewlines)) + exit(0) +} + +// ─── Text rectangles (fast "where is text" without recognition) ────────────── + +if isTextRects { + var results: [Box] = [] + let request = VNDetectTextRectanglesRequest { (req, _) in + guard let obs = req.results as? [VNTextObservation] else { return } + results = obs.map { Box(unROI($0.boundingBox), $0.confidence) } } - print(isJsonMode ? encodeJSON(ocrResults) : rawText.trimmingCharacters(in: .whitespacesAndNewlines)) + if let r = roiRect { request.regionOfInterest = r } + perform([request], "text rectangle detection") + emit(encodeJSON(results)) exit(0) } +// ─── macOS 26+: document structure & lens smudge (new Vision Swift API) ────── + +#if SDK_26 +@available(macOS 26.0, *) +func region(_ r: NormalizedRegion) -> DocRegion { + let b = r.boundingBox + return DocRegion(x: Double(b.origin.x), y: flipY(Double(b.origin.y), Double(b.height)), + width: Double(b.width), height: Double(b.height)) +} + +@available(macOS 26.0, *) +func docText(_ t: DocumentObservation.Container.Text) -> DocText { + let align: String? + switch t.textAlignment { + case .some(.leading): align = "leading" + case .some(.trailing): align = "trailing" + case .some(.center): align = "center" + default: align = nil + } + return DocText( + text: t.transcript, + alignment: align, + bbox: region(t.boundingRegion), + lines: t.lines.map { DocLine(text: $0.transcript, confidence: $0.confidence, bbox: region($0.boundingRegion)) } + ) +} + +@available(macOS 26.0, *) +func docData(_ d: DocumentObservation.Container.DataDetectorMatch, in text: String) -> DocData { + var type = "unknown" + var value: String? = nil + switch d.match.details { + case .link(let l): type = "link"; value = l.url.absoluteString + case .emailAddress(let e): type = "email"; value = e.emailAddress + case .phoneNumber(let p): type = "phone"; value = p.phoneNumber + case .postalAddress(let a): type = "address"; value = a.fullAddress + case .calendarEvent(let c): type = "date"; value = c.startDate.map { iso8601.string(from: $0) } + case .moneyAmount(let m): type = "money"; value = "\(m.amount) \(m.currency.identifier)" + case .flightNumber(let f): type = "flight"; value = "\(f.airlineCode)\(f.flightNumber)" + case .shipmentTrackingNumber(let s): type = "tracking"; value = s.trackingNumber + case .measurement(let m): type = "measurement"; value = String(m.value) + case .paymentIdentifier(let p): type = "payment"; value = p.identifier + @unknown default: break + } + let matched = d.match.range.map { String(text[$0]) } ?? "" + return DocData(type: type, text: matched, value: value, bbox: region(d.boundingRegion)) +} + +@available(macOS 26.0, *) +func runDocumentStructure() -> DocStructure? { + var req = RecognizeDocumentsRequest() + var opts = req.textRecognitionOptions + if !ocrLanguages.isEmpty { opts.recognitionLanguages = ocrLanguages.map { Locale.Language(identifier: $0) } } + if ocrAutoLang { opts.automaticallyDetectLanguage = true } + opts.useLanguageCorrection = !ocrNoCorrection + if !ocrCustomWords.isEmpty { opts.customWords = ocrCustomWords } + if let mh = ocrMinHeight { opts.minimumTextHeightFraction = mh } + req.textRecognitionOptions = opts + if let r = roiRect { req.regionOfInterest = NormalizedRect(normalizedRect: r) } + + let sema = DispatchSemaphore(value: 0) + var result: DocStructure? = nil + var failure: Error? = nil + Task { + do { + let observations = try await req.perform(on: cgImage) + if let doc = observations.first { + let c = doc.document + let full = c.text.transcript + var tables: [DocTable] = [] + for t in c.tables { + var cells: [DocCell] = [] + var rows: [[String]] = [] + for (ri, row) in t.rows.enumerated() { + var rowTexts: [String] = [] + for cell in row { + // A spanning cell appears in every row it covers; emit it once. + if cell.rowRange.lowerBound == ri { + cells.append(DocCell( + text: cell.content.text.transcript, + row: cell.rowRange.lowerBound, col: cell.columnRange.lowerBound, + rowSpan: cell.rowRange.count, colSpan: cell.columnRange.count, + bbox: region(cell.content.boundingRegion) + )) + } + rowTexts.append(cell.content.text.transcript) + } + rows.append(rowTexts) + } + tables.append(DocTable(rowCount: t.rows.count, columnCount: t.columns.count, + rows: rows, cells: cells, bbox: region(t.boundingRegion))) + } + let lists = c.lists.map { l in + DocList(items: l.items.map { DocListItem(marker: $0.markerString, text: $0.itemString, + bbox: region($0.content.boundingRegion)) }, + bbox: region(l.boundingRegion)) + } + let barcodes = c.barcodes.map { + DocBarcode(type: String(describing: $0.symbology), value: $0.payloadString ?? "", bbox: region($0.boundingRegion)) + } + result = DocStructure( + title: c.title.map(docText), + text: full, + paragraphs: c.paragraphs.map(docText), + tables: tables, + lists: lists, + barcodes: barcodes, + detectedData: c.text.detectedData.map { docData($0, in: full) } + ) + } else { + result = DocStructure(title: nil, text: "", paragraphs: [], tables: [], lists: [], barcodes: [], detectedData: []) + } + } catch { + failure = error + } + sema.signal() + } + sema.wait() + if let e = failure { + fail("Vision document recognition failed: \(e.localizedDescription)") + } + return result +} + +@available(macOS 26.0, *) +func runSmudge() -> SmudgeResult { + let req = DetectLensSmudgeRequest() + // VisionCore logs "Unable to find a valid E5 ..." straight to stdout on hardware + // without the smudge model. Divert stdout to a temp file while the request runs + // so the JSON we print afterwards stays clean, and use the noise to flag support. + let tmp = NSTemporaryDirectory() + "vision-helper-smudge-\(getpid()).log" + let savedStdout = dup(STDOUT_FILENO) + fflush(stdout) + let fd = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0o600) + if fd >= 0 { dup2(fd, STDOUT_FILENO); close(fd) } + let sema = DispatchSemaphore(value: 0) + var conf: Float = 0 + var failure: Error? = nil + Task { + do { conf = try await req.perform(on: cgImage).confidence } catch { failure = error } + sema.signal() + } + sema.wait() + fflush(stdout) + dup2(savedStdout, STDOUT_FILENO) + close(savedStdout) + let noise = (try? String(contentsOfFile: tmp, encoding: .utf8)) ?? "" + unlink(tmp) + // A throw here means the smudge model is not usable on this machine — the + // image already loaded, and CI runners without the model raise + // Foundation._GenericObjCError.nilError rather than logging the notice below. + // Report it as unsupported, which is this request's documented contract, + // instead of failing the whole call. + if failure != nil { return SmudgeResult(confidence: 0, supported: false) } + let supported = !noise.contains("Unable to find") + return SmudgeResult(confidence: conf, supported: supported) +} +#endif + +if isStructure { +#if SDK_26 + if #available(macOS 26.0, *) { + if let s = runDocumentStructure() { emit(encodeJSON(s)) } + exit(0) + } +#endif + fail("--document-structure requires macOS 26 or newer", code: 2) +} + +if isSmudge { +#if SDK_26 + if #available(macOS 26.0, *) { + emit(encodeJSON(runSmudge())) + exit(0) + } +#endif + fail("--smudge requires macOS 26 or newer", code: 2) +} + + +// ─── Pixel output helpers ──────────────────────────────────────────────────── + +func writePNG(_ cg: CGImage, to path: String) -> Bool { + let url = URL(fileURLWithPath: path) as CFURL + guard let dest = CGImageDestinationCreateWithURL(url, UTType.png.identifier as CFString, 1, nil) else { return false } + CGImageDestinationAddImage(dest, cg, nil) + return CGImageDestinationFinalize(dest) +} + +let ciContext = CIContext(options: nil) + +func cgFromPixelBuffer(_ pb: CVPixelBuffer) -> CGImage? { + let ci = CIImage(cvPixelBuffer: pb) + return ciContext.createCGImage(ci, from: ci.extent) +} + +func requireOut() -> String { + guard let out = optValue("--out") else { + fail("this mode requires --out ") + } + return out +} + +func jointsDict(_ points: [VNRecognizedPointKey: VNRecognizedPoint]) -> [String: PointResult] { + var d: [String: PointResult] = [:] + for (k, p) in points where p.confidence > 0 { + d[k.rawValue] = PointResult(x: Double(p.x), y: 1.0 - Double(p.y), confidence: p.confidence) + } + return d +} + +// ─── Image info (no Vision) ────────────────────────────────────────────────── + +if isImageInfo { + var dpi: Double? = nil, orientation: Int? = nil, format: String? = nil + if let src = CGImageSourceCreateWithURL(URL(fileURLWithPath: imagePath) as CFURL, nil) { + format = CGImageSourceGetType(src) as String? + if let props = CGImageSourceCopyPropertiesAtIndex(src, 0, nil) as? [CFString: Any] { + dpi = props[kCGImagePropertyDPIWidth] as? Double + orientation = props[kCGImagePropertyOrientation] as? Int + } + } + let info = ImageInfoResult( + width: cgImage.width, height: cgImage.height, + hasAlpha: cgImage.alphaInfo != .none && cgImage.alphaInfo != .noneSkipFirst && cgImage.alphaInfo != .noneSkipLast, + bitsPerComponent: cgImage.bitsPerComponent, + colorSpace: cgImage.colorSpace?.name as String?, + dpi: dpi, orientation: orientation, format: format + ) + emit(encodeJSON(info)) + exit(0) +} + +// ─── Crop (normalized region, top-left origin) ─────────────────────────────── + +if isCrop { + let out = requireOut() + let p = optValue("--crop")?.split(separator: ",").compactMap { Double($0) } ?? [] + guard p.count == 4 else { + fail("--crop expects x,y,w,h (normalized 0-1)") + } + let W = Double(cgImage.width), H = Double(cgImage.height) + let rect = CGRect(x: p[0] * W, y: p[1] * H, width: p[2] * W, height: p[3] * H).integral + guard let cropped = cgImage.cropping(to: rect), writePNG(cropped, to: out) else { + fail("crop failed") + } + emit(encodeJSON(CropResult(outPath: out, width: cropped.width, height: cropped.height))) + exit(0) +} + +// ─── Document crop (perspective-corrected) ─────────────────────────────────── + +if isDocCrop { + let out = requireOut() + let request = VNDetectDocumentSegmentationRequest() + try? handler.perform([request]) + guard let o = request.results?.first as? VNRectangleObservation else { + fail("no document detected") + } + let ci = CIImage(cgImage: cgImage) + let W = ci.extent.width, H = ci.extent.height + func p(_ pt: CGPoint) -> CGPoint { CGPoint(x: pt.x * W, y: pt.y * H) } + let filter = CIFilter(name: "CIPerspectiveCorrection")! + filter.setValue(ci, forKey: kCIInputImageKey) + filter.setValue(CIVector(cgPoint: p(o.topLeft)), forKey: "inputTopLeft") + filter.setValue(CIVector(cgPoint: p(o.topRight)), forKey: "inputTopRight") + filter.setValue(CIVector(cgPoint: p(o.bottomLeft)), forKey: "inputBottomLeft") + filter.setValue(CIVector(cgPoint: p(o.bottomRight)), forKey: "inputBottomRight") + guard let outCI = filter.outputImage, + let outCG = ciContext.createCGImage(outCI, from: outCI.extent), + writePNG(outCG, to: out) else { + fail("perspective correction failed") + } + emit(encodeJSON(CropResult(outPath: out, width: outCG.width, height: outCG.height))) + exit(0) +} + +// ─── Face landmarks + capture quality ──────────────────────────────────────── + +if isLandmarks { + let lm = VNDetectFaceLandmarksRequest() + let cq = VNDetectFaceCaptureQualityRequest() + perform([lm, cq], "face landmarks") + let faces = (lm.results ?? []) + let quals = (cq.results ?? []) + var results: [FaceLandmarksResult] = [] + for (i, f) in faces.enumerated() { + let b = Box(f.boundingBox, f.confidence) + var marks: [String: [[Double]]] = [:] + if let l = f.landmarks { + let regions: [(String, VNFaceLandmarkRegion2D?)] = [ + ("faceContour", l.faceContour), ("leftEye", l.leftEye), ("rightEye", l.rightEye), + ("leftEyebrow", l.leftEyebrow), ("rightEyebrow", l.rightEyebrow), ("nose", l.nose), + ("noseCrest", l.noseCrest), ("medianLine", l.medianLine), ("outerLips", l.outerLips), + ("innerLips", l.innerLips), ("leftPupil", l.leftPupil), ("rightPupil", l.rightPupil), + ] + for (name, r) in regions { + guard let r = r else { continue } + // pointsInImage gives pixel coords (bottom-left origin); normalize + flip. + let pts = r.pointsInImage(imageSize: CGSize(width: cgImage.width, height: cgImage.height)) + marks[name] = pts.map { [Double($0.x) / Double(cgImage.width), 1.0 - Double($0.y) / Double(cgImage.height)] } + } + } + let q: Float? = i < quals.count ? quals[i].faceCaptureQuality : nil + results.append(FaceLandmarksResult( + x: b.x, y: b.y, width: b.width, height: b.height, confidence: f.confidence, + roll: f.roll.map { Double(truncating: $0) * 180 / .pi }, + yaw: f.yaw.map { Double(truncating: $0) * 180 / .pi }, + pitch: f.pitch.map { Double(truncating: $0) * 180 / .pi }, + captureQuality: q, landmarks: marks + )) + } + emit(encodeJSON(results)) + exit(0) +} + +// ─── Human rectangles ──────────────────────────────────────────────────────── + +if isHumans { + let request = VNDetectHumanRectanglesRequest() + request.upperBodyOnly = false + perform([request], "human detection") + let results = (request.results ?? []).map { Box($0.boundingBox, $0.confidence) } + emit(encodeJSON(results)) + exit(0) +} + +// ─── Body / hand / animal pose ─────────────────────────────────────────────── + +if isBodyPose { + let request = VNDetectHumanBodyPoseRequest() + perform([request], "body pose") + let results = (request.results ?? []).map { o in + PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) + } + emit(encodeJSON(results)) + exit(0) +} + +if isHandPose { + let request = VNDetectHumanHandPoseRequest() + request.maximumHandCount = 4 + perform([request], "hand pose") + let results = (request.results ?? []).map { o -> PoseResult in + let ch: String + switch o.chirality { case .left: ch = "left"; case .right: ch = "right"; default: ch = "unknown" } + return PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: ch) + } + emit(encodeJSON(results)) + exit(0) +} + +if isAnimalPose { +#if SDK_14 + if #available(macOS 14.0, *) { + let request = VNDetectAnimalBodyPoseRequest() + perform([request], "animal pose") + let results = (request.results ?? []).map { o in + PoseResult(joints: jointsDict((try? o.recognizedPoints(forGroupKey: .all)) ?? [:]), confidence: o.confidence, chirality: nil) + } + emit(encodeJSON(results)) + exit(0) + } +#endif + fail("--animal-pose requires macOS 14 or newer", code: 2) +} + +// ─── Animals (cat / dog) ───────────────────────────────────────────────────── + +if isAnimals { + let request = VNRecognizeAnimalsRequest() + perform([request], "animal recognition") + let results = (request.results ?? []).map { o -> AnimalResult in + let b = Box(o.boundingBox, o.confidence) + return AnimalResult( + labels: o.labels.map { ClassificationResult(identifier: $0.identifier, confidence: $0.confidence) }, + x: b.x, y: b.y, width: b.width, height: b.height, confidence: o.confidence + ) + } + emit(encodeJSON(results)) + exit(0) +} + +// ─── Horizon ───────────────────────────────────────────────────────────────── + +if isHorizon { + let request = VNDetectHorizonRequest() + perform([request], "horizon detection") + guard let o = request.results?.first as? VNHorizonObservation else { + emit("null") + exit(0) + } + emit(encodeJSON(HorizonResult(angleDegrees: Double(o.angle) * 180 / .pi))) + exit(0) +} + +// ─── Contours ──────────────────────────────────────────────────────────────── + +if isContours { + let request = VNDetectContoursRequest() + request.detectsDarkOnLight = !args.contains("--light-on-dark") + if let r = roiRect { request.regionOfInterest = r } + perform([request], "contour detection") + guard let o = request.results?.first as? VNContoursObservation else { + emit(encodeJSON(ContoursResult(totalContours: 0, topLevel: []))) + exit(0) + } + let maxPoints = optValue("--max-points").flatMap { Int($0) } ?? 0 + var top: [ContourResult] = [] + for (i, c) in o.topLevelContours.enumerated() { + let b = Box(unROI(c.normalizedPath.boundingBox), 1) + var pts: [[Double]]? = nil + if maxPoints > 0 { + let all = c.normalizedPoints + let stride = max(1, all.count / maxPoints) + pts = Swift.stride(from: 0, to: all.count, by: stride).map { + let q = unROI(CGRect(x: CGFloat(all[$0].x), y: CGFloat(all[$0].y), width: 0, height: 0)) + return [Double(q.origin.x), 1.0 - Double(q.origin.y)] + } + } + top.append(ContourResult(index: i, pointCount: c.pointCount, childCount: c.childContourCount, + x: b.x, y: b.y, width: b.width, height: b.height, points: pts)) + } + emit(encodeJSON(ContoursResult(totalContours: o.contourCount, topLevel: top))) + exit(0) +} + +// ─── Saliency (attention | objectness) ─────────────────────────────────────── + +if isSaliency { + let kind = optValue("--saliency") ?? "attention" + let request: VNImageBasedRequest = kind == "objectness" + ? VNGenerateObjectnessBasedSaliencyImageRequest() + : VNGenerateAttentionBasedSaliencyImageRequest() + perform([request], "saliency") + guard let o = request.results?.first as? VNSaliencyImageObservation else { + emit(encodeJSON(SaliencyResult(regions: [], heatmapPath: nil))) + exit(0) + } + let regions = (o.salientObjects ?? []).map { Box($0.boundingBox, $0.confidence) } + var heat: String? = nil + if let out = optValue("--out"), let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) { heat = out } + emit(encodeJSON(SaliencyResult(regions: regions, heatmapPath: heat))) + exit(0) +} + +// ─── Foreground subject cutout / person mask ───────────────────────────────── + +if isForeground { +#if SDK_14 + if #available(macOS 14.0, *) { + let out = requireOut() + let request = VNGenerateForegroundInstanceMaskRequest() + perform([request], "foreground mask") + guard let o = request.results?.first as? VNInstanceMaskObservation else { + emit(encodeJSON(MaskResult(instances: 0, outPath: ""))) + exit(0) + } + let maskOnly = args.contains("--mask-only") + do { + let pb = maskOnly + ? try o.generateScaledMaskForImage(forInstances: o.allInstances, from: handler) + : try o.generateMaskedImage(ofInstances: o.allInstances, from: handler, croppedToInstancesExtent: args.contains("--tight")) + guard let cg = cgFromPixelBuffer(pb), writePNG(cg, to: out) else { throw NSError(domain: "vision-helper", code: 1) } + } catch { + fail("could not write foreground image: \(error.localizedDescription)") + } + emit(encodeJSON(MaskResult(instances: o.allInstances.count, outPath: out))) + exit(0) + } +#endif + fail("--foreground-mask requires macOS 14 or newer", code: 2) +} + +if isPersonMask { + let out = requireOut() + let request = VNGeneratePersonSegmentationRequest() + request.qualityLevel = .accurate + request.outputPixelFormat = kCVPixelFormatType_OneComponent8 + perform([request], "person segmentation") + guard let o = request.results?.first as? VNPixelBufferObservation, + let cg = cgFromPixelBuffer(o.pixelBuffer), writePNG(cg, to: out) else { + fail("could not write person mask") + } + emit(encodeJSON(MaskResult(instances: 1, outPath: out))) + exit(0) +} + +// ─── Aesthetics (macOS 15+) ────────────────────────────────────────────────── + +if isAesthetics { +#if SDK_15 + if #available(macOS 15.0, *) { + let request = VNCalculateImageAestheticsScoresRequest() + perform([request], "aesthetics") + guard let o = request.results?.first as? VNImageAestheticsScoresObservation else { + emit("null") + exit(0) + } + emit(encodeJSON(AestheticsResult(overallScore: o.overallScore, isUtility: o.isUtility))) + exit(0) + } +#endif + fail("--aesthetics requires macOS 15 or newer", code: 2) +} + // ─── Faces ─────────────────────────────────────────────────────────────────── if isFaces { - var results: [FaceResult] = [] + var results: [Box] = [] let request = VNDetectFaceRectanglesRequest { (req, _) in guard let obs = req.results as? [VNFaceObservation] else { return } - for o in obs { - let box = o.boundingBox - results.append(FaceResult( - x: Double(box.origin.x), - y: flipY(Double(box.origin.y), Double(box.size.height)), - w: Double(box.size.width), - h: Double(box.size.height), - confidence: o.confidence - )) - } - } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision face detection failed: \(error.localizedDescription)\n", stderr) - exit(1) + results = obs.map { Box($0.boundingBox, $0.confidence) } } - print(encodeJSON(results)) + perform([request], "face detection") + emit(encodeJSON(results)) exit(0) } @@ -148,25 +991,15 @@ if isBarcodes { let request = VNDetectBarcodesRequest { (req, _) in guard let obs = req.results as? [VNBarcodeObservation] else { return } for o in obs { - let box = o.boundingBox + let b = Box(o.boundingBox, o.confidence) results.append(BarcodeResult( - type: o.symbology.rawValue, - value: o.payloadStringValue ?? "", - x: Double(box.origin.x), - y: flipY(Double(box.origin.y), Double(box.size.height)), - w: Double(box.size.width), - h: Double(box.size.height), - confidence: o.confidence + type: o.symbology.rawValue, value: o.payloadStringValue ?? "", + x: b.x, y: b.y, width: b.width, height: b.height, confidence: o.confidence )) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision barcode detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "barcode detection") + emit(encodeJSON(results)) exit(0) } @@ -184,14 +1017,9 @@ if isRectangles { )) } } - (request as VNDetectRectanglesRequest).maximumObservations = 0 - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision rectangle detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + request.maximumObservations = 0 + perform([request], "rectangle detection") + emit(encodeJSON(results)) exit(0) } @@ -209,13 +1037,8 @@ if isDocument { )) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision document detection failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "document detection") + emit(encodeJSON(results)) exit(0) } @@ -230,12 +1053,7 @@ if isClassify { results.append(ClassificationResult(identifier: o.identifier, confidence: o.confidence)) } } - do { - try handler.perform([request]) - } catch { - fputs("ERROR: Vision classification failed: \(error.localizedDescription)\n", stderr) - exit(1) - } - print(encodeJSON(results)) + perform([request], "classification") + emit(encodeJSON(results)) exit(0) } diff --git a/src/ui.ts b/src/ui.ts new file mode 100644 index 0000000..5142b3b --- /dev/null +++ b/src/ui.ts @@ -0,0 +1,224 @@ +// UI layer: screen capture + window/display/permission introspection. +// +// Privacy invariant: no function in this module ever returns image bytes. +// Captures land on disk; only paths, geometry, and metadata flow back. +// The library never synthesizes input — eyes, not hands. + +import { readFile } from 'fs/promises'; +import { UI_BIN, runHelper, execHelper, tmpOutPath, sha256 } from './helper.js'; + +const UI_HELPER_TIMEOUT_MS = 15_000; +const SCREENCAPTURE_TIMEOUT_MS = 15_000; + +// ─── Types ─────────────────────────────────────────────────────────────────── + +export interface WindowInfo { + /** CGWindowID — stable for the window's lifetime */ + windowId: number; + /** Owning application name, e.g. 'Safari' */ + app: string; + pid: number; + /** Window title (may be empty) */ + title: string; + /** Global screen points, top-left origin (CGEvent click space) */ + x: number; + y: number; + w: number; + h: number; + /** 0 = normal app window; menu bar / dock / overlays only with `listWindows(true)` */ + layer: number; + isOnScreen: boolean; +} + +export interface DisplayInfo { + displayId: number; + isMain: boolean; + /** Global screen points, top-left origin */ + x: number; + y: number; + w: number; + h: number; + /** Backing scale factor (2 on Retina) */ + scale: number; +} + +export interface PermissionsInfo { + screenRecording: boolean; + accessibility: boolean; + /** True while the login session is locked — no capture is useful until unlocked. */ + screenLocked: boolean; +} + +/** Rectangle in global screen points, top-left origin (CGEvent click space). */ +export interface ScreenFrame { + x: number; + y: number; + w: number; + h: number; +} + +export interface CaptureResult { + /** Absolute path to the PNG on disk */ + path: string; + /** Pixel dimensions of the PNG. */ + pixelWidth: number; + pixelHeight: number; + /** Screen region the image covers, in global screen points (top-left origin). */ + frame: ScreenFrame; + /** pixelWidth / frame.w — ≈2 on Retina */ + scale: number; + /** SHA-256 of the PNG bytes — pin assertions to a specific capture, detect replaced files */ + sha256: string; + /** ISO timestamp */ + capturedAt: string; + /** Human-readable description of what was captured */ + target: string; +} + +export interface CaptureOptions { + windowId?: number; + /** App name (exact or case-insensitive prefix) — its frontmost window wins. */ + app?: string; + /** Region in global screen points, top-left origin. */ + rect?: ScreenFrame; + displayId?: number; + /** Where to write the PNG. Default: `$TMPDIR/macos-vision/capture-.png` */ + outPath?: string; +} + +// ─── ui-helper ─────────────────────────────────────────────────────────────── + +const runUi = (...args: string[]) => + runHelper(UI_BIN, args, { timeout: UI_HELPER_TIMEOUT_MS }); + +/** On-screen windows, front-to-back. Pass `true` to include menu bar, dock and overlays. */ +export function listWindows(includeAll = false): Promise { + return runUi('--windows', ...(includeAll ? ['--all'] : [])); +} + +/** Online displays (including asleep ones) with bounds and backing scale. */ +export function listDisplays(): Promise { + return runUi('--displays'); +} + +/** Screen Recording / Accessibility status for the current host process. */ +export function checkPermissions(): Promise { + return runUi('--permissions'); +} + +// ─── Capture ───────────────────────────────────────────────────────────────── + +/** Width/height straight from the PNG IHDR header (bytes 16–23). */ +function pngPixelSize(png: Buffer): { w: number; h: number } { + return { w: png.readUInt32BE(16), h: png.readUInt32BE(20) }; +} + +async function resolveWindow(opts: CaptureOptions): Promise { + const windows = await listWindows(); + if (opts.windowId != null) { + const win = windows.find((w) => w.windowId === opts.windowId); + if (!win) throw new Error(`Window ${opts.windowId} not found (use listWindows())`); + return win; + } + if (opts.app) { + const q = opts.app.toLowerCase(); + // CGWindowList is front-to-back — first match is the frontmost window of that app. + // App-name matching only: title search would silently capture the wrong app. + const win = windows.find((w) => w.app.toLowerCase() === q || w.app.toLowerCase().startsWith(q)); + if (!win) { + const apps = [...new Set(windows.map((w) => w.app))].join(', '); + throw new Error(`No on-screen window matches app "${opts.app}". Visible apps: ${apps}`); + } + return win; + } + throw new Error('window target requires windowId or app'); +} + +// Screen Recording can only change for this process via an app restart, so one +// successful preflight is valid for the process lifetime. +let screenRecordingOk = false; + +/** + * Captures a window (`windowId` / `app`), a region (`rect`) or a display + * (`displayId`, default: main) to a PNG via `/usr/sbin/screencapture`. + * Returns the file path and geometry — never the image bytes. + */ +export async function captureScreen(opts: CaptureOptions = {}): Promise { + if (!screenRecordingOk) { + const perms = await checkPermissions(); + if (!perms.screenRecording) { + throw new Error( + 'Screen Recording permission missing. Grant it to the host application ' + + '(Terminal / Claude Desktop / Cursor) in System Settings → Privacy & Security → Screen Recording, then restart it.' + ); + } + screenRecordingOk = true; + } + + const out = tmpOutPath('capture', opts.outPath); + const args = ['-x', '-t', 'png']; + let frame: ScreenFrame; + let targetDesc: string; + + if (opts.windowId != null || opts.app) { + const win = await resolveWindow(opts); + args.push('-o', '-l', String(win.windowId)); // -o: no window shadow + frame = { x: win.x, y: win.y, w: win.w, h: win.h }; + targetDesc = `window ${win.windowId} (${win.app}${win.title ? `: ${win.title}` : ''})`; + } else if (opts.rect) { + const { x, y, w, h } = opts.rect; + args.push(`-R${x},${y},${w},${h}`); + frame = { x, y, w, h }; + targetDesc = `region ${x},${y} ${w}×${h}`; + } else { + const displays = await listDisplays(); + const display = + opts.displayId != null + ? displays.find((d) => d.displayId === opts.displayId) + : (displays.find((d) => d.isMain) ?? displays[0]); + if (!display) throw new Error(`Display ${opts.displayId} not found`); + const idx = displays.indexOf(display) + 1; // screencapture -D is 1-based ordinal + args.push('-D', String(idx)); + frame = { x: display.x, y: display.y, w: display.w, h: display.h }; + targetDesc = `display ${display.displayId}`; + } + + args.push(out); + try { + await execHelper('/usr/sbin/screencapture', args, { timeout: SCREENCAPTURE_TIMEOUT_MS }); + } catch (err) { + const stderr = (err as { stderr?: string }).stderr?.trim(); + // Ask the system rather than guessing: on a locked Mac window and region + // capture fail outright and a full-screen capture returns only the lock + // screen, so retrying cannot succeed until someone unlocks. + const locked = await checkPermissions() + .then((p) => p.screenLocked) + .catch(() => false); + throw new Error( + locked + ? `screencapture failed for ${targetDesc}: the screen is locked. ` + + 'Window and region capture do not work on a locked Mac, and a full-screen ' + + 'capture would only show the lock screen. Ask the user to unlock, then retry — ' + + 'retrying while locked cannot succeed.' + : `screencapture failed for ${targetDesc}${stderr ? ` (${stderr})` : ''}. ` + + 'The window may have closed, or the display may be asleep.' + ); + } + let bytes: Buffer; + try { + bytes = await readFile(out); + } catch { + throw new Error(`screencapture produced no file for ${targetDesc} — is the window on screen?`); + } + const px = pngPixelSize(bytes); + return { + path: out, + pixelWidth: px.w, + pixelHeight: px.h, + sha256: sha256(bytes), + frame, + scale: frame.w > 0 ? px.w / frame.w : 1, + capturedAt: new Date().toISOString(), + target: targetDesc, + }; +} diff --git a/src/vision.ts b/src/vision.ts new file mode 100644 index 0000000..5d3cb3b --- /dev/null +++ b/src/vision.ts @@ -0,0 +1,492 @@ +// Extended Vision API: everything beyond the classic OCR/faces/barcodes set. +// +// All geometry is normalized 0–1 with a TOP-LEFT origin unless stated otherwise. +// Functions that produce pixels (masks, crops, heatmaps) write PNG files and +// return paths — never image bytes. + +import { resolve } from 'path'; +import { + VISION_BIN, + runHelper, + runGated, + tmpOutPath, + UnsupportedOnThisMacOSError, +} from './helper.js'; + +export { UnsupportedOnThisMacOSError }; + +// ─── Shared types ──────────────────────────────────────────────────────────── + +/** Normalized rectangle, 0–1, top-left origin. */ +export interface NormalizedRect { + x: number; + y: number; + width: number; + height: number; +} + +/** A detection: where it is and how sure Vision is. */ +export type Detection = NormalizedRect & { confidence: number }; + +/** Options shared by every text-recognition path (OCR, text regions, document structure). */ +export interface TextRecognitionOptions { + /** BCP-47 codes in priority order, e.g. `['pl-PL', 'en-US']`. See `supportedOcrLanguages()`. */ + languages?: string[]; + /** Let Vision pick the language per text run. Default false. */ + autoDetectLanguage?: boolean; + /** Apply language-model correction. Default true. Disable for IDs, codes, IBANs, hashes. */ + languageCorrection?: boolean; + /** Domain vocabulary that should win over the language model, e.g. product names. */ + customWords?: string[]; + /** `.fast` recognition level — noticeably quicker, lower accuracy. Default false. */ + fast?: boolean; + /** Only look inside this region (normalized, top-left origin). Results are still reported in full-image space. */ + regionOfInterest?: NormalizedRect; + /** Ignore text shorter than this fraction of image height (0–1). */ + minTextHeight?: number; +} + +/** Serialises shared text options into helper flags. Canonical — also used as the OCR cache key. */ +export function textOptionArgs(o: TextRecognitionOptions = {}): string[] { + const args: string[] = []; + if (o.languages?.length) args.push('--lang', o.languages.join(',')); + if (o.autoDetectLanguage) args.push('--auto-lang'); + if (o.languageCorrection === false) args.push('--no-correction'); + if (o.customWords?.length) args.push('--custom-words', o.customWords.join(',')); + if (o.fast) args.push('--fast'); + if (o.regionOfInterest) { + const r = o.regionOfInterest; + args.push('--roi', `${r.x},${r.y},${r.width},${r.height}`); + } + if (o.minTextHeight !== undefined) args.push('--min-text-height', String(o.minTextHeight)); + return args; +} + +const run = (args: string[], input?: string) => + runHelper(VISION_BIN, args, { timeout: 60_000, input }); +const gated = (args: string[], feature: string, minVersion: string) => + runGated(VISION_BIN, args, feature, minVersion, { timeout: 60_000 }); + +// ─── Capabilities ──────────────────────────────────────────────────────────── + +export interface VisionCapabilities { + /** Version of the bundled `vision-helper` protocol */ + helperVersion: string; + macosVersion: string; + /** BCP-47 codes Vision can OCR on this machine */ + ocrLanguages: string[]; + /** Feature → available on this macOS. Gate agent plans on this. */ + features: Record; +} + +let capabilitiesPromise: Promise | undefined; + +/** What this machine can do. Constant for the process lifetime, so memoized. */ +export function visionCapabilities(): Promise { + capabilitiesPromise ??= run(['--capabilities']).catch((err) => { + capabilitiesPromise = undefined; + throw err; + }); + return capabilitiesPromise; +} + +/** BCP-47 codes supported by the accurate OCR model. */ +export function supportedOcrLanguages(): Promise { + return run(['--languages']); +} + +// ─── Image info ────────────────────────────────────────────────────────────── + +export interface ImageInfo { + width: number; + height: number; + hasAlpha: boolean; + bitsPerComponent: number; + colorSpace?: string; + dpi?: number; + /** EXIF orientation 1–8 when present */ + orientation?: number; + /** UTI, e.g. 'public.png', 'public.jpeg' */ + format?: string; +} + +/** Pixel dimensions and metadata without running any model. */ +export function imageInfo(imagePath: string): Promise { + return run(['--image-info', resolve(imagePath)]); +} + +// ─── Text regions (no recognition) ─────────────────────────────────────────── + +export type TextRegion = Detection; + +/** Where text is, without reading it. Much faster than OCR — use to pick regions of interest. */ +export function detectTextRegions( + imagePath: string, + options: Pick = {} +): Promise { + return run(['--text-rects', ...textOptionArgs(options), resolve(imagePath)]); +} + +// ─── Image similarity ──────────────────────────────────────────────────────── + +export interface ImageComparison { + /** Feature-print distance. 0 = identical; < ~0.3 visually the same scene; > ~0.8 different content. */ + distance: number; +} + +/** Compare two images semantically (Vision feature prints). Robust to small shifts and compression. */ +export function compareImages(imagePathA: string, imagePathB: string): Promise { + return run(['--compare', resolve(imagePathA), resolve(imagePathB)]); +} + +// ─── Entities in text (NSDataDetector) ─────────────────────────────────────── + +export interface TextEntity { + type: 'link' | 'email' | 'phone' | 'address' | 'date' | 'transit' | 'unknown'; + /** Matched substring */ + text: string; + /** UTF-16 offsets into the input */ + start: number; + end: number; + /** Normalised value: absolute URL, e-mail, phone, ISO-8601 date, "street, city, zip" */ + value?: string; + /** Structured parts (address components, event duration, transit info) */ + components?: Record; +} + +/** Links, e-mails, phones, addresses and dates in plain text. Pure Foundation — no model. */ +export function extractEntities(text: string): Promise { + return run(['--entities'], text); +} + +// ─── Document structure (macOS 26+) ────────────────────────────────────────── + +export interface DocLine { + text: string; + confidence: number; + bbox: NormalizedRect; +} +export interface DocText { + text: string; + alignment?: 'leading' | 'center' | 'trailing'; + bbox: NormalizedRect; + lines: DocLine[]; +} +export interface DocCell { + text: string; + row: number; + col: number; + rowSpan: number; + colSpan: number; + bbox: NormalizedRect; +} +export interface DocTable { + rowCount: number; + columnCount: number; + /** Cell texts by row; spanning cells repeat in every row they cover */ + rows: string[][]; + /** Unique cells with spans */ + cells: DocCell[]; + bbox: NormalizedRect; +} +export interface DocListItem { + marker: string; + text: string; + bbox: NormalizedRect; +} +export interface DocList { + items: DocListItem[]; + bbox: NormalizedRect; +} +export interface DocDetectedData { + type: + | 'link' + | 'email' + | 'phone' + | 'address' + | 'date' + | 'money' + | 'flight' + | 'tracking' + | 'measurement' + | 'payment' + | 'unknown'; + text: string; + value?: string; + bbox: NormalizedRect; +} +export interface DocBarcode { + type: string; + value: string; + bbox: NormalizedRect; +} + +export interface DocumentStructure { + /** Detected title block, if any */ + title?: DocText; + /** Full transcript in reading order */ + text: string; + paragraphs: DocText[]; + tables: DocTable[]; + lists: DocList[]; + barcodes: DocBarcode[]; + /** Data detectors run on the transcript: links, e-mails, phones, money, dates… with positions */ + detectedData: DocDetectedData[]; +} + +/** + * Native document understanding (macOS 26+): paragraphs, tables, lists, title, + * barcodes and detected data with positions — no heuristics, no LLM. + * Throws `UnsupportedOnThisMacOSError` on older systems; check `visionCapabilities().features.documentStructure`. + */ +export function recognizeDocument( + imagePath: string, + options: TextRecognitionOptions = {} +): Promise { + return gated( + ['--document-structure', ...textOptionArgs(options), resolve(imagePath)], + 'recognizeDocument', + '26' + ); +} + +// ─── Quality signals ───────────────────────────────────────────────────────── + +export interface LensSmudge { + /** 0–1 likelihood the lens was dirty/smudged */ + confidence: number; + /** False when the smudge model is unavailable on this hardware — treat confidence as unknown */ + supported: boolean; +} + +/** Was the photo taken through a dirty lens? macOS 26+. Returns `supported:false` instead of guessing. */ +export async function detectLensSmudge(imagePath: string): Promise { + try { + return await gated(['--smudge', resolve(imagePath)], 'detectLensSmudge', '26'); + } catch (err) { + if (err instanceof UnsupportedOnThisMacOSError) return { confidence: 0, supported: false }; + throw err; + } +} + +export interface AestheticsScore { + /** -1…1, higher is nicer */ + overallScore: number; + /** True for screenshots, receipts, documents — "utility" images rather than photos */ + isUtility: boolean; +} + +/** Photo aesthetics + utility flag (macOS 15+). Good for "is this a screenshot or a photo?". */ +export function imageAesthetics(imagePath: string): Promise { + return gated( + ['--aesthetics', resolve(imagePath)], + 'imageAesthetics', + '15' + ); +} + +export interface Horizon { + /** Tilt in degrees; positive = clockwise. */ + angleDegrees: number; +} + +/** Horizon tilt for photos; null when no horizon is detected. */ +export function detectHorizon(imagePath: string): Promise { + return run(['--horizon', resolve(imagePath)]); +} + +// ─── People, faces, poses, animals ─────────────────────────────────────────── + +export interface FaceLandmarks extends Detection { + /** Head rotation in degrees, when available */ + roll?: number; + yaw?: number; + pitch?: number; + /** 0–1 sharpness/exposure quality of the face crop */ + captureQuality?: number; + /** Region name → polyline of [x, y] normalized points */ + landmarks: Record; +} + +export function detectFaceLandmarks(imagePath: string): Promise { + return run(['--face-landmarks', resolve(imagePath)]); +} + +export type HumanBox = Detection; + +/** Full-body person boxes. */ +export function detectHumans(imagePath: string): Promise { + return run(['--humans', resolve(imagePath)]); +} + +export interface Keypoint { + x: number; + y: number; + confidence: number; +} + +export interface Pose { + /** Joint name (Vision key, e.g. 'left_wrist_joint') → point. Only joints with confidence > 0. */ + joints: Record; + confidence: number; + /** Hands only */ + chirality?: 'left' | 'right' | 'unknown'; +} + +export function detectBodyPose(imagePath: string): Promise { + return run(['--body-pose', resolve(imagePath)]); +} + +export function detectHandPose(imagePath: string): Promise { + return run(['--hand-pose', resolve(imagePath)]); +} + +/** macOS 14+. */ +export async function detectAnimalPose(imagePath: string): Promise { + try { + return await run(['--animal-pose', resolve(imagePath)]); + } catch (err) { + if ((err as { code?: number }).code === 2) + throw new UnsupportedOnThisMacOSError('detectAnimalPose', '14'); + throw err; + } +} + +export interface Animal extends Detection { + /** e.g. [{ identifier: 'Cat', confidence: 0.98 }] */ + labels: Array<{ identifier: string; confidence: number }>; +} + +/** Cats and dogs with boxes. */ +export function detectAnimals(imagePath: string): Promise { + return run(['--animals', resolve(imagePath)]); +} + +// ─── Saliency, contours ────────────────────────────────────────────────────── + +export interface SaliencyOptions { + /** 'attention' = where a human would look; 'objectness' = where objects are. Default 'attention'. */ + mode?: 'attention' | 'objectness'; + /** Write the heatmap PNG here (optional). */ + heatmapPath?: string; +} + +export interface Saliency { + /** Up to 3 salient regions, sorted by the model */ + regions: Detection[]; + heatmapPath?: string; +} + +export function detectSaliency( + imagePath: string, + options: SaliencyOptions = {} +): Promise { + const args = ['--saliency', options.mode ?? 'attention']; + if (options.heatmapPath) args.push('--out', resolve(options.heatmapPath)); + args.push(resolve(imagePath)); + return run(args); +} + +export interface ContourOptions { + /** Include up to N evenly-sampled points per top-level contour. Default 0 (boxes only). */ + maxPoints?: number; + /** Detect light shapes on a dark background instead of dark-on-light. */ + lightOnDark?: boolean; + regionOfInterest?: NormalizedRect; +} + +export interface Contour extends NormalizedRect { + index: number; + pointCount: number; + childCount: number; + points?: [number, number][]; +} + +export interface Contours { + totalContours: number; + topLevel: Contour[]; +} + +/** Edge/shape contours — useful for charts, diagrams, UI boundaries. */ +export function detectContours(imagePath: string, options: ContourOptions = {}): Promise { + const args = ['--contours']; + if (options.maxPoints) args.push('--max-points', String(options.maxPoints)); + if (options.lightOnDark) args.push('--light-on-dark'); + args.push(...textOptionArgs({ regionOfInterest: options.regionOfInterest }), resolve(imagePath)); + return run(args); +} + +// ─── Pixel-producing operations (write PNG, return path) ───────────────────── + +export interface CropResult { + outPath: string; + width: number; + height: number; +} + +/** Crop a normalized region (top-left origin) to a PNG. Cheap "zoom in" for a second OCR pass. */ +export function cropImage( + imagePath: string, + region: NormalizedRect, + out?: string +): Promise { + return run([ + '--crop', + `${region.x},${region.y},${region.width},${region.height}`, + '--out', + tmpOutPath('crop', out), + resolve(imagePath), + ]); +} + +/** Detect the document in a photo and write a perspective-corrected, deskewed PNG. */ +export function cropDocument(imagePath: string, out?: string): Promise { + return run([ + '--document-crop', + '--out', + tmpOutPath('document', out), + resolve(imagePath), + ]); +} + +export interface MaskResult { + /** Number of foreground instances found (0 → nothing written) */ + instances: number; + outPath: string; +} + +export interface ForegroundOptions { + /** Write only the alpha mask instead of the masked subject. */ + maskOnly?: boolean; + /** Crop the output to the subject's extent. */ + tight?: boolean; + out?: string; +} + +/** Subject cutout (macOS 14+): transparent-background PNG of the main foreground object(s). */ +export async function extractForeground( + imagePath: string, + options: ForegroundOptions = {} +): Promise { + const args = ['--foreground-mask', '--out', tmpOutPath('foreground', options.out)]; + if (options.maskOnly) args.push('--mask-only'); + if (options.tight) args.push('--tight'); + args.push(resolve(imagePath)); + try { + return await run(args); + } catch (err) { + if ((err as { code?: number }).code === 2) + throw new UnsupportedOnThisMacOSError('extractForeground', '14'); + throw err; + } +} + +/** Person segmentation mask as an 8-bit grayscale PNG (white = person). */ +export function personMask(imagePath: string, out?: string): Promise { + return run([ + '--person-mask', + '--out', + tmpOutPath('person-mask', out), + resolve(imagePath), + ]); +} diff --git a/test/vision.test.ts b/test/vision.test.ts new file mode 100644 index 0000000..6fefbff --- /dev/null +++ b/test/vision.test.ts @@ -0,0 +1,341 @@ +import { describe, it, expect } from 'vitest'; +import { existsSync } from 'fs'; +import { resolve, dirname } from 'path'; +import { fileURLToPath } from 'url'; +import { tmpdir } from 'os'; +import { + ocr, + visionCapabilities, + supportedOcrLanguages, + imageInfo, + detectTextRegions, + compareImages, + extractEntities, + recognizeDocument, + detectLensSmudge, + detectHumans, + detectBodyPose, + detectSaliency, + detectContours, + cropImage, + cropDocument, + extractForeground, + imageAesthetics, + detectAnimalPose, + UnsupportedOnThisMacOSError, + listDisplays, + checkPermissions, + captureScreen, +} from '../src/index.js'; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const SAMPLE_IMG = resolve(__dirname, 'fixtures/sample.png'); +const SAMPLE_PDF = resolve(__dirname, 'fixtures/sample.pdf'); +const T = 30_000; + +describe('visionCapabilities()', () => { + it('reports helper version, macOS version and feature flags', async () => { + const caps = await visionCapabilities(); + expect(caps.helperVersion).toBeTruthy(); + expect(caps.macosVersion).toMatch(/^\d+\.\d+/); + expect(caps.features.ocr).toBe(true); + expect(typeof caps.features.documentStructure).toBe('boolean'); + expect(caps.ocrLanguages.length).toBeGreaterThan(5); + }); + + it('supportedOcrLanguages() includes English', async () => { + expect(await supportedOcrLanguages()).toContain('en-US'); + }); +}); + +describe('ocr() — tuning options', () => { + it( + 'regionOfInterest restricts results but reports full-image coordinates', + async () => { + const top = await ocr(SAMPLE_IMG, { + format: 'blocks', + regionOfInterest: { x: 0, y: 0, width: 1, height: 0.1 }, + }); + expect(top.length).toBeGreaterThan(0); + for (const b of top) { + expect(b.y).toBeLessThan(0.12); + expect(b.height).toBeLessThan(0.1); + } + expect(top.map((b) => b.text).join(' ')).toContain('Henry VIII'); + }, + T + ); + + it( + 'languages + languageCorrection:false still reads the fixture', + async () => { + const text = await ocr(SAMPLE_IMG, { languages: ['en-US'], languageCorrection: false }); + expect(text).toContain('Henry VIII'); + }, + T + ); + + it( + 'fast mode returns text', + async () => { + const text = await ocr(SAMPLE_IMG, { fast: true }); + expect(text).toContain('Henry'); + }, + T + ); + + it( + 'cache:true returns identical result on second call', + async () => { + const opts = { cache: true, customWords: ['Wikipedia'] } as const; + const a = await ocr(SAMPLE_IMG, opts); + const started = Date.now(); + const b = await ocr(SAMPLE_IMG, opts); + expect(b).toBe(a); + expect(Date.now() - started).toBeLessThan(200); + }, + T + ); + + it('onProgress fires once per PDF page', async () => { + const calls: Array<[number, number]> = []; + await ocr(SAMPLE_PDF, { onProgress: (d, n) => calls.push([d, n]) }); + expect(calls.length).toBeGreaterThan(0); + expect(calls[calls.length - 1][0]).toBe(calls[calls.length - 1][1]); + }, 60_000); +}); + +describe('imageInfo() / detectTextRegions() / compareImages()', () => { + it('imageInfo reads dimensions without a model', async () => { + const info = await imageInfo(SAMPLE_IMG); + expect(info.width).toBe(1088); + expect(info.height).toBe(1344); + expect(info.format).toBe('public.png'); + }); + + it( + 'detectTextRegions finds many text boxes', + async () => { + const regions = await detectTextRegions(SAMPLE_IMG); + expect(regions.length).toBeGreaterThan(20); + for (const r of regions) { + expect(r.x).toBeGreaterThanOrEqual(-0.01); + expect(r.y).toBeGreaterThanOrEqual(-0.01); + expect(r.width).toBeGreaterThan(0); + } + }, + T + ); + + it( + 'compareImages: identical → 0, different → > 0.5', + async () => { + expect((await compareImages(SAMPLE_IMG, SAMPLE_IMG)).distance).toBe(0); + const crop = await cropImage(SAMPLE_IMG, { x: 0.6, y: 0.1, width: 0.3, height: 0.3 }); + expect((await compareImages(SAMPLE_IMG, crop.outPath)).distance).toBeGreaterThan(0.3); + }, + T + ); +}); + +describe('extractEntities()', () => { + it('finds email, phone, url and date', async () => { + const ents = await extractEntities( + 'Kontakt: jan@example.com, tel. +48 601 234 567, https://prorok.pl, spotkanie 1 września 2026 o 14:00' + ); + const types = ents.map((e) => e.type); + expect(types).toContain('email'); + expect(types).toContain('phone'); + expect(types).toContain('link'); + expect(types).toContain('date'); + expect(ents.find((e) => e.type === 'email')?.value).toBe('jan@example.com'); + }); +}); + +describe('recognizeDocument()', () => { + it('returns structure on macOS 26+, throws UnsupportedOnThisMacOSError otherwise', async () => { + const caps = await visionCapabilities(); + if (!caps.features.documentStructure) { + await expect(recognizeDocument(SAMPLE_IMG)).rejects.toBeInstanceOf( + UnsupportedOnThisMacOSError + ); + return; + } + const doc = await recognizeDocument(SAMPLE_IMG, { languages: ['en-US'] }); + expect(doc.text).toContain('Henry VIII'); + expect(doc.paragraphs.length).toBeGreaterThan(5); + expect(doc.title?.text).toContain('Henry'); + expect(Array.isArray(doc.tables)).toBe(true); + }, 60_000); + + it( + 'detectLensSmudge never throws on supported systems', + async () => { + const caps = await visionCapabilities(); + const r = await detectLensSmudge(SAMPLE_IMG); + expect(typeof r.supported).toBe('boolean'); + if (!caps.features.lensSmudge) expect(r.supported).toBe(false); + }, + T + ); +}); + +describe('people / saliency / contours', () => { + it( + 'detectHumans finds the portrait on the fixture', + async () => { + const humans = await detectHumans(SAMPLE_IMG); + expect(humans.length).toBeGreaterThanOrEqual(1); + expect(humans[0].x).toBeGreaterThan(0.5); + }, + T + ); + + it( + 'detectBodyPose returns named joints', + async () => { + const poses = await detectBodyPose(SAMPLE_IMG); + expect(poses.length).toBeGreaterThanOrEqual(1); + expect(Object.keys(poses[0].joints).length).toBeGreaterThan(5); + }, + T + ); + + it( + 'detectSaliency returns regions and can write a heatmap', + async () => { + const out = resolve(tmpdir(), `macos-vision-test-heat-${Date.now()}.png`); + const s = await detectSaliency(SAMPLE_IMG, { mode: 'objectness', heatmapPath: out }); + expect(s.regions.length).toBeGreaterThan(0); + expect(s.heatmapPath).toBe(out); + expect(existsSync(out)).toBe(true); + }, + T + ); + + it( + 'detectContours counts contours and samples points', + async () => { + const c = await detectContours(SAMPLE_IMG, { maxPoints: 4 }); + expect(c.totalContours).toBeGreaterThan(10); + expect(c.topLevel[0].points?.length).toBeLessThanOrEqual(5); + }, + T + ); + + // Guards the helper↔TS wire contract: every rect-shaped result uses width/height. + it('all rect-shaped results expose numeric width/height', async () => { + const [regions, humans, contours, saliency, blocks] = await Promise.all([ + detectTextRegions(SAMPLE_IMG), + detectHumans(SAMPLE_IMG), + detectContours(SAMPLE_IMG, { maxPoints: 2 }), + detectSaliency(SAMPLE_IMG, { mode: 'objectness' }), + ocr(SAMPLE_IMG, { format: 'blocks' }), + ]); + const rects = [...regions, ...humans, ...contours.topLevel, ...saliency.regions, ...blocks]; + expect(rects.length).toBeGreaterThan(0); + for (const r of rects) { + expect(typeof r.width).toBe('number'); + expect(typeof r.height).toBe('number'); + expect(Number.isFinite(r.width)).toBe(true); + } + }, 60_000); + + it('recognizeDocument bboxes use width/height too', async () => { + const caps = await visionCapabilities(); + if (!caps.features.documentStructure) return; + const doc = await recognizeDocument(SAMPLE_IMG, { languages: ['en-US'] }); + for (const p of doc.paragraphs.slice(0, 5)) { + expect(typeof p.bbox.width).toBe('number'); + expect(typeof p.bbox.height).toBe('number'); + } + }, 60_000); +}); + +describe('pixel ops return paths, never bytes', () => { + it('cropImage writes a PNG of the requested size', async () => { + const r = await cropImage(SAMPLE_IMG, { x: 0, y: 0, width: 0.5, height: 0.5 }); + expect(existsSync(r.outPath)).toBe(true); + expect(r.width).toBe(544); + expect(r.height).toBe(672); + }); + + it( + 'cropDocument writes a perspective-corrected PNG', + async () => { + const r = await cropDocument(SAMPLE_IMG); + expect(existsSync(r.outPath)).toBe(true); + expect(r.width).toBeGreaterThan(100); + }, + T + ); + + it( + 'extractForeground writes a cutout (macOS 14+)', + async () => { + const caps = await visionCapabilities(); + if (!caps.features.foregroundMask) return; + const r = await extractForeground(SAMPLE_IMG, { tight: true }); + expect(r.instances).toBeGreaterThanOrEqual(1); + expect(existsSync(r.outPath)).toBe(true); + }, + T + ); +}); + +describe('helper error reporting', () => { + // A helper failure must surface its own message, not Node's + // "Command failed: /long/path/to/vision-helper --flags …". + it('surfaces the helper ERROR line, not the spawn command', async () => { + await expect(imageInfo('/nonexistent-xyz.png')).rejects.toThrow(/^Cannot open file:/); + }); + + it('keeps the exit status so gating still works', async () => { + await imageInfo('/nonexistent-xyz.png').catch((err) => { + expect((err as { code?: number }).code).toBe(1); + }); + }); + + it('captureScreen surfaces screencapture stderr', async () => { + await expect(captureScreen({ rect: { x: 0, y: 0, w: -5, h: -5 } })).rejects.toThrow( + /does not intersect any displays/ + ); + }); +}); + +describe('capability gating', () => { + // The helper reports a feature only when both its SDK and this macOS provide it; + // anything reported unavailable must fail with the typed error, not a raw crash. + it.each([ + ['documentStructure', () => recognizeDocument(SAMPLE_IMG)], + ['aesthetics', () => imageAesthetics(SAMPLE_IMG)], + ['foregroundMask', () => extractForeground(SAMPLE_IMG)], + ['animalPose', () => detectAnimalPose(SAMPLE_IMG)], + ])( + 'caps.%s agrees with the call', + async (feature, call) => { + const caps = await visionCapabilities(); + if (caps.features[feature]) { + await expect(call()).resolves.toBeDefined(); + } else { + await expect(call()).rejects.toBeInstanceOf(UnsupportedOnThisMacOSError); + } + }, + 60_000 + ); +}); + +describe('ui-helper', () => { + it('listDisplays reports at least one display with a scale', async () => { + const d = await listDisplays(); + expect(d.length).toBeGreaterThanOrEqual(1); + expect(d.some((x) => x.isMain)).toBe(true); + expect(d[0].scale).toBeGreaterThanOrEqual(1); + }); + + it('checkPermissions returns booleans', async () => { + const p = await checkPermissions(); + expect(typeof p.screenRecording).toBe('boolean'); + expect(typeof p.accessibility).toBe('boolean'); + }); +}); diff --git a/vitest.config.ts b/vitest.config.ts new file mode 100644 index 0000000..b186783 --- /dev/null +++ b/vitest.config.ts @@ -0,0 +1,17 @@ +import { defineConfig } from 'vitest/config'; + +export default defineConfig({ + test: { + // Every test spawns a native helper that runs a Vision model. Under the + // default 5s timeout these pass in isolation but flake when the suite runs + // in parallel and saturates the machine. + testTimeout: 30_000, + hookTimeout: 30_000, + // Vision requests are already CPU/ANE-bound; running many files at once + // buys nothing and is what pushed the suite over its timeouts. + fileParallelism: false, + // The repo's own worktrees contain copies of these tests — running them + // twice doubles an already slow, model-bound suite. + exclude: ['**/node_modules/**', '**/dist/**', '**/.claude/worktrees/**'], + }, +});