Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions signaling-server/.env.example
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,17 @@ RAG_PROVIDER_ORDER=cohere,voyage,local
RAG_PROVIDER_TIMEOUT_MS=20000
RAG_PROVIDER_MIN_INTERVAL_MS=250

# ── Content analysis / OCR (optional; TINY hosts keep off) ───────────────────
# Image-file OCR via tesseract.js — transiently allocates 120–200MB, so it is
# tier-gated: requires workload ceiling ≥300MB (STANDARD/LARGE or explicit).
RAG_OCR_ENABLED=false
RAG_OCR_LANG=eng
RAG_OCR_MAX_BYTES=5242880

# ── Health monitor ────────────────────────────────────────────────────────────
RAG_HEALTH_INTERVAL_MS=60000
RAG_DISCONTINUE_AFTER_CHECKS=3

# ── Adaptive memory tiers ─────────────────────────────────────────────────────
# The server auto-detects its cgroup memory limit at boot:
# TINY <768MB · STANDARD ≤2048MB · LARGE >2048MB
Expand Down
115 changes: 115 additions & 0 deletions signaling-server/package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions signaling-server/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@
"multer": "^2.1.1",
"pino": "^10.3.1",
"pino-pretty": "^13.1.3",
"tesseract.js": "^7.0.0",
"typescript": "^5.9.3",
"unpdf": "^1.8.1",
"ws": "^8.19.0",
Expand Down
8 changes: 8 additions & 0 deletions signaling-server/src/config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,14 @@ export const CONFIG = {
RAG_ALLOW_LOCAL_TINY: process.env.RAG_ALLOW_LOCAL_TINY === 'true',
// Disable the local embedder entirely (APIs only), any tier.
RAG_DISABLE_LOCAL: process.env.RAG_DISABLE_LOCAL === 'true',
// ── Content analysis / OCR (P2) ─────────────────────────────────────────
// Image-file OCR via tesseract.js. OFF by default: OCR allocates 120–200MB
// transiently (WASM runtime + language data), which cannot be guaranteed
// under a 480MB target — TINY hosts keep it off unless explicitly enabled,
// and every page is RSS-guarded before recognition starts.
RAG_OCR_ENABLED: process.env.RAG_OCR_ENABLED === 'true',
RAG_OCR_LANG: process.env.RAG_OCR_LANG || 'eng',
RAG_OCR_MAX_BYTES: parseInt(process.env.RAG_OCR_MAX_BYTES ?? String(5 * 1024 * 1024), 10),
// Direct-stuffing budget is TOKEN-based, tied to the active Groq model's
// context window (review §4) — never a bare character constant.
LLM_CONTEXT_TOKENS: parseInt(process.env.LLM_CONTEXT_TOKENS ?? '131072', 10),
Expand Down
76 changes: 76 additions & 0 deletions signaling-server/src/rag/analyzer/file-analyzer.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,76 @@
import type { ExtractedDoc } from '../types'
import type { WorkloadClass } from '../embedding/selector'

// ── File Analyzer (adaptive services ①) ─────────────────────────────────────
// Cheap post-extraction classification: consumes the file's metadata plus
// its ExtractedDoc (never re-reads the source) and produces everything the
// selector/router needs:
//
// kind pdf | docx | xlsx | text | image | unknown
// requiresOcr extracted text ~empty while pages exist ⇒ scanned/image
// estimatedTokens conservative ceil(chars/3)
// workload low | medium | high | very_high (drives model selection)
//
// File SIZE is never used alone (doc §16): a 200KB txt can out-work a 5MB
// scanned PDF only in byte terms — the score reflects EXTRACTED reality.

export interface AnalyzerInput {
name: string
mimeType: string
sizeBytes: number
doc: ExtractedDoc | null // null ⇒ unsupported/failed before extraction
}

export interface FileAnalysis {
kind: 'pdf' | 'docx' | 'xlsx' | 'text' | 'image' | 'unknown'
requiresOcr: boolean
totalChars: number
pageCount: number
emptyPages: number
estimatedTokens: number
workload: WorkloadClass
/** Human-readable note surfaced through aiStats when content was skipped. */
notice?: string
}

function extOf(name: string): string {
const i = name.lastIndexOf('.')
return i >= 0 ? name.slice(i + 1).toLowerCase() : ''
}

export function analyzeFile(input: AnalyzerInput): FileAnalysis {
const ext = extOf(input.name)
let kind: FileAnalysis['kind'] = 'unknown'
if (ext === 'pdf' || input.mimeType === 'application/pdf') kind = 'pdf'
else if (['docx'].includes(ext) || input.mimeType.includes('wordprocessingml')) kind = 'docx'
else if (['xlsx', 'xls', 'xlsm'].includes(ext) || input.mimeType.includes('spreadsheetml')) kind = 'xlsx'
else if (input.mimeType.startsWith('image/') || ['png', 'jpg', 'jpeg', 'webp', 'bmp'].includes(ext)) kind = 'image'
else kind = 'text'

const pages = input.doc?.pages ?? []
const totalChars = pages.reduce((n, p) => n + p.text.length, 0)
const pageCount = pages.length
const emptyPages = pages.filter(p => p.text.trim().length === 0).length

// Scanned/OCR signal: pages exist but carry (almost) no text.
const requiresOcr =
kind === 'image' ||
(pageCount > 0 && emptyPages > 0 && totalChars < pageCount * 40)

const estimatedTokens = Math.ceil(totalChars / 3)

let workload: WorkloadClass
if (!requiresOcr && estimatedTokens <= 4_000) workload = 'low'
else if (!requiresOcr && estimatedTokens <= 32_000) workload = 'medium'
else if (!requiresOcr && estimatedTokens <= 120_000) workload = 'high'
else workload = 'very_high'

let notice: string | undefined
if (kind === 'image' && !input.doc) {
notice = 'Image file — enable RAG_OCR_ENABLED (on hosts with memory headroom) to extract text.'
} else if (requiresOcr && kind === 'pdf') {
notice = `PDF appears scanned (${emptyPages}/${pageCount} pages without text).`
}

return { kind, requiresOcr, totalChars, pageCount, emptyPages, estimatedTokens, workload, notice }
}
Loading
Loading