-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmemory.toml
More file actions
48 lines (42 loc) · 2.37 KB
/
Copy pathmemory.toml
File metadata and controls
48 lines (42 loc) · 2.37 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
[daemon]
# Unload ONNX models (embedding + reranker) after this many seconds of inactivity.
# Env var VODOU_MEMORY_MODEL_IDLE_SECS overrides this. Set to 0 to disable.
model_idle_secs = 0
# Alpha: keep embed/rerank warm — cold reload was blowing Cursor's 5s hook
# budget (sequential prompt timeouts at exactly 5001ms). Set to e.g. 120 to
# reclaim RAM on idle machines; 0 = never unload.
# Reranker model: "bge-base" (default, ~1 GB) | "jina-turbo" (~150 MB, English) | "bge-v2-m3" (~1.1 GB)
# Switch to jina-turbo to cut reranker RAM by ~7x with minimal quality loss for English content.
# bge-base is the quality default and what the 0.70 library-match floor is
# calibrated against. jina-turbo (~150 MB vs ~1 GB) was a local RAM choice; its
# logits run compressed (5.16 -> 1.94 on an identical pair), so swapping models
# silently rescales every calibrated threshold. Chosen deliberately 2026-08-10.
rerank_model = "bge-base"
# Set to "off" to disable the cross-encoder reranker entirely (~1.2 GB savings at peak).
# Search falls back to RRF hybrid fusion (FTS5 + vector) — still good quality.
# reranker = "off"
# How often the LIVE extraction cycle runs — the loop that turns a recorded
# turn into a searchable fact. Was hardcoded at 300s. Measured 2026-08-29:
# a phone turn took 3m51s to become retrievable, of which ~20s was actual work
# and ~3m31s was waiting for this timer. Lowered to 60s. Env
# VODOU_EXTRACT_INTERVAL_SECS overrides; floor is 10s.
# A tick with nothing pending is a cheap queue peek, and the low-priority
# import drain still only runs on ticks where live extraction did no work.
extract_interval_secs = 60
[extraction]
# "auto" reads the LLM provider currently selected in the gateway (gateway.db
# `gateway_settings.llm_provider`) and uses that provider's credentials for
# extraction. Set to a specific provider (claude / anthropic / openai / google /
# ollama / etc.) to override. VODOU_MEMORY_EXTRACTION_PROVIDER in .env wins
# over this file.
provider = "auto"
[extraction.claude]
# Haiku is 3-5x faster than sonnet for tagging-style extraction; drops avg flush from ~30-60s to ~6-15s.
model = "haiku"
[extraction.anthropic]
# Direct-API model. The code default (claude-3-5-haiku-20241022) is RETIRED —
# API returns not_found_error for it (2026-07-17).
model = "claude-haiku-4-5-20251001"
[extraction.ollama]
model = "qwen2.5:3b-instruct"
base_url = "http://localhost:11434"