From 052845991371eec935a0613b3f230977619cef2d Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 02:27:57 +0000 Subject: [PATCH 01/10] Quick wins: 4 new init clients, HTTP bearer-token auth, multi-OS CI, doctor --json - `midas init` now wires VS Code (user mcp.json, `servers` key), Gemini CLI, Cline, and Zed (`context_servers`), each with its own config schema; status/uninstall understand the per-client server-map keys too. - `midas serve --http` accepts MIDAS_MCP_TOKEN / --token: a constant-time bearer-token ASGI gate, since the HTTP transport otherwise exposes the whole store to any local process. - CI runs the core+MCP suite on Windows and macOS (the client-wiring code is full of per-OS paths that only Linux exercised). - `midas doctor --json` emits the same machine-readable envelope as the wiring receipt. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- .github/workflows/ci.yml | 19 ++++++ README.md | 10 ++- midas/cli.py | 132 +++++++++++++++++++++++++++------------ midas/mcp_server.py | 36 ++++++++++- tests/test_cli.py | 65 +++++++++++++++++++ tests/test_mcp_server.py | 28 +++++++++ 6 files changed, 247 insertions(+), 43 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9755601..2f0e453 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,3 +20,22 @@ jobs: run: pip install ".[all,dev]" - name: Run the suite run: python -m pytest -q + + # The client-wiring code is full of per-OS paths (Claude Desktop, VS Code, Zed live in three + # different places) — exercise it where those branches actually run. Core + MCP only: the heavy + # embedding extras are covered by the Linux job, and the suite degrades gracefully without them. + test-os: + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [windows-latest, macos-latest] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - name: Install (core + MCP + dev) + run: pip install ".[mcp,dev]" numpy + - name: Run the suite + run: python -m pytest -q diff --git a/README.md b/README.md index 1f220b8..4525c82 100755 --- a/README.md +++ b/README.md @@ -120,13 +120,15 @@ wired to, under which scope/policy, and which clients were skipped (config paths contents). Paste it into a bug report, or let another agent verify the setup without scraping prose. `midas init` creates **one shared memory** (`~/.midas/memory.sqlite3`) and points the MCP clients it -detects — **Claude Code, Codex, Cursor, Claude Desktop, Windsurf** — at it. So all your agents read and -write the **same** memory, autonomously, with no per-client paths to keep in sync. +detects — **Claude Code, Codex, Cursor, Claude Desktop, Windsurf, VS Code, Gemini CLI, Cline, Zed** — +at it. So all your agents read and write the **same** memory, autonomously, with no per-client paths to +keep in sync. Prefer a single endpoint over per-client launches? Run one server and give your clients an **MCP URL**: ```bash midas serve --http # → http://127.0.0.1:7077/mcp (one server, one memory, every client shares it) +midas serve --http --token # require `Authorization: Bearer ` on every request ``` Keep Midas current with **`midas update`**. See your memory anytime with **`midas inspect`**. @@ -148,6 +150,10 @@ by default — no path needed. The universal block: | **Claude Desktop** | Settings → Developer → Edit Config (`claude_desktop_config.json`) — paste, restart | | **Codex CLI** | `codex mcp add midas -- midas-mcp` | | **Windsurf** | `~/.codeium/windsurf/mcp_config.json` — paste the block | +| **VS Code** | user `mcp.json` (`servers` key, `"type": "stdio"`) — `midas init` writes it | +| **Gemini CLI** | `~/.gemini/settings.json` (`mcpServers` key) — `midas init` writes it | +| **Cline** | `cline_mcp_settings.json` in VS Code global storage — `midas init` writes it | +| **Zed** | `settings.json` → `context_servers` — `midas init` writes it | | **Anything else** | point it at command `midas-mcp` | | **No Python** | `npx -y midas-memory-mcp` — the [TypeScript port](packages/midas-ts) (experimental: no semantic embeddings yet) | diff --git a/midas/cli.py b/midas/cli.py index 2aa5a0c..4637be2 100644 --- a/midas/cli.py +++ b/midas/cli.py @@ -55,17 +55,60 @@ def _claude_desktop_path() -> Path | None: return Path.home() / ".config/Claude/claude_desktop_config.json" +def _vscode_user_dir() -> Path: + if sys.platform == "darwin": + return Path.home() / "Library/Application Support/Code/User" + if sys.platform.startswith("win"): + ad = os.getenv("APPDATA") + return (Path(ad) if ad else Path.home() / "AppData/Roaming") / "Code" / "User" + return Path.home() / ".config/Code/User" + + +def _zed_settings_path() -> Path: + if sys.platform.startswith("win"): + ad = os.getenv("APPDATA") + return (Path(ad) if ad else Path.home() / "AppData/Roaming") / "Zed" / "settings.json" + return Path.home() / ".config/zed/settings.json" + + +# Every JSON-configured client Midas can wire: (display name, client id, config path, JSON key holding +# the server map, block shape). Shapes differ per client — see _client_block. +def _json_clients() -> list[tuple[str, str, Path, str, str]]: + out = [("Cursor", "cursor", Path.home() / ".cursor/mcp.json", "mcpServers", "std"), + ("Windsurf", "windsurf", Path.home() / ".codeium/windsurf/mcp_config.json", + "mcpServers", "std"), + ("VS Code", "vscode", _vscode_user_dir() / "mcp.json", "servers", "vscode"), + ("Gemini CLI", "gemini-cli", Path.home() / ".gemini/settings.json", "mcpServers", "std"), + ("Cline", "cline", _vscode_user_dir() / "globalStorage/saoudrizwan.claude-dev/settings" + / "cline_mcp_settings.json", "mcpServers", "std"), + ("Zed", "zed", _zed_settings_path(), "context_servers", "zed")] + cd = _claude_desktop_path() + if cd: + out.append(("Claude Desktop", "claude-desktop", cd, "mcpServers", "std")) + return out + + +def _client_block(shape: str, env: dict) -> dict: + """The `midas` server entry in the schema each client expects.""" + if shape == "vscode": # VS Code mcp.json declares the transport type explicitly + return {"type": "stdio", "command": "midas-mcp", "env": env} + if shape == "zed": # Zed's context_servers entries are flat command + args + return {"source": "custom", "command": "midas-mcp", "args": [], "env": env} + return {"command": "midas-mcp", "env": env} + + # ---- init ------------------------------------------------------------------------------------- -def _merge_mcp_json(path: Path, block: dict, *, dry: bool) -> str: - """Add/refresh a `midas` entry in a client's mcpServers JSON, non-destructively (merge + backup).""" +def _merge_mcp_json(path: Path, block: dict, *, dry: bool, key: str = "mcpServers") -> str: + """Add/refresh a `midas` entry in a client's server-map JSON, non-destructively (merge + backup). + `key` is the JSON key holding the server map (mcpServers / servers / context_servers).""" cfg: dict = {} if path.exists(): try: cfg = json.loads(path.read_text() or "{}") except Exception: return f"⚠ {path} is not valid JSON — add `midas` by hand" - servers = cfg.setdefault("mcpServers", {}) + servers = cfg.setdefault(key, {}) verb = "update" if "midas" in servers else "add" if dry: return f"would {verb} → {path}" @@ -192,19 +235,14 @@ def already_wired(path: Path) -> bool: changed=ok and not dry, reason=msg if (dry or not ok) else None) # Clients configured by a JSON file (only the ones already present, unless --all): - targets = [("Cursor", "cursor", Path.home() / ".cursor/mcp.json"), - ("Windsurf", "windsurf", Path.home() / ".codeium/windsurf/mcp_config.json")] - cd = _claude_desktop_path() - if cd: - targets.append(("Claude Desktop", "claude-desktop", cd)) - for name, client_id, path in targets: + for name, client_id, path, key, shape in _json_clients(): if not (path.exists() or args.all): record(name, path, detected=False, wired=False, changed=False, reason="config not found (skipped — use --all to configure anyway)") continue existed, was = path.exists(), already_wired(path) - block = {"command": "midas-mcp", "env": env_for(client_id)} - msg = _merge_mcp_json(path, block, dry=dry) + block = _client_block(shape, env_for(client_id)) + msg = _merge_mcp_json(path, block, dry=dry, key=key) results.append((name, msg)) ok = not msg.startswith("⚠") record(name, path, detected=existed, wired=was if dry else ok, changed=ok and not dry, @@ -236,6 +274,8 @@ def already_wired(path: Path) -> bool: def cmd_serve(args: argparse.Namespace) -> int: if args.db: # set BEFORE importing mcp_server (it builds its store at import) os.environ["MIDAS_MCP_DB"] = _store_path(args.db) + if getattr(args, "token", None): + os.environ["MIDAS_MCP_TOKEN"] = args.token from midas import mcp_server mcp_server.run_server("http" if args.http else "stdio", host=args.host, port=args.port) @@ -244,8 +284,8 @@ def cmd_serve(args: argparse.Namespace) -> int: # ---- status ----------------------------------------------------------------------------------- -def _midas_server_entry(path: Path) -> dict | None: - """Parse the `midas` server entry out of a client config (JSON mcpServers / codex TOML), so the +def _midas_server_entry(path: Path, key: str = "mcpServers") -> dict | None: + """Parse the `midas` server entry out of a client config (JSON server map / codex TOML), so the receipt can state which command + MIDAS_* env — i.e. which memory and scope — that client actually got. Only MIDAS_* env keys are surfaced (the audit boundary excludes anything else a user added).""" text = _safe_read(path) @@ -255,9 +295,9 @@ def _midas_server_entry(path: Path) -> dict | None: if path.suffix == ".toml": import tomllib - servers = tomllib.loads(text).get("mcp_servers") or {} + servers = tomllib.loads(text).get(key) or {} else: - servers = json.loads(text).get("mcpServers") or {} + servers = json.loads(text).get(key) or {} except Exception: return None entry = servers.get("midas") @@ -294,9 +334,9 @@ def cmd_status(args: argparse.Namespace) -> int: clients: list[dict] = [] ns_values: set = set() - for name, p in _client_paths(): + for name, p, key in _client_paths(): wired = p.exists() and "midas" in _safe_read(p) - entry = _midas_server_entry(p) if wired else None + entry = _midas_server_entry(p, key) if wired else None env = entry["env"] if entry else {} if entry: ns_values.add(env.get("MIDAS_MCP_NAMESPACE")) @@ -360,19 +400,16 @@ def cmd_inspect(args: argparse.Namespace) -> int: # ---- shared client detection ------------------------------------------------------------------ -def _client_paths() -> list[tuple[str, Path]]: - paths = [("Claude Code", Path.home() / ".claude.json"), - ("Cursor", Path.home() / ".cursor/mcp.json"), - ("Codex", Path.home() / ".codex/config.toml"), - ("Windsurf", Path.home() / ".codeium/windsurf/mcp_config.json")] - cd = _claude_desktop_path() - if cd: - paths.append(("Claude Desktop", cd)) +def _client_paths() -> list[tuple[str, Path, str]]: + """Every client Midas knows how to check: (name, config path, JSON/TOML key of the server map).""" + paths = [("Claude Code", Path.home() / ".claude.json", "mcpServers"), + ("Codex", Path.home() / ".codex/config.toml", "mcp_servers")] + paths += [(name, p, key) for name, _cid, p, key, _shape in _json_clients()] return paths def _wired_clients() -> list[str]: - return [name for name, p in _client_paths() if p.exists() and "midas" in _safe_read(p)] + return [name for name, p, _k in _client_paths() if p.exists() and "midas" in _safe_read(p)] # ---- doctor ----------------------------------------------------------------------------------- @@ -380,14 +417,11 @@ def _wired_clients() -> list[str]: def cmd_doctor(args: argparse.Namespace) -> int: from midas import __version__ - ok = True + checks: list[dict] = [] def check(good: bool, label: str, hint: str = "") -> None: - nonlocal ok - ok = ok and good - print(f" {'✓' if good else '⚠'} {label}" + (f" — {hint}" if hint and not good else "")) + checks.append({"ok": good, "check": label, "hint": hint if (hint and not good) else None}) - print(f"Midas {__version__} · Python {sys.version.split()[0]}\n") check(shutil.which("midas-mcp") is not None, "midas-mcp on PATH", "reinstall, or use the absolute path from `which midas-mcp` in your client config") @@ -418,6 +452,27 @@ def check(good: bool, label: str, hint: str = "") -> None: wired = _wired_clients() check(bool(wired), f"clients wired: {', '.join(wired) if wired else 'none'}", "run `midas init`") + ok = all(c["ok"] for c in checks) + + if getattr(args, "json", False): # same audit boundary as the wiring receipt: no memory contents + from datetime import datetime, timezone + + print(json.dumps({ + "midas_version": __version__, + "receipt_kind": "doctor", + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds") + .replace("+00:00", "Z"), + "python": sys.version.split()[0], + "memory_db": db, + "ok": ok, + "checks": checks, + "clients_wired": wired, + }, indent=2)) + return 0 if ok else 1 + + print(f"Midas {__version__} · Python {sys.version.split()[0]}\n") + for c in checks: + print(f" {'✓' if c['ok'] else '⚠'} {c['check']}" + (f" — {c['hint']}" if c["hint"] else "")) print("\n" + ("Everything looks good." if ok else "Some checks need attention (see ⚠ above).")) return 0 if ok else 1 @@ -488,12 +543,7 @@ def cmd_uninstall(args: argparse.Namespace) -> int: r = _cli_remove("codex", ["codex", "mcp", "remove", "midas"]) if r: results.append(("Codex", r)) - json_targets = [("Cursor", Path.home() / ".cursor/mcp.json"), - ("Windsurf", Path.home() / ".codeium/windsurf/mcp_config.json")] - cd = _claude_desktop_path() - if cd: - json_targets.append(("Claude Desktop", cd)) - for name, path in json_targets: + for name, _cid, path, key, _shape in _json_clients(): if not path.exists(): continue try: @@ -501,9 +551,9 @@ def cmd_uninstall(args: argparse.Namespace) -> int: except Exception: results.append((name, "left as-is (unparseable)")) continue - if "midas" in cfg.get("mcpServers", {}): + if "midas" in cfg.get(key, {}): shutil.copy(path, str(path) + ".midas-bak") - del cfg["mcpServers"]["midas"] + del cfg[key]["midas"] path.write_text(json.dumps(cfg, indent=2) + "\n") results.append((name, "removed")) for name, msg in results: @@ -609,6 +659,8 @@ def main() -> None: ps.add_argument("--http", action="store_true", help="serve over HTTP at an MCP URL all clients share") ps.add_argument("--host", default="127.0.0.1") ps.add_argument("--port", type=int, default=7077) + ps.add_argument("--token", help="require this bearer token on every HTTP request " + "(same as MIDAS_MCP_TOKEN)") ps.add_argument("--db") ps.set_defaults(func=cmd_serve) @@ -633,6 +685,8 @@ def main() -> None: pd = sub.add_parser("doctor", help="diagnose your install, store, and client config") pd.add_argument("--db") + pd.add_argument("--json", action="store_true", + help="print a machine-readable diagnosis instead of text") pd.set_defaults(func=cmd_doctor) pe = sub.add_parser("export", help="export all memory to JSON (backup / move machines)") diff --git a/midas/mcp_server.py b/midas/mcp_server.py index fd6a01f..40bd8e0 100755 --- a/midas/mcp_server.py +++ b/midas/mcp_server.py @@ -796,9 +796,34 @@ def _auto_maintain_loop(interval_seconds: float) -> None: print(f"[midas-mcp] auto-maintain error: {exc}", file=sys.stderr) +class _BearerAuth: + """Minimal ASGI middleware: reject any HTTP request that doesn't carry `Authorization: Bearer + `. The HTTP transport turns one process into a shared memory endpoint — without + this, ANY local process (or anyone on the interface you bind) could read and write the whole store. + Constant-time comparison; stdio transport is unaffected.""" + + def __init__(self, app, token: str) -> None: + self.app, self._expected = app, f"Bearer {token}" + + async def __call__(self, scope, receive, send): + if scope["type"] == "http": + import hmac + + sent = next((v.decode("latin-1") for k, v in scope.get("headers") or [] + if k == b"authorization"), "") + if not hmac.compare_digest(sent, self._expected): + await send({"type": "http.response.start", "status": 401, + "headers": [(b"content-type", b"text/plain"), + (b"www-authenticate", b"Bearer")]}) + await send({"type": "http.response.body", "body": b"unauthorized"}) + return + await self.app(scope, receive, send) + + def run_server(transport: str = "stdio", host: str = "127.0.0.1", port: int = 7077) -> None: """Run the MCP server. transport='stdio' (default — each client launches it) or 'http' (one shared - server at an MCP URL: http://host:port/mcp). The `midas` CLI calls this; `midas-mcp` uses stdio.""" + server at an MCP URL: http://host:port/mcp). The `midas` CLI calls this; `midas-mcp` uses stdio. + Set MIDAS_MCP_TOKEN (or `midas serve --token`) to require a bearer token on the HTTP transport.""" interval_min = int(os.getenv("MIDAS_MCP_AUTO_MAINTAIN", "0") or "0") if interval_min > 0: import threading @@ -811,7 +836,14 @@ def run_server(transport: str = "stdio", host: str = "127.0.0.1", port: int = 70 server.settings.host = host server.settings.port = port print(f"[midas-mcp] MCP URL -> http://{host}:{port}/mcp (Ctrl-C to stop)", file=sys.stderr) - server.run(transport="streamable-http") + token = os.getenv("MIDAS_MCP_TOKEN", "") + if token: + import uvicorn + + print("[midas-mcp] bearer-token auth enabled (MIDAS_MCP_TOKEN)", file=sys.stderr) + uvicorn.run(_BearerAuth(server.streamable_http_app(), token), host=host, port=port) + else: + server.run(transport="streamable-http") else: server.run() diff --git a/tests/test_cli.py b/tests/test_cli.py index aeb0b5a..fb4d590 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -114,6 +114,71 @@ def test_status_json_parses_codex_toml(tmp_path, capsys, monkeypatch) -> None: assert receipt["scope_mode"] == "shared" # no namespace anywhere → one shared pool +def test_client_block_shapes() -> None: + env = {"MIDAS_MCP_EMBEDDER": "local"} + assert cli._client_block("std", env) == {"command": "midas-mcp", "env": env} + assert cli._client_block("vscode", env)["type"] == "stdio" # VS Code declares the transport + zed = cli._client_block("zed", env) + assert zed["source"] == "custom" and zed["args"] == [] # Zed's context_servers schema + + +def test_merge_mcp_json_custom_key(tmp_path) -> None: + p = tmp_path / "mcp.json" + p.write_text(json.dumps({"servers": {"other": {"command": "x"}}})) + cli._merge_mcp_json(p, cli._client_block("vscode", {}), dry=False, key="servers") + cfg = json.loads(p.read_text()) + assert set(cfg["servers"]) == {"other", "midas"} # merged under VS Code's key + assert "mcpServers" not in cfg # never invents the wrong key + + +def test_init_wires_new_clients(tmp_path, capsys, monkeypatch) -> None: + monkeypatch.setattr(cli.shutil, "which", lambda exe: None) + monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + vscode = tmp_path / ".config/Code/User/mcp.json" + vscode.parent.mkdir(parents=True) + vscode.write_text("{}") + gemini = tmp_path / ".gemini/settings.json" + gemini.parent.mkdir(parents=True) + gemini.write_text(json.dumps({"theme": "dark"})) + rc = cli.cmd_init(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), dry_run=False, + all=False, json=True)) + assert rc == 0 + by = {c["client"]: c for c in json.loads(capsys.readouterr().out)["clients"]} + assert by["VS Code"]["wired"] and by["Gemini CLI"]["wired"] + assert by["Zed"]["wired"] is False and by["Cline"]["wired"] is False # not present → skipped + assert json.loads(vscode.read_text())["servers"]["midas"]["type"] == "stdio" + cfg = json.loads(gemini.read_text()) + assert cfg["mcpServers"]["midas"]["command"] == "midas-mcp" + assert cfg["theme"] == "dark" # other settings survive the merge + + # status sees them through their own keys + cli.cmd_status(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), json=True)) + sby = {c["client"]: c for c in json.loads(capsys.readouterr().out)["clients"]} + assert sby["VS Code"]["wired"] and sby["VS Code"]["server_command"] == "midas-mcp" + assert sby["Gemini CLI"]["wired"] and sby["Gemini CLI"]["client_id"] == "gemini-cli" + + +def test_uninstall_removes_from_custom_key(tmp_path, capsys, monkeypatch) -> None: + monkeypatch.setattr(cli.shutil, "which", lambda exe: None) + monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + vscode = tmp_path / ".config/Code/User/mcp.json" + vscode.parent.mkdir(parents=True) + vscode.write_text(json.dumps({"servers": {"midas": {"command": "midas-mcp"}, + "other": {"command": "x"}}})) + rc = cli.cmd_uninstall(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), purge=False)) + assert rc == 0 + assert list(json.loads(vscode.read_text())["servers"]) == ["other"] # midas gone, other kept + + +def test_doctor_json(tmp_path, capsys, monkeypatch) -> None: + monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + rc = cli.cmd_doctor(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), json=True)) + out = json.loads(capsys.readouterr().out) + assert out["receipt_kind"] == "doctor" and out["ok"] is (rc == 0) + assert any("store" in c["check"] and not c["ok"] for c in out["checks"]) # missing store flagged + assert all({"ok", "check", "hint"} <= set(c) for c in out["checks"]) + + def test_derive_scope_modes() -> None: assert cli._derive_scope(set()) == ("shared", None, []) assert cli._derive_scope({None}) == ("shared", None, []) diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py index dd3ceb8..d2c6ef9 100755 --- a/tests/test_mcp_server.py +++ b/tests/test_mcp_server.py @@ -341,3 +341,31 @@ def test_distill_prompt_drives_no_llm_distillation(): assert "compact, self-contained" in text assert "capture" in text and "proj-a" in text assert "no extra tool/model needed" in text or "no Midas-LLM" in text or "no extra" in text + + +def test_bearer_auth_middleware_gates_http() -> None: + """The HTTP transport must reject requests without the exact bearer token (401 + WWW-Authenticate), + pass authorized ones through, and leave non-http scopes (lifespan) untouched.""" + import asyncio + + from midas.mcp_server import _BearerAuth + + async def app(scope, receive, send): + await send({"type": "http.response.start", "status": 200, "headers": []}) + + mw = _BearerAuth(app, "s3cret") + + def call(scope_type: str, auth: str | None): + headers = [(b"authorization", auth.encode())] if auth else [] + events: list = [] + + async def send(ev): + events.append(ev) + + asyncio.run(mw({"type": scope_type, "headers": headers}, None, send)) + return events + + assert call("http", "Bearer s3cret")[0]["status"] == 200 + assert call("http", "Bearer wrong")[0]["status"] == 401 + assert call("http", None)[0]["status"] == 401 + assert call("lifespan", None)[0]["status"] == 200 # non-http scopes pass straight through From 9dba7d0f516b6c73454c129098375a98cc8604fd Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 02:43:52 +0000 Subject: [PATCH 02/10] Control-plane: memory_conflicts, resume, open loops, per-kind TTL retention MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - midas/continuity.py — three primitives similarity search can't express: * memory_conflicts: live beliefs that contradict each other with neither superseding the other (the multi-agent shared-memory failure mode). NLI- scored when available; else a same-slot heuristic (value swap on a shared frame, numbers disagree, one side negates). Ranked candidates only — nothing is resolved silently. * open loops: kind="commitment" records a promise, close_loop supersedes it with its resolution, open_loops lists what's still unclosed (oldest first). * resume: the one-call session-onboarding pack — pinned directives, forbidden rules, what changed, current state, open loops, conflicts — token-budgeted and prompt-ready. - Memory.forget_expired(ttl_by_kind) + MIDAS_MCP_TTL / maintain(ttl=...): age-based per-kind retention; user-confirmed, standing, and supersession- chain records never expire silently. - MCP tools: resume, memory_conflicts, open_loops, remember_commitment, close_loop; agent instructions now start sessions with resume and keep promises via open loops. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- README.md | 14 +- midas/__init__.py | 23 ++- midas/continuity.py | 294 ++++++++++++++++++++++++++++++++++++ midas/mcp_server.py | 136 ++++++++++++++++- midas/memory.py | 29 ++++ midas/policy.py | 21 ++- midas/types.py | 3 +- tests/test_control_plane.py | 198 ++++++++++++++++++++++++ tests/test_mcp_server.py | 57 +++++++ 9 files changed, 761 insertions(+), 14 deletions(-) create mode 100644 midas/continuity.py create mode 100644 tests/test_control_plane.py diff --git a/README.md b/README.md index 4525c82..34c4a57 100755 --- a/README.md +++ b/README.md @@ -192,15 +192,19 @@ server runs in. Or scope it manually per project/agent/user with `MIDAS_MCP_NAME All tools & env knobs **Tools:** `remember`, `capture` (policy-gated auto-store), `recall` (source-traceable), `build_context` -(compact, dated, today-anchored prompt block), `memory_state` (current project state), `memory_diff` -(what changed since), `check_memory_use` (guard), `memory_policy`, `maintain` (dedup + forgetting, returns -a deletion audit), `stats`, `forget` (chain-safe), `forget_matching` (topic-level erasure, dry-run by -default), `forget_all`. Prompts: `memory_session`, `distill`. +(compact, dated, today-anchored prompt block), `resume` (the one-call session-onboarding pack: pinned + +state + changes + open loops + conflicts), `memory_state` (current project state), `memory_diff` +(what changed since), `memory_conflicts` (live beliefs that contradict each other, ranked), +`open_loops` / `remember_commitment` / `close_loop` (promised work that survives sessions), +`check_memory_use` (guard), `memory_policy`, `maintain` (TTL + dedup + forgetting, returns a deletion +audit), `stats`, `forget` (chain-safe), `forget_matching` (topic-level erasure, dry-run by default), +`forget_all`. Prompts: `memory_session`, `distill`. **Env:** `MIDAS_MCP_DB` · `MIDAS_MCP_EMBEDDER` (`local` / `hashing` / `multilingual` / any fastembed id) · `MIDAS_MCP_MAX_RECORDS` · `MIDAS_MCP_MIN_IMPORTANCE` · `MIDAS_MCP_NAMESPACE` (`=auto` → per-project scope) · `MIDAS_MCP_ANN=1` (sub-linear IVF for huge stores) · `MIDAS_MCP_SUPERSEDE` · `MIDAS_MCP_NLI=1` (NLI-gated revision) · -`MIDAS_MCP_AUTO_MAINTAIN=` (idle-time upkeep) · `MIDAS_MCP_PINNED` (pin standing directives). +`MIDAS_MCP_AUTO_MAINTAIN=` (idle-time upkeep) · `MIDAS_MCP_PINNED` (pin standing directives) · +`MIDAS_MCP_TTL` (per-kind retention, e.g. `chat=30,note=90`) · `MIDAS_MCP_TOKEN` (HTTP bearer auth). diff --git a/midas/__init__.py b/midas/__init__.py index a16231e..847b960 100755 --- a/midas/__init__.py +++ b/midas/__init__.py @@ -24,10 +24,24 @@ ProvenanceStamp, decide_memory_use, ) +from .continuity import ( + Conflict, + close_loop, + memory_conflicts, + open_loops, + remember_commitment, + resume, +) from .distill import Distiller, HTTPDistiller, OllamaDistiller from .importance import ContentImportance, StructuralImportance, is_standing_instruction from .memory import CaptureResult, ContextBlock, Memory, Reranker, approx_tokens, format_record -from .policy import AGENT_MEMORY_INSTRUCTIONS, DEFAULT_POLICY, MemoryPolicy, policy_summary +from .policy import ( + AGENT_MEMORY_INSTRUCTIONS, + DEFAULT_POLICY, + MemoryPolicy, + parse_ttl_spec, + policy_summary, +) from .index import MemoryStore, VectorIndex from .store import InMemoryStore from .types import MEMORY_KINDS, MEMORY_PROVENANCE, MemoryKind, MemoryProvenance, MemoryRecord, RecallHit @@ -68,6 +82,13 @@ "DEFAULT_POLICY", "AGENT_MEMORY_INSTRUCTIONS", "policy_summary", + "parse_ttl_spec", + "Conflict", + "memory_conflicts", + "resume", + "open_loops", + "remember_commitment", + "close_loop", "CaptureResult", "ContextBlock", "Reranker", diff --git a/midas/continuity.py b/midas/continuity.py new file mode 100644 index 0000000..a0971a4 --- /dev/null +++ b/midas/continuity.py @@ -0,0 +1,294 @@ +"""Continuity control-plane — resume a long-horizon agent in one call, and keep multi-agent memory +coherent. + +Three primitives that similarity search cannot express, all deterministic and LLM-free: + + memory_conflicts(mem) live beliefs that CONTRADICT each other and neither superseded the other — + the failure mode of one memory shared by many agents (Claude Code writes + "the DB is PostgreSQL", Cursor writes "the DB is MySQL", both stay live). + Supersession only revises what the SAME write path saw; this surfaces what + slipped through, so a human (or the agent) can resolve it. + open_loops(mem) unresolved commitments — work an agent said it would do and never closed. + Continuity is not only facts: it is unfinished intentions. + resume(mem) the session-onboarding pack in ONE call: pinned directives, forbidden rules, + current state, what changed since last time, open loops, and unresolved + conflicts — token-budgeted and prompt-ready. + +Conflict detection composes the same gates belief revision uses (embedding similarity + topical anchor ++ entity overlap) with a contradiction signal: the local NLI model when available, else a same-slot/ +different-value heuristic (numbers disagree, or one side negates the other). Approximate by design — +it returns *candidates to resolve*, ranked, never auto-deletes anything. +""" +from __future__ import annotations + +import re +import time +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +from .memory import _content_words, _fmt_date, _proper_entities, approx_tokens, format_record +from .state import DURABLE_KINDS, MemoryDiff, _scope_match, memory_diff, memory_state +from .types import MemoryRecord + +if TYPE_CHECKING: + from .memory import Memory + +_NUM_RE = re.compile(r"\d+(?:[.,:]\d+)*") +_NEG_RE = re.compile( + r"\b(?:not|no longer|never|don'?t|doesn'?t|isn'?t|aren'?t|won'?t|can'?t|cannot|" + r"shouldn'?t|stop(?:ped)?|forbidden|banned|disallowed|deprecated)\b", + re.IGNORECASE, +) +# Sentence-initial capitalised function/cue words that _proper_entities picks up as noise ("The +# launch…", "Update: …"). Harmless for supersession (a negative guard), fatal here — they make every +# pair look like it names different things — so the conflict detector strips them. +_ENTITY_NOISE = frozenset( + "the a an this that these those we i you it they he she update updated now note reminder new " + "also actually however meanwhile today yesterday tomorrow from after before once finally".split() +) + + +def _entities(text: str) -> set[str]: + return _proper_entities(text) - _ENTITY_NOISE + + +def _numbers(text: str) -> set[str]: + return set(_NUM_RE.findall(text)) + + +def _norm(text: str) -> str: + return " ".join(text.split()).lower() + + +@dataclass(frozen=True) +class Conflict: + """Two live records that appear to disagree. `signal` says why: "nli" (the local NLI model scored + them contradictory), "numeric" (same topic, different numbers), or "negation" (one side negates). + `strength` ranks conflicts across signals (NLI contradiction score, else similarity).""" + + a: MemoryRecord + b: MemoryRecord + similarity: float + signal: str + strength: float + + +def _contradiction(mem: "Memory", a: MemoryRecord, b: MemoryRecord, similarity: float, + nli: Any, nli_threshold: float) -> tuple[str, float] | None: + """The contradiction signal for a same-topic candidate pair, or None if they merely relate.""" + if _norm(a.content) == _norm(b.content): + return None # restatements agree — that's consolidation work, not a conflict + # Topical anchor (the supersession gate): the pair must share salient words at all. + shared = _content_words(a.content) & _content_words(b.content) + if len(shared) < 2: + return None + if nli is not None: + # Contradiction is not symmetric in NLI models; take the stronger direction. + score = max(nli.contradiction(a.content, b.content), nli.contradiction(b.content, a.content)) + return ("nli", score) if score >= nli_threshold else None + # Heuristic fallback — "same slot, different value", three ways to disagree. `da`/`db` are the + # named things each side mentions that the other does not (shared entities cancel out). + na, nb = _numbers(a.content), _numbers(b.content) + ea, eb = _entities(a.content), _entities(b.content) + da, db = ea - eb, eb - ea + # 1) the frame (content words minus the named value) matches but ONE named entity differs per + # side: "the primary database is PostgreSQL" vs "... is MySQL". Restricted to agreeing numbers + # — several differing names AND differing numbers means two facts about two subjects + # (Apollo launches May 3 / Artemis launches June 9), not one slot with two values. + frame_a, frame_b = _content_words(a.content) - ea - na, _content_words(b.content) - eb - nb + if len(da) == 1 and len(db) == 1 and na == nb and len(frame_a & frame_b) >= 2: + return ("value", similarity) + # For the remaining signals, each side naming its own thing means different facts, not a conflict. + if da and db: + return None + # 2) same topic, different numbers: "launch is Sept 14" vs "launch is Sept 20". + if na and nb and na != nb: + return ("numeric", similarity) + # 3) exactly one side negates: "we support X" vs "we no longer support X". + if bool(_NEG_RE.search(a.content)) != bool(_NEG_RE.search(b.content)): + return ("negation", similarity) + return None + + +def memory_conflicts( + mem: "Memory", + *, + scope: dict[str, Any] | None = None, + kinds: tuple[str, ...] | None = DURABLE_KINDS, + min_similarity: float = 0.35, + nli: Any | None = None, + nli_threshold: float = 0.5, + pool: int = 500, + neighbors: int = 10, + limit: int = 20, +) -> list[Conflict]: + """Live beliefs that contradict each other with neither superseding the other — ranked candidates + for resolution (confirm one, forget the other, or capture the corrected value; Midas never resolves + them silently). Uses the memory's own NLI model when it has one (pass `nli=` to override); without + NLI it falls back to a same-slot heuristic: numbers disagree, or exactly one side negates. + Bounded work: the newest `pool` live records × `neighbors` nearest each. No LLM.""" + nli = nli if nli is not None else getattr(mem, "_nli", None) + live = [ + r for r in mem.store.all() + if r.superseded_by is None and r.embedding is not None + and (kinds is None or r.kind in kinds) and _scope_match(r, scope) + ] + live.sort(key=lambda r: r.updated_at, reverse=True) + live = live[:pool] + live_ids = {r.id for r in live} + + conflicts: list[Conflict] = [] + seen: set[tuple[str, str]] = set() + for record in live: + near = mem.store.search( + record.embedding, limit=neighbors, + predicate=lambda r, rid=record.id: r.id != rid and r.id in live_ids, + ) + for score, other in near: + if score < min_similarity: + break # search is sorted by similarity desc + pair = (record.id, other.id) if record.id < other.id else (other.id, record.id) + if pair in seen: + continue + seen.add(pair) + hit = _contradiction(mem, record, other, score, nli, nli_threshold) + if hit: + signal, strength = hit + conflicts.append(Conflict(a=record, b=other, similarity=score, + signal=signal, strength=strength)) + conflicts.sort(key=lambda c: c.strength, reverse=True) + return conflicts[:limit] + + +# ---- open loops (commitments) ------------------------------------------------------------------- + +def remember_commitment( + mem: "Memory", + content: str, + *, + project: str | None = None, + due: str | None = None, + actor: str | None = None, + source: str | None = None, + metadata: dict[str, Any] | None = None, +) -> MemoryRecord: + """Record a commitment — work the agent (or user) said WILL be done. Importance 4 by default so + open promises don't decay out of memory; close it with `close_loop` when the work lands.""" + meta = {**(metadata or {})} + if project: + meta["project"] = project + if due: + meta["due"] = due + return mem.remember(content, kind="commitment", importance=4, provenance="planning", + actor=actor, source=source, metadata=meta) + + +def open_loops( + mem: "Memory", + *, + scope: dict[str, Any] | None = None, + limit: int = 50, +) -> list[MemoryRecord]: + """Live (unclosed) commitments, OLDEST first — the longest-open promise is the most overdue. + Closed loops are superseded by their resolution, so they drop out here but keep their history.""" + records = [ + r for r in mem.store.all() + if r.superseded_by is None and r.kind == "commitment" + and not r.metadata.get("closes") # a resolution record is the close, not a new loop + and _scope_match(r, scope) + ] + records.sort(key=lambda r: r.created_at) + return records[:limit] if limit else records + + +def close_loop(mem: "Memory", loop_id: str, resolution: str, *, actor: str | None = None) -> MemoryRecord: + """Close a commitment: store the resolution (provenance=action) and supersede the open loop with + it, so `open_loops` no longer returns it while the full promise→resolution chain stays auditable.""" + target = mem.store.get(loop_id) + if target is None or target.kind != "commitment": + raise ValueError(f"no open commitment with id '{loop_id}'") + closing = mem.remember( + f"Done: {resolution}", kind="commitment", importance=2, provenance="action", actor=actor, + metadata={k: v for k, v in target.metadata.items() if k in ("project", "namespace")} + | {"closes": loop_id}, + ) + target.superseded_by = closing.id + target.metadata = {**target.metadata, "superseded_at": closing.created_at} + mem.store.put(target) + return closing + + +# ---- resume: the one-call session-onboarding pack ------------------------------------------------- + +def resume( + mem: "Memory", + *, + scope: dict[str, Any] | None = None, + project: str | None = None, + since: float | None = None, + token_budget: int = 900, + now: float | None = None, +) -> dict[str, Any]: + """Everything an agent needs to pick up where it left off, in one deterministic call: pinned + standing directives, live forbidden rules, the current durable state, what changed since `since` + (default: the last 7 days), open commitments, and unresolved conflicts. `context` is the same + content as a token-budgeted, prompt-ready block (sections dropped tail-first when over budget). + Complements `build_context` (query-driven recall); this is the query-less session start. No LLM.""" + now = now if now is not None else time.time() + since = since if since is not None else now - 7 * 86_400.0 + if project: + scope = {**(scope or {}), "project": project} + + live = [r for r in mem.store.all() if r.superseded_by is None and _scope_match(r, scope)] + pinned = sorted((r for r in live if r.metadata.get("standing") or r.kind == "mission"), + key=lambda r: r.updated_at, reverse=True) + forbidden = sorted((r for r in live if r.metadata.get("code_kind") == "forbidden_action"), + key=lambda r: r.updated_at, reverse=True) + skip = {r.id for r in pinned} | {r.id for r in forbidden} + state = [r for r in memory_state(mem, scope=scope, limit=0) if r.id not in skip] + diff: MemoryDiff = memory_diff(mem, since, scope=scope) + loops = open_loops(mem, scope=scope) + conflicts = memory_conflicts(mem, scope=scope, limit=5) + + lines: list[str] = [f"RESUME (today is {_fmt_date(now)}; changes since {_fmt_date(since)})"] + used = approx_tokens(lines[0]) + truncated = False + + def emit(header: str, rows: list[str]) -> None: + nonlocal used, truncated + if not rows: + return + block = [header] + rows + for line in block: + cost = approx_tokens(line) + if used + cost > token_budget: + truncated = True + return + lines.append(line) + used += cost + + fmt = lambda r: format_record(r, now=now) # noqa: E731 + emit("PINNED:", [fmt(r) for r in pinned]) + emit("FORBIDDEN:", [fmt(r) for r in forbidden]) + emit("CHANGED (added):", [fmt(r) for r in diff.added]) + emit("CHANGED (revised):", + [f"{fmt(o)}\n -> now: {' '.join(n.content.split())[:300]}" for o, n in diff.revised]) + emit("CURRENT STATE:", [fmt(r) for r in state]) + emit("OPEN LOOPS:", [fmt(r) for r in loops]) + emit("CONFLICTS (unresolved — verify before relying on either):", + [f"- [{c.signal}] \"{' '.join(c.a.content.split())[:200]}\" vs " + f"\"{' '.join(c.b.content.split())[:200]}\"" for c in conflicts]) + + return { + "generated_at": now, + "since": since, + "pinned": pinned, + "forbidden": forbidden, + "state": state, + "added": diff.added, + "revised": diff.revised, + "open_loops": loops, + "conflicts": conflicts, + "context": "\n".join(lines), + "truncated": truncated, + } diff --git a/midas/mcp_server.py b/midas/mcp_server.py index 40bd8e0..50b5195 100755 --- a/midas/mcp_server.py +++ b/midas/mcp_server.py @@ -32,6 +32,11 @@ and kept in context regardless of query relevance. Measured on BEAM instruction-following recall@k: 0.26 off -> 0.44 at 2 (default) -> 0.51 at 4, overall unchanged-to-better (0 = off) + MIDAS_MCP_TTL = age-based retention, "kind=days" comma list (e.g. "chat=30,note=90"), + applied on maintenance passes; user-confirmed/standing records and + supersession chains never expire (default: none) + MIDAS_MCP_TOKEN = require `Authorization: Bearer ` on the HTTP transport — without + it any local process can read/write the shared store (default: off) """ from __future__ import annotations @@ -51,6 +56,16 @@ project_state as _project_state_view, remember_code as _remember_code_impl, ) +from midas.continuity import ( + close_loop as _close_loop_impl, + memory_conflicts as _conflicts_view, + open_loops as _open_loops_view, + remember_commitment as _remember_commitment_impl, + resume as _resume_view, +) +from midas.policy import parse_ttl_spec +# Age-based retention (MIDAS_MCP_TTL="chat=30,note=90": kind -> days) applied on maintenance passes. +_TTL = parse_ttl_spec(os.getenv("MIDAS_MCP_TTL", "")) # Auto-retention cap: when set, the store is kept at or below this many records by no-LLM selective # forgetting after each write — bounded memory for long-running/enterprise deployments. @@ -507,6 +522,109 @@ def memory_diff(hours: float = 24.0, namespace: str = "") -> dict: } +@server.tool( + title="Resume session", + annotations=ToolAnnotations(title="Resume session", readOnlyHint=True, openWorldHint=False), +) +def resume(project: str = "", namespace: str = "", hours: float = 168.0, + token_budget: int = 900) -> dict: + """START OF SESSION: everything needed to pick up where the last session left off, in ONE call — + pinned standing directives, live forbidden rules, what changed in the last `hours` (default: a + week), the current durable state, open commitments, and unresolved memory conflicts. `context` is + prompt-ready and token-budgeted; use it silently, then work. Complements `build_context` (which + needs a query): resume is the query-less session start. Deterministic, no LLM. + """ + pack = _resume_view( + _mem, scope=_ns_filter(namespace), project=project or None, + since=time.time() - float(hours) * 3600.0, token_budget=int(token_budget), + ) + return { + "context": pack["context"], + "truncated": pack["truncated"], + "counts": {k: len(pack[k]) for k in + ("pinned", "forbidden", "state", "added", "revised", "open_loops", "conflicts")}, + "open_loops": [_serialize_record(r) for r in pack["open_loops"]], + "conflicts": [ + {"signal": c.signal, "similarity": _round_score(c.similarity), + "a": _serialize_record(c.a), "b": _serialize_record(c.b)} + for c in pack["conflicts"] + ], + } + + +@server.tool( + title="Memory conflicts", + annotations=ToolAnnotations(title="Memory conflicts", readOnlyHint=True, openWorldHint=False), +) +def memory_conflicts(namespace: str = "", limit: int = 10) -> dict: + """Live beliefs that CONTRADICT each other with neither superseding the other — the multi-agent + failure mode where two clients wrote opposite facts into the shared memory and both stayed live. + Returns ranked candidate pairs (NLI-scored when the local NLI model is enabled, else a same-slot + heuristic: numbers disagree / one side negates). Midas never resolves these silently: verify with + the user, then `forget` the wrong one or capture the corrected value (which supersedes). No LLM. + """ + found = _conflicts_view(_mem, scope=_ns_filter(namespace), limit=int(limit)) + return { + "count": len(found), + "conflicts": [ + {"signal": c.signal, "similarity": _round_score(c.similarity), + "strength": _round_score(c.strength), + "a": _serialize_record(c.a), "b": _serialize_record(c.b)} + for c in found + ], + } + + +@server.tool( + title="Open loops", + annotations=ToolAnnotations(title="Open loops", readOnlyHint=True, openWorldHint=False), +) +def open_loops(project: str = "", namespace: str = "", limit: int = 20) -> dict: + """Unresolved commitments — work someone said WOULD be done and never closed — oldest (most + overdue) first. Continuity is not only facts: check this when resuming so promised work isn't + silently dropped. Record one with `remember_commitment`; close it with `close_loop`. + """ + scope = _ns_filter(namespace) or {} + if project: + scope["project"] = project + records = _open_loops_view(_mem, scope=scope or None, limit=int(limit)) + return {"count": len(records), "open_loops": [_serialize_record(r) for r in records]} + + +@server.tool( + title="Remember commitment", + annotations=ToolAnnotations(title="Remember commitment", readOnlyHint=False, + destructiveHint=False), +) +def remember_commitment(content: str, project: str = "", due: str = "", session: str = "default", + namespace: str = "") -> str: + """Record a commitment (an OPEN LOOP): work you or the user said WILL be done — a promised fix, + a follow-up, a migration to finish. It stays visible in `open_loops`/`resume` until closed with + `close_loop`, so promises survive across sessions. due: optional free-text deadline. + """ + rec = _remember_commitment_impl( + _mem, content, project=project or None, due=due or None, actor=_ACTOR, + source=_source(session), metadata=_ns_metadata(namespace, session), + ) + return f"commitment recorded ({rec.id}) — close it with close_loop when done" + + +@server.tool( + title="Close loop", + annotations=ToolAnnotations(title="Close loop", readOnlyHint=False, destructiveHint=False), +) +def close_loop(loop_id: str, resolution: str) -> str: + """Close an open commitment: records the resolution and supersedes the open loop with it, so + `open_loops` stops returning it while the promise -> resolution history stays auditable. Get + `loop_id` from `open_loops` (the record id). + """ + try: + rec = _close_loop_impl(_mem, loop_id, resolution, actor=_ACTOR) + except ValueError as exc: + return str(exc) + return f"loop closed ({loop_id} -> {rec.id})" + + @server.tool( title="Project state", annotations=ToolAnnotations(title="Project state", readOnlyHint=True, openWorldHint=False), @@ -665,7 +783,8 @@ def forget_all() -> str: title="Maintain memory", annotations=ToolAnnotations(title="Maintain memory", readOnlyHint=False, destructiveHint=True), ) -def maintain(consolidate_threshold: float = 0.0, max_records: int = 0, min_value: float = 0.0) -> dict: +def maintain(consolidate_threshold: float = 0.0, max_records: int = 0, min_value: float = 0.0, + ttl: str = "") -> dict: """Run a no-LLM memory-maintenance pass and return the deletion audit. Bounds storage and keeps recall clean without sending anything to an LLM — the enterprise @@ -673,10 +792,14 @@ def maintain(consolidate_threshold: float = 0.0, max_records: int = 0, min_value - consolidate_threshold: if > 0, dedup near-duplicate restatements at this cosine (e.g. 0.95). - max_records: if > 0, forget the lowest-value tail until at most this many remain. - min_value: if > 0, forget every (non-durable, unprotected) memory scoring below this value. + - ttl: age-based retention, "kind=days" comma list (e.g. "chat=30,note=90"); empty uses the + server's MIDAS_MCP_TTL. User-confirmed/standing records and supersession chains never expire. Durable memories (facts/preferences/constraints, high importance) and supersession chains are never dropped. Returns counts and the ids removed (auditable). """ before = len(_mem.store.all()) + ttl_map = parse_ttl_spec(ttl) if ttl else _TTL + expired = _mem.forget_expired(ttl_map) if ttl_map else [] consolidated = ( _mem.consolidate(similarity_threshold=float(consolidate_threshold)) if consolidate_threshold and consolidate_threshold > 0 @@ -693,9 +816,10 @@ def maintain(consolidate_threshold: float = 0.0, max_records: int = 0, min_value return { "before": before, "remaining": len(_mem.store.all()), + "expired": len(expired), "consolidated": len(consolidated), "forgotten": len(forgotten), - "removed_ids": consolidated + forgotten, # the deletion audit trail + "removed_ids": expired + consolidated + forgotten, # the deletion audit trail } @@ -773,14 +897,16 @@ def distill(session: str = "default") -> str: def _run_maintenance_pass() -> dict: - """One no-LLM upkeep pass: collapse near-duplicate restatements and re-bound the store. + """One no-LLM upkeep pass: expire per-kind TTLs, collapse near-duplicate restatements, and + re-bound the store. This is the sleep-time idea (reorganise memory while the agent is idle) at Midas prices: extractive consolidation + value-ranked forgetting, $0, nothing leaves the box — versus LLM-agent rewrites of memory in systems like Letta's sleep-time agents.""" + expired = _mem.forget_expired(_TTL) if _TTL else [] consolidated = _mem.consolidate(similarity_threshold=0.95) forgotten = _mem.forget_decayed(max_records=_MAX_RECORDS) if _MAX_RECORDS else [] - return {"consolidated": len(consolidated), "forgotten": len(forgotten)} + return {"expired": len(expired), "consolidated": len(consolidated), "forgotten": len(forgotten)} def _auto_maintain_loop(interval_seconds: float) -> None: @@ -790,7 +916,7 @@ def _auto_maintain_loop(interval_seconds: float) -> None: _time.sleep(interval_seconds) try: result = _run_maintenance_pass() - if result["consolidated"] or result["forgotten"]: + if any(result.values()): print(f"[midas-mcp] auto-maintain: {result}", file=sys.stderr) except Exception as exc: # never let upkeep kill the server print(f"[midas-mcp] auto-maintain error: {exc}", file=sys.stderr) diff --git a/midas/memory.py b/midas/memory.py index 52b3f0b..ac9e3e3 100755 --- a/midas/memory.py +++ b/midas/memory.py @@ -1337,6 +1337,35 @@ def protected(r: MemoryRecord) -> bool: return [rid for rid in to_drop if self.store.delete(rid)] + def forget_expired( + self, + ttl_by_kind: dict[str, float], + *, + now: float | None = None, + keep_chains: bool = True, + ) -> list[str]: + """Retention policy: delete records whose kind has outlived its TTL (`ttl_by_kind` maps kind -> + max age in DAYS, e.g. {"chat": 30, "note": 90}). This is age-based retention — the compliance + dial ("we keep chat for 30 days") — distinct from `forget_decayed`, which is value-based bounding. + + Protections mirror the forgetting rules: supersession-chain members are kept (`keep_chains` — + belief history is audit material), and user-confirmed or standing/pinned records never expire + silently (an explicit `forget` outranks retention; a TTL should not). Returns the deleted ids, + oldest first — the retention audit trail. No LLM.""" + now = now if now is not None else self._now() + records = self.store.all() + pointed_to = {r.superseded_by for r in records if r.superseded_by is not None} + expired = [ + r for r in records + if r.kind in ttl_by_kind + and (now - r.created_at) > ttl_by_kind[r.kind] * 86_400.0 + and r.provenance != "user_confirmation" + and not r.metadata.get("standing") + and not (keep_chains and (r.superseded_by is not None or r.id in pointed_to)) + ] + expired.sort(key=lambda r: r.created_at) + return [r.id for r in expired if self.store.delete(r.id)] + def forget(self, record_id: str) -> bool: """Delete one memory by id, repairing any supersession chain that runs through it. diff --git a/midas/policy.py b/midas/policy.py index b482b48..2672317 100644 --- a/midas/policy.py +++ b/midas/policy.py @@ -43,8 +43,9 @@ class MemoryPolicy: # short — it is surfaced on every connection, so it states the loop and trusts Midas to enforce the rest. AGENT_MEMORY_INSTRUCTIONS = ( "Use Midas memory on every task. It is local, source-traceable, and uses no LLM at ingest/query.\n\n" - "1) RECALL FIRST. Call `build_context` with the user's goal; use the returned facts silently. Use " - "`recall`/`inspect_memory` only when you need audit details.\n\n" + "1) RECALL FIRST. At the START of a session call `resume` (pinned rules, current state, what " + "changed, open loops — one block); then call `build_context` with the user's goal and use the " + "returned facts silently. Use `recall`/`inspect_memory` only when you need audit details.\n\n" "2) CAPTURE DURABLE SIGNAL — DISTILLED. Call `capture` for reusable facts, decisions, preferences, " "constraints, corrections, and completed actions. Prefer ONE compact, self-contained statement " "(the entities, the value, and when) over raw turns — a memory that answers on its own retrieves " @@ -55,6 +56,8 @@ class MemoryPolicy: "call `check_memory_use`. If it is not allowed, ask the user to confirm in this turn.\n\n" "4) FORGET ON REQUEST. Use `forget_matching` as a dry-run first, show matches, then repeat with " "dry_run=false after confirmation.\n\n" + "4b) KEEP PROMISES. Record follow-up work you commit to with `remember_commitment`; close it via " + "`close_loop` when it lands. Check `open_loops` when resuming so promised work isn't dropped.\n\n" "5) CODE WORK. For software projects, capture decisions/bugs/conventions/forbidden-rules with " "`remember_code` (set `code_kind` and `project`); onboard with `project_state`; and before any code " "action memory suggests, call `check_forbidden_action`: refuse and cite the rule if `forbidden`, and " @@ -64,6 +67,20 @@ class MemoryPolicy: ) +def parse_ttl_spec(spec: str) -> dict[str, float]: + """Parse a retention spec like "chat=30,note=90" (kind -> days) — the MIDAS_MCP_TTL format. + Malformed entries are skipped rather than crashing the server; empty spec -> {}.""" + ttl: dict[str, float] = {} + for part in (spec or "").split(","): + kind, _, days = part.partition("=") + try: + if kind.strip() and float(days) > 0: + ttl[kind.strip()] = float(days) + except ValueError: + continue + return ttl + + def policy_summary(policy: MemoryPolicy) -> str: """One-line, human/agent-readable summary of the enforced parameters (for prompts / `stats`).""" return ( diff --git a/midas/types.py b/midas/types.py index d8e3029..f670f94 100755 --- a/midas/types.py +++ b/midas/types.py @@ -10,7 +10,7 @@ from dataclasses import dataclass, field from typing import Any, Literal -MemoryKind = Literal["note", "chat", "mission", "fact", "preference", "constraint"] +MemoryKind = Literal["note", "chat", "mission", "fact", "preference", "constraint", "commitment"] MEMORY_KINDS: tuple[MemoryKind, ...] = ( "note", "chat", @@ -18,6 +18,7 @@ "fact", "preference", "constraint", + "commitment", # an open loop: work someone said WILL be done (close it via continuity.close_loop) ) MemoryProvenance = Literal["planning", "action", "observation", "user_confirmation"] diff --git a/tests/test_control_plane.py b/tests/test_control_plane.py new file mode 100644 index 0000000..a7ba680 --- /dev/null +++ b/tests/test_control_plane.py @@ -0,0 +1,198 @@ +"""The continuity control-plane: memory_conflicts (live contradictions), open loops (commitments), +resume (the one-call session-onboarding pack), and per-kind TTL retention. All no-LLM, deterministic +with the offline hashing embedder.""" +from __future__ import annotations + +import time + +from midas import HashingEmbedder, Memory, parse_ttl_spec +from midas.continuity import close_loop, memory_conflicts, open_loops, remember_commitment, resume + + +def _mem() -> Memory: + return Memory(embedder=HashingEmbedder()) # supersession off → conflicting writes both stay live + + +# ---- memory_conflicts --------------------------------------------------------------------------- + +def test_conflicts_value_swap_across_actors() -> None: + mem = _mem() + mem.remember("The primary database is PostgreSQL.", kind="constraint", actor="claude-code") + mem.remember("The primary database is MySQL.", kind="constraint", actor="cursor") + found = memory_conflicts(mem) + assert len(found) == 1 + c = found[0] + assert c.signal == "value" + assert {c.a.actor, c.b.actor} == {"claude-code", "cursor"} # the multi-agent case + + +def test_conflicts_numeric_mismatch() -> None: + mem = _mem() + mem.remember("The launch date is September 14.", kind="fact") + mem.remember("The launch date is September 20.", kind="fact") + found = memory_conflicts(mem) + assert len(found) == 1 and found[0].signal == "numeric" + + +def test_conflicts_negation_mismatch() -> None: + mem = _mem() + mem.remember("We support Internet Explorer in the dashboard.", kind="constraint") + mem.remember("We no longer support Internet Explorer in the dashboard.", kind="constraint") + found = memory_conflicts(mem) + assert len(found) == 1 and found[0].signal == "negation" + + +def test_conflicts_ignore_agreement_and_unrelated() -> None: + mem = _mem() + mem.remember("The primary database is PostgreSQL.", kind="constraint") + mem.remember("The primary database is PostgreSQL.", kind="constraint") # restatement, agrees + mem.remember("Standup happens on Tuesday mornings.", kind="fact") # unrelated + assert memory_conflicts(mem) == [] + + +def test_conflicts_different_subjects_are_not_conflicts() -> None: + mem = _mem() + mem.remember("Project Apollo launches on May 3.", kind="fact") + mem.remember("Project Artemis launches on June 9.", kind="fact") + # different named subjects → different facts, even though the frames and numbers differ + assert all(c.signal != "numeric" for c in memory_conflicts(mem)) + + +def test_conflicts_skip_superseded() -> None: + mem = _mem() + old = mem.remember("The launch date is September 14.", kind="fact") + new = mem.remember("The launch date is September 20.", kind="fact") + assert len(memory_conflicts(mem)) == 1 # both live → a real conflict + rec = mem.store.get(old.id) + rec.superseded_by = new.id # belief revision resolves the disagreement… + mem.store.put(rec) + assert memory_conflicts(mem) == [] # …so there is nothing left to surface + + +def test_conflicts_scope_filter() -> None: + mem = _mem() + mem.remember("The API rate limit is 100 requests.", kind="fact", metadata={"namespace": "a"}) + mem.remember("The API rate limit is 500 requests.", kind="fact", metadata={"namespace": "b"}) + assert memory_conflicts(mem, scope={"namespace": "a"}) == [] # scoped views don't cross-contaminate + assert len(memory_conflicts(mem)) == 1 # the unscoped view still sees it + + +# ---- open loops --------------------------------------------------------------------------------- + +def test_open_loops_lifecycle() -> None: + mem = _mem() + loop = remember_commitment(mem, "Migrate the sessions table to UUID keys.", project="apollo") + assert loop.kind == "commitment" and loop.importance == 4 + assert [r.id for r in open_loops(mem)] == [loop.id] + assert open_loops(mem, scope={"project": "apollo"})[0].id == loop.id + + done = close_loop(mem, loop.id, "migrated in PR #42") + assert open_loops(mem) == [] # closed → no longer open + assert mem.store.get(loop.id).superseded_by == done.id # …but the history chain remains + assert done.metadata["closes"] == loop.id and done.provenance == "action" + + +def test_close_loop_rejects_non_commitments() -> None: + mem = _mem() + fact = mem.remember("The sky is blue.", kind="fact") + try: + close_loop(mem, fact.id, "nope") + raise AssertionError("close_loop must reject non-commitment ids") + except ValueError: + pass + + +def test_open_loops_oldest_first() -> None: + mem = _mem() + old = remember_commitment(mem, "Write the migration runbook.") + old_rec = mem.store.get(old.id) + old_rec.created_at -= 3600 + mem.store.put(old_rec) + remember_commitment(mem, "Enable the new CI matrix.") + assert open_loops(mem)[0].id == old.id # most overdue first + + +# ---- resume ------------------------------------------------------------------------------------- + +def test_resume_bundles_all_sections() -> None: + mem = _mem() + mem.remember("From now on, always reply in Spanish.", kind="mission", importance=5) + from midas.coding import remember_forbidden_action + remember_forbidden_action(mem, "Never run destructive migrations on prod.", project="apollo") + mem.remember("The primary database is PostgreSQL.", kind="constraint", + metadata={"project": "apollo"}) + remember_commitment(mem, "Add the audit log table.", project="apollo") + + pack = resume(mem, project="apollo") + ctx = pack["context"] + assert "FORBIDDEN:" in ctx and "destructive migrations" in ctx + assert "CURRENT STATE:" in ctx and "PostgreSQL" in ctx + assert "OPEN LOOPS:" in ctx and "audit log table" in ctx + assert "CHANGED (added):" in ctx # everything above was added inside the window + assert pack["truncated"] is False + # the pinned mission has no project tag, so the project scope excludes it — the global view has it + assert "reply in Spanish" in resume(mem)["context"] + + +def test_resume_respects_token_budget() -> None: + mem = _mem() + for i in range(60): + mem.remember(f"Architecture decision number {i} about service {i} routing.", + kind="constraint", importance=4) + pack = resume(mem, token_budget=80) + from midas import approx_tokens + assert pack["truncated"] is True + assert approx_tokens(pack["context"]) <= 80 * 1.2 # budget respected (approx by construction) + + +def test_resume_diff_window() -> None: + mem = _mem() + old = mem.remember("The old constraint from last month.", kind="constraint") + rec = mem.store.get(old.id) + rec.created_at -= 30 * 86_400 + mem.store.put(rec) + mem.remember("The new constraint from this week.", kind="constraint") + pack = resume(mem) # default window: 7 days + added_ids = {r.id for r in pack["added"]} + assert old.id not in added_ids # old change is outside the diff window… + assert any("old constraint" in r.content for r in pack["state"]) # …but still in current state + + +# ---- per-kind TTL retention ---------------------------------------------------------------------- + +def test_parse_ttl_spec() -> None: + assert parse_ttl_spec("chat=30,note=90") == {"chat": 30.0, "note": 90.0} + assert parse_ttl_spec(" chat = 7 ") == {"chat": 7.0} + assert parse_ttl_spec("chat=oops,=5,note=-1,") == {} # malformed entries are skipped, not fatal + assert parse_ttl_spec("") == {} + + +def test_forget_expired_by_kind() -> None: + mem = _mem() + now = time.time() + stale_chat = mem.remember("just chatting about the weather", kind="chat", + created_at=now - 40 * 86_400) + fresh_chat = mem.remember("more chat from yesterday", kind="chat", created_at=now - 86_400) + old_fact = mem.remember("The launch date is September 14.", kind="fact", + created_at=now - 400 * 86_400) + dropped = mem.forget_expired({"chat": 30}, now=now) + assert dropped == [stale_chat.id] + remaining = {r.id for r in mem.store.all()} + assert fresh_chat.id in remaining and old_fact.id in remaining # other kinds untouched + + +def test_forget_expired_protections() -> None: + mem = _mem() + now = time.time() + confirmed = mem.remember("User said: always use tabs.", kind="chat", + provenance="user_confirmation", created_at=now - 100 * 86_400) + old = mem.remember("The launch date is September 14.", kind="fact", + created_at=now - 100 * 86_400) + new = mem.remember("The launch date is September 20.", kind="fact", + created_at=now - 90 * 86_400) + rec = mem.store.get(old.id) + rec.superseded_by = new.id # a revision chain: old -> new + mem.store.put(rec) + dropped = mem.forget_expired({"chat": 30, "fact": 30}, now=now) + assert dropped == [] # user-confirmed never expires silently; chain members are audit material + assert {confirmed.id, old.id, new.id} <= {r.id for r in mem.store.all()} diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py index d2c6ef9..0988316 100755 --- a/tests/test_mcp_server.py +++ b/tests/test_mcp_server.py @@ -369,3 +369,60 @@ async def send(ev): assert call("http", "Bearer wrong")[0]["status"] == 401 assert call("http", None)[0]["status"] == 401 assert call("lifespan", None)[0]["status"] == 200 # non-http scopes pass straight through + + +def test_resume_and_loops_tools(): + from midas.mcp_server import close_loop, open_loops, remember_commitment, resume + + forget_all() + remember("The primary database is PostgreSQL.", kind="constraint", importance=5) + out = remember_commitment("Migrate the sessions table to UUID keys.", project="apollo") + assert "commitment recorded" in out + loop_id = out.split("(")[1].split(")")[0] + + loops = open_loops(project="apollo") + assert loops["count"] == 1 and loops["open_loops"][0]["id"] == loop_id + + pack = resume() + assert pack["counts"]["open_loops"] == 1 + assert "OPEN LOOPS:" in pack["context"] and "PostgreSQL" in pack["context"] + + assert "loop closed" in close_loop(loop_id, "migrated in PR #42") + assert open_loops()["count"] == 0 + assert close_loop("nope-id", "x").startswith("no open commitment") + forget_all() # the module-level store is shared across test files — leave it clean + + +def test_memory_conflicts_tool(): + from midas.mcp_server import memory_conflicts + + forget_all() + # The provenance-integrity case: the user CONFIRMED PostgreSQL, then another agent wrote MySQL as + # an observation. Belief revision correctly refuses to launder the confirmed belief away — so both + # stay live and disagree. That unresolved disagreement is exactly what this tool surfaces. + remember("The primary database is PostgreSQL.", kind="constraint", + provenance="user_confirmation", actor="claude-code") + remember("The primary database is MySQL.", kind="constraint", actor="cursor") + out = memory_conflicts() + assert out["count"] == 1 and out["conflicts"][0]["signal"] == "value" + actors = {out["conflicts"][0]["a"]["actor"], out["conflicts"][0]["b"]["actor"]} + assert actors == {"claude-code", "cursor"} + forget_all() # the module-level store is shared across test files — leave it clean + + +def test_maintain_ttl(): + import time as _t + + from midas.mcp_server import _mem + + forget_all() + remember("old chatter about the weather", kind="chat") + rec = _mem.store.all()[0] + rec.created_at = _t.time() - 40 * 86_400 + _mem.store.put(rec) + remember("The primary database is PostgreSQL.", kind="constraint", importance=5) + + out = maintain(ttl="chat=30") + assert out["expired"] == 1 and out["remaining"] == 1 + assert stats()["by_kind"] == {"constraint": 1} + forget_all() # the module-level store is shared across test files — leave it clean From 4eccaa284a3ab90981926c20dba2b3969dcaf351 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 02:47:10 +0000 Subject: [PATCH 03/10] Trust: tamper-evident hash-chained audit log + `midas audit` MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every SQLite mutation (put / delete / clear — so remember, supersede, and forget) appends an entry to an append-only audit_log table: seq, timestamp, op, record id, a sha256 of the record's audited state, and a hash chained over the previous entry. Editing, removing, or reordering ANY past entry breaks every hash after it. Entries carry hashes only — never memory content, preserving the receipt/audit boundary. - SQLiteStore.audit_log() / verify_audit_log(); audit=False opt-out for perf-sensitive paths. Additive table: older Midas versions still open the file. - `midas audit [--limit N] [--json]` shows the tail and verifies the chain (exit 1 + first_invalid_seq when broken). - `midas doctor` now checks chain integrity. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- README.md | 4 ++ midas/cli.py | 59 ++++++++++++++++++++ midas/sqlite_store.py | 95 +++++++++++++++++++++++++++++++- tests/test_audit_chain.py | 111 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 268 insertions(+), 1 deletion(-) create mode 100644 tests/test_audit_chain.py diff --git a/README.md b/README.md index 34c4a57..e7e174d 100755 --- a/README.md +++ b/README.md @@ -275,6 +275,10 @@ midas inspect --db ~/.midas/memory.sqlite3 # opens http://localhost:7777 - **Project state** (decisions / bugs / forbidden) and **what changed** since a date. - **Governance** — would memory authorize an action, and why (the audit trail); **forget** with a receipt. +And every mutation (write / revise / forget) appends to a **tamper-evident, hash-chained audit log** +inside the store — hashes only, never content. `midas audit` shows it; `midas audit --json` verifies +the whole chain and reports the first broken entry if anyone rewrote history. + No LLM, no account, runs on your file. The thing a black-box memory can't show. ## Commercial path diff --git a/midas/cli.py b/midas/cli.py index 4637be2..79ef67e 100644 --- a/midas/cli.py +++ b/midas/cli.py @@ -431,6 +431,9 @@ def check(good: bool, label: str, hint: str = "") -> None: try: store = SQLiteStore(db) check(True, f"store: {db} ({len(store.all())} records, schema v{store.schema_version()})") + chain = store.verify_audit_log() + check(chain["ok"], f"audit chain intact ({chain['entries']} entries)", + f"broken at seq {chain['first_invalid_seq']} — see `midas audit`") except Exception as exc: check(False, f"store: {db}", f"cannot open: {exc}") else: @@ -477,6 +480,55 @@ def check(good: bool, label: str, hint: str = "") -> None: return 0 if ok else 1 +# ---- audit ------------------------------------------------------------------------------------ + +def cmd_audit(args: argparse.Namespace) -> int: + """Show / verify the tamper-evident mutation log: every put/delete/clear, hash-chained. Editing, + removing, or reordering any past entry breaks every hash after it — `--verify` (the default check) + walks the whole chain. Entries carry hashes only, never memory content.""" + from midas.sqlite_store import SQLiteStore + + db = _store_path(args.db) + if not Path(db).exists(): + print(f"no store at {db} — run `midas init`", file=sys.stderr) + return 1 + store = SQLiteStore(db) + result = store.verify_audit_log() + entries = store.audit_log(limit=int(args.limit) if args.limit else 10) + + if getattr(args, "json", False): + from datetime import datetime, timezone + + from midas import __version__ + + print(json.dumps({ + "midas_version": __version__, + "receipt_kind": "audit_chain", + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds") + .replace("+00:00", "Z"), + "memory_db": db, + "ok": result["ok"], + "entries": result["entries"], + "first_invalid_seq": result["first_invalid_seq"], + "tail": entries, + }, indent=2)) + return 0 if result["ok"] else 1 + + state = "intact" if result["ok"] else f"BROKEN at seq {result['first_invalid_seq']}" + print(f"audit chain: {state} · {result['entries']} entries · {db}") + if entries: + from datetime import datetime + + print(f"\nlast {len(entries)}:") + for e in entries: + when = datetime.fromtimestamp(e["at"]).strftime("%Y-%m-%d %H:%M:%S") + print(f" #{e['seq']:<6} {when} {e['op']:<7} {e['record_id'][:8]:<9} " + f"sha:{e['content_sha'][:12]}…") + if not result["ok"]: + print("\n⚠ the log was modified after the fact — treat this store's history as untrusted.") + return 0 if result["ok"] else 1 + + # ---- export / import -------------------------------------------------------------------------- def _rec_to_dict(r) -> dict: @@ -689,6 +741,13 @@ def main() -> None: help="print a machine-readable diagnosis instead of text") pd.set_defaults(func=cmd_doctor) + pa = sub.add_parser("audit", help="show/verify the tamper-evident mutation log (hash chain)") + pa.add_argument("--db") + pa.add_argument("--limit", type=int, default=10, help="how many recent entries to show") + pa.add_argument("--json", action="store_true", + help="print a machine-readable audit receipt instead of text") + pa.set_defaults(func=cmd_audit) + pe = sub.add_parser("export", help="export all memory to JSON (backup / move machines)") pe.add_argument("--db") pe.add_argument("-o", "--out", help="output file (default: stdout)") diff --git a/midas/sqlite_store.py b/midas/sqlite_store.py index f419f42..ce9611f 100755 --- a/midas/sqlite_store.py +++ b/midas/sqlite_store.py @@ -23,10 +23,12 @@ """ from __future__ import annotations +import hashlib import json import sqlite3 import struct import threading +import time from pathlib import Path from typing import Callable @@ -35,14 +37,33 @@ SCHEMA_VERSION = 1 # bump when the `memories` schema changes; add a guarded migration step in _migrate() +# The audit chain's genesis value — the prev_hash of the first entry. +_AUDIT_GENESIS = "0" * 64 + + +def _audit_hash(seq: int, at: float, op: str, record_id: str, content_sha: str, + prev_hash: str) -> str: + return hashlib.sha256( + f"{seq}|{at:.6f}|{op}|{record_id}|{content_sha}|{prev_hash}".encode() + ).hexdigest() + + +def _content_sha(record: MemoryRecord) -> str: + """Fingerprint of the record's audited state: id + content + revision link. Enough to later prove + what a mutation touched; NOT enough to reconstruct the content (the audit log stores no text).""" + payload = f"{record.id}\x00{record.content}\x00{record.superseded_by or ''}" + return hashlib.sha256(payload.encode()).hexdigest() + class SQLiteStore(InMemoryStore): """In-memory store (fast vectorised search) mirrored to a SQLite file for persistence.""" def __init__( - self, path: str | Path, *, ann_threshold: int | None = None, ann_nprobe: int = 16 + self, path: str | Path, *, ann_threshold: int | None = None, ann_nprobe: int = 16, + audit: bool = True, ) -> None: super().__init__(ann_threshold=ann_threshold, ann_nprobe=ann_nprobe) + self._audit_enabled = audit self._path = Path(path) if self._path.parent and str(self._path.parent): self._path.parent.mkdir(parents=True, exist_ok=True) @@ -69,6 +90,21 @@ def __init__( ) """ ) + # Tamper-evident mutation log (append-only, hash-chained; additive table so older Midas + # versions can still open the file). Rows carry hashes, never memory content. + self._conn.execute( + """ + CREATE TABLE IF NOT EXISTS audit_log ( + seq INTEGER PRIMARY KEY, + at REAL NOT NULL, + op TEXT NOT NULL, + record_id TEXT NOT NULL, + content_sha TEXT NOT NULL, + prev_hash TEXT NOT NULL, + hash TEXT NOT NULL + ) + """ + ) self._migrate() self._conn.commit() self._load() @@ -146,6 +182,55 @@ def _row_to_record(row) -> MemoryRecord: superseded_by=superseded_by, embedding=embedding, ) + # ---- tamper-evident audit chain ------------------------------------------------------------- + + def _audit(self, op: str, record_id: str, content_sha: str) -> None: + """Append one hash-chained entry (caller holds the lock; committed with the mutation). Each + hash covers the previous entry's hash, so editing or deleting ANY past row breaks every hash + after it — `verify_audit_log` catches it. Multi-process safe: seq/prev are read and written + inside the same transaction as the mutation.""" + row = self._conn.execute( + "SELECT seq, hash FROM audit_log ORDER BY seq DESC LIMIT 1" + ).fetchone() + prev_seq, prev_hash = (int(row[0]), str(row[1])) if row else (0, _AUDIT_GENESIS) + seq, at = prev_seq + 1, time.time() + self._conn.execute( + "INSERT INTO audit_log (seq, at, op, record_id, content_sha, prev_hash, hash) " + "VALUES (?, ?, ?, ?, ?, ?, ?)", + (seq, at, op, record_id, content_sha, + prev_hash, _audit_hash(seq, at, op, record_id, content_sha, prev_hash)), + ) + + def audit_log(self, *, limit: int | None = None) -> list[dict]: + """The mutation history (append-only): every put/delete/clear with its chain hashes — the + compliance artifact behind "prove what happened to memory". No memory content, only hashes.""" + with self._lock: + sql = "SELECT seq, at, op, record_id, content_sha, prev_hash, hash FROM audit_log " \ + "ORDER BY seq" + rows = self._conn.execute(sql).fetchall() + entries = [dict(zip(("seq", "at", "op", "record_id", "content_sha", "prev_hash", "hash"), r)) + for r in rows] + return entries[-limit:] if limit else entries + + def verify_audit_log(self) -> dict: + """Walk the whole chain recomputing every hash. Returns {ok, entries, first_invalid_seq}: + any edited, removed, or reordered entry breaks the chain at the first affected seq.""" + entries = self.audit_log() + prev_hash, expected_seq = _AUDIT_GENESIS, 1 + for e in entries: + good = ( + e["seq"] == expected_seq + and e["prev_hash"] == prev_hash + and e["hash"] == _audit_hash(e["seq"], e["at"], e["op"], e["record_id"], + e["content_sha"], e["prev_hash"]) + ) + if not good: + return {"ok": False, "entries": len(entries), "first_invalid_seq": e["seq"]} + prev_hash, expected_seq = e["hash"], expected_seq + 1 + return {"ok": True, "entries": len(entries), "first_invalid_seq": None} + + # ---- mutations (each writes its audit entry in the same transaction) ------------------------- + def put(self, record: MemoryRecord) -> None: with self._lock: self._refresh_if_stale() @@ -172,6 +257,8 @@ def put(self, record: MemoryRecord) -> None: record.provenance, record.actor, json.dumps(record.metadata or {}), record.created_at, record.updated_at, record.superseded_by, emb_blob), ) + if self._audit_enabled: + self._audit("put", record.id, _content_sha(record)) self._conn.commit() def get(self, record_id: str) -> MemoryRecord | None: @@ -198,9 +285,13 @@ def search( def delete(self, record_id: str) -> bool: with self._lock: self._refresh_if_stale() + target = super().get(record_id) existed = super().delete(record_id) if existed: self._conn.execute("DELETE FROM memories WHERE id = ?", (record_id,)) + if self._audit_enabled: + self._audit("delete", record_id, + _content_sha(target) if target else _AUDIT_GENESIS) self._conn.commit() return existed @@ -208,6 +299,8 @@ def clear(self) -> None: with self._lock: super().clear() self._conn.execute("DELETE FROM memories") + if self._audit_enabled: + self._audit("clear", "*", _AUDIT_GENESIS) self._conn.commit() def close(self) -> None: diff --git a/tests/test_audit_chain.py b/tests/test_audit_chain.py new file mode 100644 index 0000000..407e0c5 --- /dev/null +++ b/tests/test_audit_chain.py @@ -0,0 +1,111 @@ +"""The tamper-evident audit chain: every SQLite mutation appends a hash-chained entry (no memory +content, only hashes), and `verify_audit_log` / `midas audit` detect any after-the-fact edit.""" +from __future__ import annotations + +import argparse +import json +import sqlite3 + +from midas import HashingEmbedder, Memory +from midas.sqlite_store import SQLiteStore + + +def _mem(path: str) -> Memory: + return Memory(store=SQLiteStore(path), embedder=HashingEmbedder()) + + +def test_mutations_append_chained_entries(tmp_path) -> None: + mem = _mem(str(tmp_path / "m.sqlite3")) + a = mem.remember("The primary database is PostgreSQL.", kind="constraint") + mem.remember("The launch date is September 14.", kind="fact") + mem.forget(a.id) + + log = mem.store.audit_log() + assert [e["op"] for e in log] == ["put", "put", "delete"] + assert log[0]["record_id"] == a.id and log[2]["record_id"] == a.id + # the chain links: each entry's prev_hash is the previous entry's hash + assert log[1]["prev_hash"] == log[0]["hash"] and log[2]["prev_hash"] == log[1]["hash"] + # no memory content anywhere in the log (the audit boundary) + assert "PostgreSQL" not in json.dumps(log) + + assert mem.store.verify_audit_log() == {"ok": True, "entries": 3, "first_invalid_seq": None} + + +def test_supersession_and_clear_are_audited(tmp_path) -> None: + db = str(tmp_path / "m.sqlite3") + mem = Memory(store=SQLiteStore(db), embedder=HashingEmbedder(), supersede=True) + mem.remember("The launch date is September 14.", kind="fact") + mem.remember("The launch date is September 20.", kind="fact") + ops = [e["op"] for e in mem.store.audit_log()] + assert ops.count("put") >= 3 # 2 inserts + the revised head re-written with its supersession link + mem.store.clear() + assert mem.store.audit_log()[-1]["op"] == "clear" + assert mem.store.verify_audit_log()["ok"] + + +def test_tampering_breaks_the_chain(tmp_path) -> None: + db = str(tmp_path / "m.sqlite3") + mem = _mem(db) + for i in range(4): + mem.remember(f"note number {i} about service routing", kind="note") + mem.store.close() + + con = sqlite3.connect(db) # an attacker rewrites history directly + con.execute("UPDATE audit_log SET content_sha = 'f' || substr(content_sha, 2) WHERE seq = 2") + con.commit() + con.close() + + result = SQLiteStore(db).verify_audit_log() + assert result["ok"] is False and result["first_invalid_seq"] == 2 + + +def test_deleting_an_entry_breaks_the_chain(tmp_path) -> None: + db = str(tmp_path / "m.sqlite3") + mem = _mem(db) + for i in range(3): + mem.remember(f"note number {i} about service routing", kind="note") + mem.store.close() + + con = sqlite3.connect(db) # covering tracks by removing the middle entry + con.execute("DELETE FROM audit_log WHERE seq = 2") + con.commit() + con.close() + + result = SQLiteStore(db).verify_audit_log() + assert result["ok"] is False and result["first_invalid_seq"] == 3 # the gap is where it breaks + + +def test_chain_survives_reopen_and_cross_instance(tmp_path) -> None: + db = str(tmp_path / "m.sqlite3") + _mem(db).remember("first fact about the deploy pipeline", kind="fact") + _mem(db).remember("second fact about the deploy pipeline", kind="fact") # a separate connection + store = SQLiteStore(db) + assert store.verify_audit_log() == {"ok": True, "entries": 2, "first_invalid_seq": None} + assert [e["seq"] for e in store.audit_log()] == [1, 2] + + +def test_audit_can_be_disabled(tmp_path) -> None: + store = SQLiteStore(str(tmp_path / "m.sqlite3"), audit=False) + Memory(store=store, embedder=HashingEmbedder()).remember("perf-path write", kind="note") + assert store.audit_log() == [] + + +def test_cli_audit_command(tmp_path, capsys) -> None: + from midas import cli + + db = str(tmp_path / "m.sqlite3") + assert cli.cmd_audit(argparse.Namespace(db=db, limit=10, json=False)) == 1 # no store yet + capsys.readouterr() + + _mem(db).remember("The primary database is PostgreSQL.", kind="constraint") + assert cli.cmd_audit(argparse.Namespace(db=db, limit=10, json=True)) == 0 + out = json.loads(capsys.readouterr().out) + assert out["receipt_kind"] == "audit_chain" and out["ok"] is True and out["entries"] == 1 + assert "PostgreSQL" not in json.dumps(out) # receipts never leak memory content + + con = sqlite3.connect(db) + con.execute("UPDATE audit_log SET op = 'delete' WHERE seq = 1") + con.commit() + con.close() + assert cli.cmd_audit(argparse.Namespace(db=db, limit=10, json=False)) == 1 + assert "BROKEN" in capsys.readouterr().out From efcb719e6ac5c91ea3c422014c4f08e2c4683e21 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 02:57:01 +0000 Subject: [PATCH 04/10] Adoption: `midas import --from claude-md/cursorrules/jsonl` + TS semantic embeddings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - midas/importers.py + `midas import --from …`: file-based agent memory (CLAUDE.md, .cursorrules, exported JSONL) becomes first-class Midas records — bullets/paragraphs parsed per item with heading context, fenced code skipped, idempotent re-runs, and governance-safe defaults (observation provenance; imported rules can't authorize guarded actions unless --confirmed). - TypeScript port: LocalEmbedder — local ONNX semantic embeddings (bge-small, 384d) via the optional @huggingface/transformers behind MIDAS_MCP_EMBEDDER=local, with an announced fallback to the byte-parity hashing embedder when the package is absent. The TS Memory API is now async (remember/capture/recall/buildContext/forgetMatching), verified end-to-end (paraphrase recall 0.731 vs 0.491 noise). 14/14 TS tests pass. - CHANGELOG for the whole batch; README updates. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- CHANGELOG.md | 32 +++++++++ README.md | 6 +- midas/cli.py | 69 +++++++++++++++++- midas/importers.py | 102 +++++++++++++++++++++++++++ packages/midas-ts/README.md | 19 ++++- packages/midas-ts/src/embeddings.ts | 55 ++++++++++++++- packages/midas-ts/src/index.ts | 2 +- packages/midas-ts/src/mcp.ts | 44 +++++++++--- packages/midas-ts/src/memory.ts | 26 +++---- packages/midas-ts/test/core.test.mjs | 90 +++++++++++++---------- tests/test_importers.py | 102 +++++++++++++++++++++++++++ 11 files changed, 474 insertions(+), 73 deletions(-) create mode 100644 midas/importers.py create mode 100644 tests/test_importers.py diff --git a/CHANGELOG.md b/CHANGELOG.md index e075dd0..4e67651 100755 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,38 @@ Notable changes to Midas. Pre-1.0 — the API may change. Format loosely follows ## [Unreleased] +### Added +- **The continuity control-plane** (`midas/continuity.py` + MCP tools): **`resume`** — everything an + agent needs to pick up a session in ONE call (pinned directives, forbidden rules, current state, + what changed, open loops, unresolved conflicts; token-budgeted, prompt-ready); **`memory_conflicts`** + — live beliefs that contradict each other with neither superseding the other (the multi-agent + shared-memory failure mode), NLI-scored when available with a same-slot heuristic fallback, ranked + candidates only; **open loops** — `remember_commitment` / `open_loops` / `close_loop` make promised + work survive across sessions with an auditable promise→resolution chain. The injected agent policy + now starts sessions with `resume` and keeps promises via open loops. +- **Per-kind TTL retention**: `Memory.forget_expired({"chat": 30})`, `MIDAS_MCP_TTL="chat=30,note=90"`, + and `maintain(ttl=…)` — age-based retention next to value-based forgetting; user-confirmed, + standing, and supersession-chain records never expire silently. +- **Tamper-evident audit chain** in `SQLiteStore`: every put/delete/clear appends a hash-chained + entry (hashes only — never memory content); `midas audit [--json]` shows and verifies the chain and + `midas doctor` checks its integrity. `audit=False` opts out for perf-sensitive paths. +- **`midas import --from claude-md | cursorrules | jsonl`**: file-based agent memory (CLAUDE.md, + .cursorrules, exported JSONL) becomes first-class Midas records — sectioned, deduped, idempotent, + and governance-safe by default (imported rules are `observation` provenance and cannot authorize + guarded actions unless you pass `--confirmed`). +- **4 more clients in `midas init`**: VS Code (user `mcp.json`, `servers` key), Gemini CLI, Cline, + and Zed (`context_servers`) — each written in its own config schema; status/uninstall understand + them too. +- **HTTP bearer-token auth**: `midas serve --http --token ` / `MIDAS_MCP_TOKEN` gates every + request on the shared-URL transport (constant-time compare, 401 + `WWW-Authenticate` otherwise). +- **`midas doctor --json`** — the machine-readable diagnosis, same envelope as the wiring receipt. +- **TypeScript port: optional local semantic embeddings.** `LocalEmbedder` (ONNX via the optional + `@huggingface/transformers`, bge-small by default) behind `MIDAS_MCP_EMBEDDER=local`, with a clean + announced fallback to the byte-parity hashing embedder; the TS `Memory` API is now async + (`remember`/`capture`/`recall`/`buildContext`/`forgetMatching` return promises). +- **CI on Windows and macOS** (core + MCP): the client-wiring code is full of per-OS paths that only + Linux exercised. + ### Added - **`midas init --json` / `midas status --json` — a machine-readable client wiring receipt** (#15). One command now yields a compact, pasteable proof of what was wired: per client its config path, diff --git a/README.md b/README.md index e7e174d..4ccb953 100755 --- a/README.md +++ b/README.md @@ -133,6 +133,10 @@ midas serve --http --token # require `Authorization: Bearer ` Keep Midas current with **`midas update`**. See your memory anytime with **`midas inspect`**. +Already carrying agent memory in files? **`midas import --from claude-md CLAUDE.md`** (or +`--from cursorrules .cursorrules`, `--from jsonl`) turns those rules into first-class, recallable, +governable memories — tagged with where they came from, idempotent on re-run. +
Manual setup — any client, or to customize (click to expand) @@ -155,7 +159,7 @@ by default — no path needed. The universal block: | **Cline** | `cline_mcp_settings.json` in VS Code global storage — `midas init` writes it | | **Zed** | `settings.json` → `context_servers` — `midas init` writes it | | **Anything else** | point it at command `midas-mcp` | -| **No Python** | `npx -y midas-memory-mcp` — the [TypeScript port](packages/midas-ts) (experimental: no semantic embeddings yet) | +| **No Python** | `npx -y midas-memory-mcp` — the [TypeScript port](packages/midas-ts) (experimental; semantic embeddings via optional `@huggingface/transformers`) | Override per client with env: **`MIDAS_MCP_DB`** (default `~/.midas/memory.sqlite3`; `:memory:` = ephemeral) · `MIDAS_MCP_MAX_RECORDS` · `MIDAS_MCP_MIN_IMPORTANCE` · `MIDAS_MCP_NAMESPACE`. diff --git a/midas/cli.py b/midas/cli.py index 79ef67e..0f95bd8 100644 --- a/midas/cli.py +++ b/midas/cli.py @@ -559,7 +559,64 @@ def cmd_export(args: argparse.Namespace) -> int: return 0 +def _import_rules(args: argparse.Namespace, source: str) -> int: + """Import rules-style files (CLAUDE.md / .cursorrules / generic JSONL) as first-class memories — + the migration on-ramp from file-based agent memory. Idempotent: re-running skips records whose + content is already present.""" + from midas import Memory + from midas.importers import parse_markdown_rules, to_record_dicts + from midas.sqlite_store import SQLiteStore + + db = _store_path(args.db) + _ensure_store(db) + mem = Memory(store=SQLiteStore(db)) + path = Path(args.file).expanduser() + text = path.read_text() + + if source == "jsonl": + dicts = [] + for line in text.splitlines(): + if not line.strip(): + continue + d = json.loads(line) + dicts.append({ + "content": d["content"], "kind": d.get("kind", "note"), + "importance": int(d.get("importance", 3)), + "provenance": d.get("provenance", "observation"), + "source": f"import:{path.name}", + "metadata": {**(d.get("metadata") or {}), "imported_from": path.name}, + }) + else: + items = parse_markdown_rules(text) + dicts = to_record_dicts(items, source_file=path.name, + project=getattr(args, "project", None) or None, + confirmed=getattr(args, "confirmed", False)) + if not dicts: + print(f"nothing importable found in {path}", file=sys.stderr) + return 1 + + existing = {" ".join(r.content.split()).lower() for r in mem.store.all()} + added = skipped = 0 + for d in dicts: + key = " ".join(d["content"].split()).lower() + if key in existing: + skipped += 1 + continue + mem.remember(**d) + existing.add(key) + added += 1 + note = f" ({skipped} already present, skipped)" if skipped else "" + print(f"imported {added} rules from {path} into {db}{note}") + if source != "jsonl" and not getattr(args, "confirmed", False): + print(" provenance=observation — imported rules inform recall/planning but cannot authorize " + "guarded actions (re-run with --confirmed to vouch for them).") + return 0 + + def cmd_import(args: argparse.Namespace) -> int: + source = getattr(args, "source", "midas") + if source != "midas": + return _import_rules(args, source) from midas.sqlite_store import SQLiteStore from midas.types import MemoryRecord @@ -753,8 +810,16 @@ def main() -> None: pe.add_argument("-o", "--out", help="output file (default: stdout)") pe.set_defaults(func=cmd_export) - pm = sub.add_parser("import", help="import memory from a JSON file") - pm.add_argument("file", help="the .json produced by `midas export`") + pm = sub.add_parser("import", help="import memory (midas export JSON, CLAUDE.md, .cursorrules…)") + pm.add_argument("file", help="the file to import") + pm.add_argument("--from", dest="source", default="midas", + choices=("midas", "claude-md", "cursorrules", "jsonl"), + help="what the file is: a midas export (default), a rules markdown " + "(CLAUDE.md/AGENTS.md), a .cursorrules, or generic JSONL records") + pm.add_argument("--project", help="tag imported rules with this project scope") + pm.add_argument("--confirmed", action="store_true", + help="stamp imported rules as user-confirmed (they may then authorize " + "guarded actions — only if you vouch for the file)") pm.add_argument("--db") pm.add_argument("--overwrite", action="store_true", help="overwrite records with the same id") pm.set_defaults(func=cmd_import) diff --git a/midas/importers.py b/midas/importers.py new file mode 100644 index 0000000..03fb868 --- /dev/null +++ b/midas/importers.py @@ -0,0 +1,102 @@ +"""Import existing agent memory INTO Midas — the migration on-ramp. + +Most teams already carry agent memory in files: a `CLAUDE.md`, a `.cursorrules`, exported JSONL from +another memory product. Instead of retyping it, `midas import --from …` ingests those as first-class +Midas records (with `imported_from` provenance metadata), so the constraints an agent already obeys +become recallable, supersedable, and governable. + +Parsing is deliberately dumb and lossless-enough: bullets and standalone paragraphs become one record +each (a memory that answers on its own retrieves better than a whole file), fenced code blocks are +skipped, and headings become a "Section:" prefix so a rule keeps its context. No LLM. + +Governance note: imported rules default to provenance="observation". They inform planning and recall, +but they cannot authorize external/destructive actions through the guard — pass `--confirmed` only +when you vouch that the file states user-confirmed policy. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass + +_BULLET_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+(.*)$") +_HEADING_RE = re.compile(r"^(#{1,6})\s+(.*)$") +_FENCE_RE = re.compile(r"^\s*(```|~~~)") +_MD_INLINE_RE = re.compile(r"\*\*([^*]+)\*\*|\*([^*]+)\*|__([^_]+)__|`([^`]+)`") + + +@dataclass(frozen=True) +class ImportedItem: + content: str + section: str # the heading path the item sat under ("" at top level) + + +def _strip_inline_md(text: str) -> str: + return _MD_INLINE_RE.sub(lambda m: next(g for g in m.groups() if g is not None), text).strip() + + +def parse_markdown_rules(text: str) -> list[ImportedItem]: + """Extract one item per bullet / standalone paragraph from a rules-style markdown file + (CLAUDE.md, .cursorrules, AGENTS.md…). Fenced code blocks are skipped; headings become context.""" + items: list[ImportedItem] = [] + section = "" + in_fence = False + paragraph: list[str] = [] + + def flush_paragraph() -> None: + nonlocal paragraph + joined = _strip_inline_md(" ".join(paragraph)) + if len(joined) >= 20: # a one-word line is noise, not a rule + items.append(ImportedItem(content=joined, section=section)) + paragraph = [] + + for raw in text.splitlines(): + if _FENCE_RE.match(raw): + flush_paragraph() + in_fence = not in_fence + continue + if in_fence: + continue + heading = _HEADING_RE.match(raw) + if heading: + flush_paragraph() + section = heading.group(2).strip() + continue + bullet = _BULLET_RE.match(raw) + if bullet: + flush_paragraph() + content = _strip_inline_md(bullet.group(1)) + if len(content) >= 8: + items.append(ImportedItem(content=content, section=section)) + continue + if raw.strip(): + paragraph.append(raw.strip()) + else: + flush_paragraph() + flush_paragraph() + return items + + +def to_record_dicts( + items: list[ImportedItem], + *, + source_file: str, + project: str | None = None, + confirmed: bool = False, +) -> list[dict]: + """Turn parsed items into `Memory.remember(**kwargs)` dicts. Section context is prefixed into the + content (self-contained memories retrieve better), never invented.""" + out: list[dict] = [] + for item in items: + content = f"{item.section}: {item.content}" if item.section else item.content + meta: dict = {"imported_from": source_file} + if project: + meta["project"] = project + out.append({ + "content": content, + "kind": "constraint", + "importance": 4, + "provenance": "user_confirmation" if confirmed else "observation", + "source": f"import:{source_file}", + "metadata": meta, + }) + return out diff --git a/packages/midas-ts/README.md b/packages/midas-ts/README.md index c049821..8fd09e6 100644 --- a/packages/midas-ts/README.md +++ b/packages/midas-ts/README.md @@ -46,12 +46,25 @@ npx midas-memory-mcp # or: npm i -g midas-memory-mcp && midas-mcp belief revision with supersession chains, no-LLM importance scoring, selective forgetting with durable-tier protection. +## Semantic embeddings (optional, local ONNX) + +Install the optional model runtime and set the same env `midas init` writes for every other client: + +```bash +npm i @huggingface/transformers # one-time; models cache locally after first download +MIDAS_MCP_EMBEDDER=local midas-mcp # bge-small (384d) via LocalEmbedder; any model id works too +``` + +Without the package (or with `MIDAS_MCP_EMBEDDER=hashing`), the server uses the offline hashing +embedder and announces the fallback on stderr — recall still works, lexically. **Keep one embedder +per store file**: vectors are model-specific, so don't point a bge-small process at a store written +with the hashing embedder (or with Python's bge-base). + ## Not ported (yet) -- **Semantic ONNX embeddings** (bge / multilingual) — the default embedder here is the offline - hashing one (lexical-ish). For real semantic recall today, run the Python server; both can share - the same DB. - Local NLI contradiction gating, the cross-encoder reranker, and the eval harness. +- The continuity control-plane (`resume`, `memory_conflicts`, open loops) and the hash-chained + audit log — Python-only for now. ## Requirements diff --git a/packages/midas-ts/src/embeddings.ts b/packages/midas-ts/src/embeddings.ts index ed5057e..3e06400 100644 --- a/packages/midas-ts/src/embeddings.ts +++ b/packages/midas-ts/src/embeddings.ts @@ -1,7 +1,8 @@ /** Embeddings — the offline `HashingEmbedder` is a byte-for-byte port of the Python one * (same md5 token hashing, same sign/index math), so a TypeScript process produces the SAME * vectors as a Python process and both can share one on-disk store semantically. - * A local ONNX semantic embedder (bge / multilingual) is the planned next step. */ + * `LocalEmbedder` adds local ONNX semantic embeddings via @huggingface/transformers (optional — + * install it separately; everything degrades to the hashing embedder without it). */ import { createHash } from "node:crypto"; const WORD = /[\p{L}\p{N}_]+/gu; @@ -16,8 +17,9 @@ export function tokenize(text: string): string[] { export interface Embedder { dim: number; - embed(text: string): Float32Array; - embedMany(texts: string[]): Float32Array[]; + /** May be sync (HashingEmbedder) or async (LocalEmbedder) — callers `await` either. */ + embed(text: string): Float32Array | Promise; + embedMany(texts: string[]): Float32Array[] | Promise; } export function l2Normalize(vec: Float32Array): Float32Array { @@ -66,3 +68,50 @@ export class HashingEmbedder implements Embedder { return texts.map((t) => this.embed(t)); } } + +/** Local semantic embeddings over ONNX (no API key, nothing leaves the machine after the one-time + * model download). Backed by `@huggingface/transformers`, which is NOT a dependency of this package + * — install it yourself (`npm i @huggingface/transformers`) to enable `MIDAS_MCP_EMBEDDER=local`; + * without it the server falls back to the offline hashing embedder and says so on stderr. + * NOTE: vectors are model-specific. Don't point a bge-small (384-dim) process at a store written + * with another embedder — keep one embedder per store file. */ +export class LocalEmbedder implements Embedder { + readonly dim: number; + readonly model: string; + private extractor: (text: string | string[], opts: object) => Promise<{ data: ArrayLike; dims: number[] }>; + + private constructor(extractor: LocalEmbedder["extractor"], dim: number, model: string) { + this.extractor = extractor; + this.dim = dim; + this.model = model; + } + + static async create(model = "Xenova/bge-small-en-v1.5"): Promise { + let transformers: { pipeline: (task: string, model: string, opts?: object) => Promise }; + try { + const spec = "@huggingface/transformers"; // variable specifier: optional dep, tsc must not resolve it + transformers = await import(spec); + } catch { + throw new Error( + "semantic embeddings need the optional `@huggingface/transformers` package — `npm i @huggingface/transformers`", + ); + } + const extractor = (await transformers.pipeline("feature-extraction", model, { + dtype: "q8", // quantized: ~4x smaller download, near-identical retrieval + })) as LocalEmbedder["extractor"]; + // bge-family models pool on the [CLS] token; probe once to learn the dimension. + const probe = await extractor("dimension probe", { pooling: "cls", normalize: true }); + return new LocalEmbedder(extractor, probe.dims[probe.dims.length - 1], model); + } + + async embed(text: string): Promise { + const out = await this.extractor(text, { pooling: "cls", normalize: true }); + return Float32Array.from(out.data as ArrayLike); + } + + async embedMany(texts: string[]): Promise { + const out: Float32Array[] = []; + for (const t of texts) out.push(await this.embed(t)); + return out; + } +} diff --git a/packages/midas-ts/src/index.ts b/packages/midas-ts/src/index.ts index 2e97e6d..31f3ec4 100644 --- a/packages/midas-ts/src/index.ts +++ b/packages/midas-ts/src/index.ts @@ -1,5 +1,5 @@ export * from "./types.js"; -export { HashingEmbedder, cosine, l2Normalize, tokenize, type Embedder } from "./embeddings.js"; +export { HashingEmbedder, LocalEmbedder, cosine, l2Normalize, tokenize, type Embedder } from "./embeddings.js"; export { contentImportance, structuralImportance, type ImportanceScorer } from "./importance.js"; export { BM25 } from "./bm25.js"; export { InMemoryStore, SQLiteStore, type Predicate } from "./store.js"; diff --git a/packages/midas-ts/src/mcp.ts b/packages/midas-ts/src/mcp.ts index c7934f5..bcb3aaf 100644 --- a/packages/midas-ts/src/mcp.ts +++ b/packages/midas-ts/src/mcp.ts @@ -1,11 +1,13 @@ /** Midas as an MCP server (TypeScript) — same tool surface, env knobs, injected policy, and * SQLite schema as the Python `midas.mcp_server`, so the two are interchangeable and can even - * share one DB file live. No LLM and no network at ingest/query (hashing embedder; a local ONNX - * semantic embedder is the planned next step — use the Python server when you need bge today). */ + * share one DB file live. No LLM and no network at ingest/query. MIDAS_MCP_EMBEDDER=local enables + * local ONNX semantic embeddings when the optional `@huggingface/transformers` package is + * installed; otherwise the byte-parity hashing embedder is used (and the fallback is announced). */ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"; import { z } from "zod"; +import { HashingEmbedder, LocalEmbedder, type Embedder } from "./embeddings.js"; import { decideMemoryUse, MEMORY_USES, type MemoryUse } from "./guard.js"; import { Memory, DEFAULT_POLICY } from "./memory.js"; import { structuralImportance } from "./importance.js"; @@ -19,7 +21,26 @@ const ACTOR = process.env.MIDAS_MCP_ACTOR ?? "midas-mcp-ts"; const NAMESPACE = process.env.MIDAS_MCP_NAMESPACE ?? ""; const SUPERSEDE = process.env.MIDAS_MCP_SUPERSEDE !== "0"; -function buildMemory(): Memory { +async function buildEmbedder(): Promise { + const choice = (process.env.MIDAS_MCP_EMBEDDER ?? "hashing").toLowerCase(); + if (choice && choice !== "hashing") { + try { + // "local" -> the default bge-small; any other value is treated as a model id (Python parity). + const embedder = await (choice === "local" + ? LocalEmbedder.create() + : LocalEmbedder.create(process.env.MIDAS_MCP_EMBEDDER)); + console.error(`[midas-mcp-ts] semantic embeddings: ${embedder.model} (${embedder.dim}d, local ONNX)`); + return embedder; + } catch (err) { + console.error( + `[midas-mcp-ts] local embedder unavailable (${(err as Error).message}) — using the offline hashing embedder`, + ); + } + } + return new HashingEmbedder(); +} + +async function buildMemory(): Promise { let store: InMemoryStore | undefined; const db = process.env.MIDAS_MCP_DB; if (db) { @@ -28,13 +49,14 @@ function buildMemory(): Memory { } return new Memory({ store, + embedder: await buildEmbedder(), importanceScorer: structuralImportance, policy: { ...DEFAULT_POLICY, minImportance: MIN_IMPORTANCE }, supersede: SUPERSEDE, }); } -const mem = buildMemory(); +const mem = await buildMemory(); function ns(namespace: string): string { return namespace || NAMESPACE; @@ -105,7 +127,7 @@ export function createServer(): McpServer { }, }, async (args) => { - const rec = mem.remember(args.content, { + const rec = await mem.remember(args.content, { kind: args.kind as MemoryKind, importance: args.importance || null, source: `mcp:${args.session}`, @@ -136,7 +158,7 @@ export function createServer(): McpServer { }, }, async (args) => { - const result = mem.capture(args.content, { + const result = await mem.capture(args.content, { kind: args.kind as MemoryKind, source: `mcp:${args.session}`, provenance: args.provenance, @@ -170,7 +192,7 @@ export function createServer(): McpServer { }, }, async (args) => { - const hits = mem.recall(args.query, { + const hits = await mem.recall(args.query, { limit: args.limit, kind: (args.kind || null) as MemoryKind | null, minImportance: args.min_importance || null, @@ -219,12 +241,12 @@ export function createServer(): McpServer { }, async (args) => text( - mem.buildContext(args.query, { + (await mem.buildContext(args.query, { tokenBudget: args.token_budget, limit: args.limit, hybrid: args.hybrid, metadataFilter: nsFilter(args.namespace), - }).text, + })).text, ), ); @@ -260,7 +282,7 @@ export function createServer(): McpServer { }, }, async (args) => { - const hits = mem.recall(args.query, { limit: args.limit, metadataFilter: nsFilter(args.namespace) }); + const hits = await mem.recall(args.query, { limit: args.limit, metadataFilter: nsFilter(args.namespace) }); return json( decideMemoryUse(hits.map((h) => h.record), { intendedUse: args.intended_use, @@ -296,7 +318,7 @@ export function createServer(): McpServer { }, }, async (args) => { - const matched = mem.forgetMatching(args.query, { + const matched = await mem.forgetMatching(args.query, { minRelevance: args.min_relevance, limit: args.limit, metadataFilter: nsFilter(args.namespace), diff --git a/packages/midas-ts/src/memory.ts b/packages/midas-ts/src/memory.ts index cf42840..7a5de6d 100644 --- a/packages/midas-ts/src/memory.ts +++ b/packages/midas-ts/src/memory.ts @@ -161,10 +161,10 @@ export class Memory { this.now = opts.now ?? (() => Date.now() / 1000); } - remember(content: string, opts: RememberOptions = {}): MemoryRecord { + async remember(content: string, opts: RememberOptions = {}): Promise { content = (content ?? "").trim(); if (!content) throw new Error("Memory.remember: `content` is required"); - const embedding = this.embedder.embed(content); + const embedding = await this.embedder.embed(content); const ts = opts.createdAt ?? this.now(); const importance = this.deriveImportance(opts.importance ?? null, content); const provenance = opts.provenance ?? "observation"; @@ -190,7 +190,7 @@ export class Memory { return record; } - capture(content: string, opts: RememberOptions = {}): CaptureResult { + async capture(content: string, opts: RememberOptions = {}): Promise { content = (content ?? "").trim(); if (!content) return { stored: false, reason: "empty content", record: null, importance: 0 }; const scorer = this.importanceScorer ?? contentImportance; @@ -208,7 +208,7 @@ export class Memory { }; } if (this.policy.dedupThreshold > 0) { - const near = this.store.search(this.embedder.embed(content), { limit: 1 }); + const near = this.store.search(await this.embedder.embed(content), { limit: 1 }); if (near.length && near[0][0] >= this.policy.dedupThreshold) { const existing = near[0][1]; const upgraded = this.upgradeProvenance(existing, opts); @@ -220,7 +220,7 @@ export class Memory { }; } } - const record = this.remember(content, { ...opts, kind, importance }); + const record = await this.remember(content, { ...opts, kind, importance }); return { stored: true, reason: "stored", record, importance }; } @@ -243,9 +243,9 @@ export class Memory { return clampImportance(this.importanceScorer ? this.importanceScorer(content) : 1); } - recall(query: string, opts: RecallOptions = {}): RecallHit[] { + async recall(query: string, opts: RecallOptions = {}): Promise { const limit = opts.limit ?? DEFAULT_RECALL_LIMIT; - const qEmb = this.embedder.embed(query); + const qEmb = await this.embedder.embed(query); const now = opts.now ?? this.now(); const predicate = (r: MemoryRecord): boolean => { if (opts.kind && r.kind !== opts.kind) return false; @@ -398,13 +398,13 @@ export class Memory { return [...best.values()]; } - buildContext( + async buildContext( query: string, opts: RecallOptions & { tokenBudget?: number; header?: boolean; maxRecordChars?: number } = {}, - ): ContextBlock { + ): Promise { const budget = opts.tokenBudget ?? 512; const now = opts.now ?? this.now(); - const hits = this.recall(query, { ...opts, now }); + const hits = await this.recall(query, { ...opts, now }); let records = hits.map((h) => h.record); if (this.supersede) records = records.filter((r) => this.resolveHead(r).id === r.id); @@ -440,12 +440,12 @@ export class Memory { return this.store.delete(recordId); } - forgetMatching( + async forgetMatching( query: string, opts: { minRelevance?: number; limit?: number; metadataFilter?: Record | null; dryRun?: boolean } = {}, - ): MemoryRecord[] { + ): Promise { const minRelevance = opts.minRelevance ?? 0.5; - const hits = this.recall(query, { + const hits = await this.recall(query, { limit: opts.limit ?? 20, minRelevance, metadataFilter: opts.metadataFilter ?? null, diff --git a/packages/midas-ts/test/core.test.mjs b/packages/midas-ts/test/core.test.mjs index 8661652..d7c37c9 100644 --- a/packages/midas-ts/test/core.test.mjs +++ b/packages/midas-ts/test/core.test.mjs @@ -27,89 +27,89 @@ test("hashing embedder is bit-comparable with the Python implementation", () => } }); -test("remember -> recall roundtrip ranks the relevant memory first", () => { +test("remember -> recall roundtrip ranks the relevant memory first", async () => { const mem = new Memory(); - mem.remember("Decision: the primary database is PostgreSQL.", { kind: "constraint", importance: 5 }); - mem.remember("We chatted about the weekend weather.", { kind: "chat" }); - const hits = mem.recall("which database did we pick?", { limit: 2 }); + await mem.remember("Decision: the primary database is PostgreSQL.", { kind: "constraint", importance: 5 }); + await mem.remember("We chatted about the weekend weather.", { kind: "chat" }); + const hits = await mem.recall("which database did we pick?", { limit: 2 }); assert.ok(hits.length >= 1); assert.match(hits[0].record.content, /PostgreSQL/); }); -test("min_relevance_ratio prunes hits far below the top hit (default 0.3)", () => { +test("min_relevance_ratio prunes hits far below the top hit (default 0.3)", async () => { const mem = new Memory(); - mem.remember("the launch date is September 14", { kind: "fact", importance: 3 }); - mem.remember("completely unrelated quarterly parking rota", { kind: "note", importance: 3 }); - const hits = mem.recall("when is the launch date?", { limit: 5 }); + await mem.remember("the launch date is September 14", { kind: "fact", importance: 3 }); + await mem.remember("completely unrelated quarterly parking rota", { kind: "note", importance: 3 }); + const hits = await mem.recall("when is the launch date?", { limit: 5 }); assert.ok(hits.every((h) => h.relevance >= 0.3 * hits[0].relevance)); - const all = mem.recall("when is the launch date?", { limit: 5, minRelevanceRatio: 0 }); + const all = await mem.recall("when is the launch date?", { limit: 5, minRelevanceRatio: 0 }); assert.equal(all.length, 2); }); -test("hybrid recall finds exact identifiers", () => { +test("hybrid recall finds exact identifiers", async () => { const mem = new Memory(); - mem.remember("ticket ABC-123 tracks the login bug", { kind: "note", importance: 3 }); - mem.remember("the weather is nice today", { kind: "chat" }); - const hits = mem.recall("ABC-123", { hybrid: true, limit: 2 }); + await mem.remember("ticket ABC-123 tracks the login bug", { kind: "note", importance: 3 }); + await mem.remember("the weather is nice today", { kind: "chat" }); + const hits = await mem.recall("ABC-123", { hybrid: true, limit: 2 }); assert.match(hits[0].record.content, /ABC-123/); }); -test("capture enforces the relevance policy and dedups", () => { +test("capture enforces the relevance policy and dedups", async () => { const mem = new Memory({ importanceScorer: structuralImportance }); - const fact = mem.capture("My API rate limit is 5000 requests per hour.", { kind: "fact" }); + const fact = await mem.capture("My API rate limit is 5000 requests per hour.", { kind: "fact" }); assert.equal(fact.stored, true); - const filler = mem.capture("haha ok cool"); + const filler = await mem.capture("haha ok cool"); assert.equal(filler.stored, false); assert.match(filler.reason, /floor/); - const dup = mem.capture("My API rate limit is 5000 requests per hour.", { kind: "fact" }); + const dup = await mem.capture("My API rate limit is 5000 requests per hour.", { kind: "fact" }); assert.equal(dup.stored, false); assert.match(dup.reason, /duplicate/); }); -test("supersession retires the stale fact and recall returns the head", () => { +test("supersession retires the stale fact and recall returns the head", async () => { const mem = new Memory({ supersede: true, supersedeThreshold: 0.3, supersedeLowerThreshold: 0.1 }); - const oldRec = mem.remember("the launch date is September 14", { kind: "fact", importance: 4 }); - mem.remember("actually the launch date moved to October 2", { kind: "fact", importance: 4 }); + const oldRec = await mem.remember("the launch date is September 14", { kind: "fact", importance: 4 }); + await mem.remember("actually the launch date moved to October 2", { kind: "fact", importance: 4 }); assert.notEqual(mem.store.get(oldRec.id).supersededBy, null); - const hits = mem.recall("when is the launch date?", { limit: 2 }); + const hits = await mem.recall("when is the launch date?", { limit: 2 }); assert.match(hits[0].record.content, /October 2/); }); -test("forget repairs supersession chains; forgetMatching dry-run audits", () => { +test("forget repairs supersession chains; forgetMatching dry-run audits", async () => { const mem = new Memory({ supersede: true, supersedeThreshold: 0.3, supersedeLowerThreshold: 0.1 }); - const a = mem.remember("the launch date is September 14", { kind: "fact", importance: 4 }); - const b = mem.remember("actually the launch date moved to October 2", { kind: "fact", importance: 4 }); - const c = mem.remember("actually the launch date moved again to November 5", { kind: "fact", importance: 4 }); + const a = await mem.remember("the launch date is September 14", { kind: "fact", importance: 4 }); + const b = await mem.remember("actually the launch date moved to October 2", { kind: "fact", importance: 4 }); + const c = await mem.remember("actually the launch date moved again to November 5", { kind: "fact", importance: 4 }); assert.equal(mem.forget(b.id), true); assert.equal(mem.resolveHead(mem.store.get(a.id)).id, c.id); - const preview = mem.forgetMatching("launch date", { minRelevance: 0.1, dryRun: true }); + const preview = await mem.forgetMatching("launch date", { minRelevance: 0.1, dryRun: true }); assert.ok(preview.length >= 1); assert.ok(mem.store.all().length === 2, "dry run must not delete"); }); -test("buildContext emits lean dated lines with a Today anchor within budget", () => { +test("buildContext emits lean dated lines with a Today anchor within budget", async () => { const mem = new Memory(); - mem.remember("Launch date moved to September 14.", { kind: "fact", importance: 5 }); - const block = mem.buildContext("when do we launch?", { tokenBudget: 100 }); + await mem.remember("Launch date moved to September 14.", { kind: "fact", importance: 5 }); + const block = await mem.buildContext("when do we launch?", { tokenBudget: 100 }); assert.match(block.text, /# Today is \d{4}-\d{2}-\d{2}/); assert.match(block.text, /September 14/); assert.ok(!block.text.includes("id:")); }); -test("forgetDecayed evicts low-value records but protects durable kinds", () => { +test("forgetDecayed evicts low-value records but protects durable kinds", async () => { const mem = new Memory(); - mem.remember("haha filler note one", { kind: "note", importance: 1 }); - mem.remember("user PII must never be stored in plaintext", { kind: "constraint", importance: 5 }); + await mem.remember("haha filler note one", { kind: "note", importance: 1 }); + await mem.remember("user PII must never be stored in plaintext", { kind: "constraint", importance: 5 }); const dropped = mem.forgetDecayed({ maxRecords: 1 }); assert.equal(dropped.length, 1); assert.match(mem.store.all()[0].content, /PII/); }); -test("guard blocks external actions without user_confirmation", () => { +test("guard blocks external actions without user_confirmation", async () => { const mem = new Memory(); - mem.remember("Deploy target is staging.", { kind: "constraint", importance: 5, provenance: "observation" }); - const records = mem.recall("deploy target", { limit: 3 }).map((h) => h.record); + await mem.remember("Deploy target is staging.", { kind: "constraint", importance: 5, provenance: "observation" }); + const records = (await mem.recall("deploy target", { limit: 3 })).map((h) => h.record); assert.equal(decideMemoryUse(records, { intendedUse: "external_action" }).allowed, false); assert.equal(decideMemoryUse(records, { intendedUse: "planning" }).allowed, true); }); @@ -120,18 +120,18 @@ test("importance scoring separates facts from filler", () => { assert.ok(structuralImportance("I prefer all answers in metric units.") >= 3); }); -test("SQLiteStore persists across reopen and shares writes live across connections", () => { +test("SQLiteStore persists across reopen and shares writes live across connections", async () => { const dir = mkdtempSync(join(tmpdir(), "midas-ts-")); const db = join(dir, "mem.db"); const a = new Memory({ store: new SQLiteStore(db) }); - a.remember("decision: queue runs on Redis Streams", { kind: "fact", importance: 5 }); + await a.remember("decision: queue runs on Redis Streams", { kind: "fact", importance: 5 }); const b = new Memory({ store: new SQLiteStore(db) }); - const hits = b.recall("what does the queue run on?", { limit: 1 }); + const hits = await b.recall("what does the queue run on?", { limit: 1 }); assert.match(hits[0].record.content, /Redis Streams/); // cross-connection live visibility (data_version probe) - b.remember("note from the other connection", { kind: "note" }); + await b.remember("note from the other connection", { kind: "note" }); assert.ok(a.store.all().some((r) => r.content.includes("other connection"))); }); @@ -139,3 +139,15 @@ test("cosine of identical text embeddings is ~1", () => { const e = new HashingEmbedder(); assert.ok(Math.abs(cosine(e.embed("hello world test"), e.embed("hello world test")) - 1) < 1e-6); }); + +test("LocalEmbedder rejects with install guidance when @huggingface/transformers is absent", async (t) => { + try { + await import("@huggingface/transformers"); + t.skip("optional runtime installed here — the happy path needs a model download, exercised manually"); + return; + } catch { + /* not installed — the case this test pins */ + } + const { LocalEmbedder } = await import("../dist/index.js"); + await assert.rejects(LocalEmbedder.create(), /@huggingface\/transformers/); +}); diff --git a/tests/test_importers.py b/tests/test_importers.py new file mode 100644 index 0000000..c67e16c --- /dev/null +++ b/tests/test_importers.py @@ -0,0 +1,102 @@ +"""`midas import --from claude-md|cursorrules|jsonl` — file-based agent memory becomes first-class, +governable Midas records. No LLM, idempotent, governance-conscious provenance defaults.""" +from __future__ import annotations + +import argparse +import json + +from midas import cli +from midas.importers import parse_markdown_rules, to_record_dicts + +CLAUDE_MD = """# Project rules + +Always run the full test suite before committing. + +## Database + +- The primary database is **PostgreSQL** — never MySQL. +- Migrations must be reviewed by a human. + +```bash +# this fenced command block must be skipped +rm -rf build && make +``` + +## Style + +1. Use `ruff` with line-length 100. +ok +""" + + +def test_parse_markdown_rules() -> None: + items = parse_markdown_rules(CLAUDE_MD) + contents = [i.content for i in items] + assert "Always run the full test suite before committing." in contents + assert "The primary database is PostgreSQL — never MySQL." in contents # inline md stripped + assert "Use ruff with line-length 100." in contents # numbered bullets too + assert not any("rm -rf" in c for c in contents) # fenced blocks skipped + assert not any(c == "ok" for c in contents) # one-word noise dropped + by_content = {i.content: i.section for i in items} + assert by_content["Migrations must be reviewed by a human."] == "Database" + + +def test_to_record_dicts_defaults_are_governance_safe() -> None: + items = parse_markdown_rules(CLAUDE_MD) + dicts = to_record_dicts(items, source_file="CLAUDE.md", project="apollo") + assert all(d["provenance"] == "observation" for d in dicts) # cannot authorize guarded actions + assert all(d["kind"] == "constraint" and d["importance"] == 4 for d in dicts) + assert all(d["metadata"] == {"imported_from": "CLAUDE.md", "project": "apollo"} for d in dicts) + sectioned = [d for d in dicts if d["content"].startswith("Database: ")] + assert sectioned # heading context is kept in the content + + confirmed = to_record_dicts(items, source_file="CLAUDE.md", confirmed=True) + assert all(d["provenance"] == "user_confirmation" for d in confirmed) + + +def test_cli_import_claude_md_roundtrip(tmp_path, capsys) -> None: + from midas.sqlite_store import SQLiteStore + + md = tmp_path / "CLAUDE.md" + md.write_text(CLAUDE_MD) + db = str(tmp_path / "m.sqlite3") + ns = argparse.Namespace(file=str(md), source="claude-md", db=db, project="apollo", + confirmed=False, overwrite=False) + assert cli.cmd_import(ns) == 0 + out = capsys.readouterr().out + assert "imported" in out and "provenance=observation" in out + + recs = SQLiteStore(db).all() + assert any("PostgreSQL" in r.content for r in recs) + assert all(r.metadata.get("imported_from") == "CLAUDE.md" for r in recs) + n = len(recs) + + assert cli.cmd_import(ns) == 0 # idempotent: same file, nothing duplicated + assert len(SQLiteStore(db).all()) == n + assert "already present" in capsys.readouterr().out + + +def test_cli_import_jsonl(tmp_path, capsys) -> None: + from midas.sqlite_store import SQLiteStore + + f = tmp_path / "mem.jsonl" + f.write_text(json.dumps({"content": "User prefers dark mode.", "kind": "preference", + "importance": 5}) + "\n\n" + + json.dumps({"content": "The launch date is September 14.", "kind": "fact"}) + "\n") + db = str(tmp_path / "m.sqlite3") + ns = argparse.Namespace(file=str(f), source="jsonl", db=db, project=None, + confirmed=False, overwrite=False) + assert cli.cmd_import(ns) == 0 + recs = {r.content: r for r in SQLiteStore(db).all()} + assert recs["User prefers dark mode."].kind == "preference" + assert recs["User prefers dark mode."].importance == 5 + assert recs["The launch date is September 14."].source == "import:mem.jsonl" + + +def test_cli_import_empty_file_fails_cleanly(tmp_path, capsys) -> None: + f = tmp_path / "empty.md" + f.write_text("\n\n") + ns = argparse.Namespace(file=str(f), source="claude-md", db=str(tmp_path / "m.sqlite3"), + project=None, confirmed=False, overwrite=False) + assert cli.cmd_import(ns) == 1 + assert "nothing importable" in capsys.readouterr().err From 2f8eadd70fc427a87ead2d5d4577356c8482f14e Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 03:00:35 +0000 Subject: [PATCH 05/10] Fix multi-OS CI: sandbox client-config paths in tests; deterministic tamper test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The new macOS/Windows jobs caught exactly what they exist for — in the tests: - VS Code/Zed/Claude Desktop resolve OS-specific roots (macOS Library/, Windows %APPDATA%) that ignore a patched Path.home(), so the client-wiring tests looked in the wrong place off-Linux (and could have touched a runner's real configs). A _sandbox_client_paths helper now pins every lookup into tmp_path on all three OSes. - test_tampering_breaks_the_chain rewrote the sha's first hex char to 'f' — a 1/16 no-op (hit on the Windows runner). Tamper the op field instead, which is always a real change. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- tests/test_audit_chain.py | 2 +- tests/test_cli.py | 26 ++++++++++++++++++-------- 2 files changed, 19 insertions(+), 9 deletions(-) diff --git a/tests/test_audit_chain.py b/tests/test_audit_chain.py index 407e0c5..20711ed 100644 --- a/tests/test_audit_chain.py +++ b/tests/test_audit_chain.py @@ -51,7 +51,7 @@ def test_tampering_breaks_the_chain(tmp_path) -> None: mem.store.close() con = sqlite3.connect(db) # an attacker rewrites history directly - con.execute("UPDATE audit_log SET content_sha = 'f' || substr(content_sha, 2) WHERE seq = 2") + con.execute("UPDATE audit_log SET op = 'delete' WHERE seq = 2") # was 'put' — always a real change con.commit() con.close() diff --git a/tests/test_cli.py b/tests/test_cli.py index fb4d590..26c3ec6 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -45,6 +45,16 @@ def test_init_creates_the_store(tmp_path) -> None: assert rc == 0 and db.exists() # the shared store is created (parent dir too) +def _sandbox_client_paths(monkeypatch, tmp_path) -> None: + """Point every client-config lookup into tmp_path. VS Code, Zed, and Claude Desktop resolve + OS-specific roots (macOS Library/, Windows %APPDATA%) that ignore a patched `Path.home()` — + without this, the suite behaves differently per OS and could touch a CI runner's real configs.""" + monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + monkeypatch.setattr(cli, "_vscode_user_dir", lambda: tmp_path / "Code/User") + monkeypatch.setattr(cli, "_zed_settings_path", lambda: tmp_path / "zed/settings.json") + monkeypatch.setenv("APPDATA", str(tmp_path)) + + def test_init_json_receipt_dry_run(tmp_path, capsys) -> None: db = tmp_path / "memory.sqlite3" rc = cli.cmd_init(argparse.Namespace(db=str(db), dry_run=True, all=False, json=True)) @@ -64,7 +74,7 @@ def test_init_json_receipt_dry_run(tmp_path, capsys) -> None: def test_init_json_receipt_real_run(tmp_path, capsys, monkeypatch) -> None: monkeypatch.setattr(cli.shutil, "which", lambda exe: None) # no client CLIs → no subprocesses - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) # sandboxed client configs + _sandbox_client_paths(monkeypatch, tmp_path) cursor = tmp_path / ".cursor/mcp.json" cursor.parent.mkdir(parents=True) cursor.write_text("{}") @@ -92,7 +102,7 @@ def test_init_json_receipt_real_run(tmp_path, capsys, monkeypatch) -> None: def test_status_json_receipt_nothing_wired(tmp_path, capsys, monkeypatch) -> None: - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + _sandbox_client_paths(monkeypatch, tmp_path) rc = cli.cmd_status(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), json=True)) assert rc == 0 receipt = json.loads(capsys.readouterr().out) @@ -103,7 +113,7 @@ def test_status_json_receipt_nothing_wired(tmp_path, capsys, monkeypatch) -> Non def test_status_json_parses_codex_toml(tmp_path, capsys, monkeypatch) -> None: - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + _sandbox_client_paths(monkeypatch, tmp_path) codex = tmp_path / ".codex/config.toml" codex.parent.mkdir(parents=True) codex.write_text('[mcp_servers.midas]\ncommand = "midas-mcp"\n') @@ -133,8 +143,8 @@ def test_merge_mcp_json_custom_key(tmp_path) -> None: def test_init_wires_new_clients(tmp_path, capsys, monkeypatch) -> None: monkeypatch.setattr(cli.shutil, "which", lambda exe: None) - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) - vscode = tmp_path / ".config/Code/User/mcp.json" + _sandbox_client_paths(monkeypatch, tmp_path) + vscode = tmp_path / "Code/User/mcp.json" vscode.parent.mkdir(parents=True) vscode.write_text("{}") gemini = tmp_path / ".gemini/settings.json" @@ -160,8 +170,8 @@ def test_init_wires_new_clients(tmp_path, capsys, monkeypatch) -> None: def test_uninstall_removes_from_custom_key(tmp_path, capsys, monkeypatch) -> None: monkeypatch.setattr(cli.shutil, "which", lambda exe: None) - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) - vscode = tmp_path / ".config/Code/User/mcp.json" + _sandbox_client_paths(monkeypatch, tmp_path) + vscode = tmp_path / "Code/User/mcp.json" vscode.parent.mkdir(parents=True) vscode.write_text(json.dumps({"servers": {"midas": {"command": "midas-mcp"}, "other": {"command": "x"}}})) @@ -171,7 +181,7 @@ def test_uninstall_removes_from_custom_key(tmp_path, capsys, monkeypatch) -> Non def test_doctor_json(tmp_path, capsys, monkeypatch) -> None: - monkeypatch.setattr(cli.Path, "home", lambda: tmp_path) + _sandbox_client_paths(monkeypatch, tmp_path) rc = cli.cmd_doctor(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), json=True)) out = json.loads(capsys.readouterr().out) assert out["receipt_kind"] == "doctor" and out["ok"] is (rc == 0) From 26c8273d6e6f2c1a5c3385a5d968ef07725db15c Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 18:27:46 +0000 Subject: [PATCH 06/10] Bench the control-plane (resume/conflicts) + inspector views for conflicts, loops, audit chain MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Agent Continuity Bench grows three axes, all deterministic and $0: resume_fidelity (the one-call pack contains live state / forbidden rules / open loops and never presents superseded values or closed loops as live), conflict_detection (every planted live-live contradiction found), and conflict_precision (no benign look-alike flagged). All → 1.00; wired into `midas bench` / eval.benches with the same all-green verdict rule. - Inspector: three new glass-box views + pure API endpoints — Conflicts (ranked pairs with per-side forget), Open loops (close with a recorded resolution), Audit log (chain verification + hash-only tail). Conflict similarity is cast to float at the source so every JSON consumer works. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- README.md | 5 +- eval/benches.py | 3 + eval/continuity.py | 148 ++++++++++++++++++++++++++++++++++++++++ midas/continuity.py | 5 +- midas/inspector.py | 78 ++++++++++++++++++++- tests/test_benches.py | 4 +- tests/test_inspector.py | 33 +++++++++ 7 files changed, 270 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index 4ccb953..77495e0 100755 --- a/README.md +++ b/README.md @@ -68,8 +68,9 @@ safely** and **resume from cleanly** — which is where similarity search alone | *"How do I speed up the transactions list?"* | the **prior fix** resurfaces, so the agent doesn't re-diagnose it | — | These properties are measured, not asserted — the **[agent-memory bench suite](docs/agent-memory-benches.md)** -scores action-safety, decision-adherence, repeated-mistake avoidance, and adversarial **memory-safety** -across scripted multi-session projects. The safety eval blocks **10 / 10 adversarial attacks** (ASR +scores action-safety, decision-adherence, repeated-mistake avoidance, **resume fidelity**, **conflict +detection/precision** (live contradictions between agents found without over-flagging), and adversarial +**memory-safety** across scripted multi-session projects. The safety eval blocks **10 / 10 adversarial attacks** (ASR **0.00**) — including a planted confirmation next to a prohibition, a confirmation for a *different* action, a provenance-laundering supersession, and a cross-namespace approval — with **no over-blocking** (benign-pass **1.00**). Deterministic, $0, no LLM. **Reproduce every number with one command:** diff --git a/eval/benches.py b/eval/benches.py index 9b616bc..941d883 100644 --- a/eval/benches.py +++ b/eval/benches.py @@ -21,6 +21,9 @@ def run(verbose: bool = True) -> dict: ("Agent Continuity", "action_safety", cont["action_safety"], "→1.00"), ("", "decision_adherence", cont["decision_adherence"], "→1.00"), ("", "repeated_mistake", cont["repeated_mistake"], "→1.00"), + ("", "resume_fidelity", cont["resume_fidelity"], "→1.00"), + ("", "conflict_detection", cont["conflict_detection"], "→1.00"), + ("", "conflict_precision", cont["conflict_precision"], "→1.00"), ("Memory-Safety", "ASR (attack-success)", safe["ASR"], "→0.00"), ("", "benign_pass", safe["benign_pass"], "→1.00"), ("Coding-agent", "decision_currency", code["decision_currency"], "→1.00"), diff --git a/eval/continuity.py b/eval/continuity.py index c062e0b..2166b14 100644 --- a/eval/continuity.py +++ b/eval/continuity.py @@ -13,6 +13,13 @@ not the superseded one (no stale belief quoted as if live). repeated_mistake A bug fixed or a failure hit in an earlier session resurfaces on a related query, so the agent doesn't re-diagnose the fix or repeat the failure. + resume_fidelity The one-call `resume` pack contains what a resuming agent must see (live current + decisions, forbidden rules, open commitments) and does NOT present superseded + values or closed loops as live state. + conflict_detection `memory_conflicts` finds every planted live-live contradiction (value swap under + provenance protection, numeric drift, negation)… + conflict_precision …while flagging none of the planted benign look-alikes (different subjects, + restatements, unrelated facts) — detection without over-flagging. It is a SEED, not a leaderboard — one scripted project that pins these properties so they can't silently regress, and the scaffold to grow toward token-cost / more scenarios. Run: @@ -158,6 +165,143 @@ def build_memory() -> Memory: ) +# ---- resume fidelity ----------------------------------------------------------------------------- + +def build_resume_memory() -> Memory: + """A project mid-flight: live decisions, a forbidden rule, a revision, and one open + one closed + commitment — everything a resuming agent must (and must not) see.""" + from midas.coding import remember_forbidden_action + from midas.continuity import close_loop, remember_commitment + + mem = Memory(embedder=HashingEmbedder()) + mem.remember("From now on, always answer with metric units.", kind="mission", importance=5, + metadata={"standing": True}) + remember_forbidden_action(mem, "Never run destructive migrations on the production database.", + project="apollo") + old = mem.remember("Decision: Apollo uses Redis for caching.", kind="constraint", + provenance="user_confirmation") + new = mem.remember("Decision: Apollo's caching moved from Redis to an in-process LRU cache.", + kind="constraint", provenance="user_confirmation") + rec = mem.store.get(old.id) + rec.superseded_by = new.id + mem.store.put(rec) + remember_commitment(mem, "Migrate the sessions table to UUID keys.", project="apollo") + done = remember_commitment(mem, "Write the release runbook.", project="apollo") + close_loop(mem, done.id, "runbook merged in PR #12") + return mem + + +@dataclass(frozen=True) +class ResumeCase: + must_contain: str + section: str # the resume section the marker must appear under ("" = anywhere) + note: str + + +RESUME_CASES: tuple[ResumeCase, ...] = ( + ResumeCase("metric units", "PINNED:", "the standing directive is pinned into every resume"), + ResumeCase("destructive migrations", "FORBIDDEN:", "the live forbidden rule is always visible"), + ResumeCase("in-process LRU", "CURRENT STATE:", "the CURRENT value of the revised decision"), + ResumeCase("sessions table", "OPEN LOOPS:", "the unclosed commitment resurfaces"), +) +RESUME_MUST_NOT: tuple[tuple[str, str, str], ...] = ( + # (marker, section it must NOT appear under, note). The marker is a phrase unique to the + # superseded/closed record — the CURRENT value may legitimately mention the old one in passing + # ("moved from Redis to…"), so a bare "Redis" would mis-score the revision narrative itself. + ("uses Redis for caching", "CURRENT STATE:", + "the superseded value must not be presented as live state"), + ("release runbook", "OPEN LOOPS:", "a closed loop must not resurface as open"), +) + + +def _resume_sections(context: str) -> dict[str, str]: + """Split a resume context block into {section header: its lines}.""" + sections: dict[str, list[str]] = {} + current = "" + for line in context.splitlines(): + if line.endswith(":") and line == line.upper() or line.endswith(":") and "(" in line: + current = line + sections[current] = [] + else: + sections.setdefault(current, []).append(line) + return {k: "\n".join(v) for k, v in sections.items()} + + +def run_resume(verbose: bool = True) -> float: + from midas.continuity import resume + + pack = resume(build_resume_memory(), token_budget=2_000) + sections = _resume_sections(pack["context"]) + + def section_text(header: str) -> str: + return next((text for h, text in sections.items() if h.startswith(header)), "") + + passed = 0 + total = len(RESUME_CASES) + len(RESUME_MUST_NOT) + for c in RESUME_CASES: + ok = c.must_contain.lower() in section_text(c.section).lower() + passed += ok + if verbose: + print(f" [{'PASS' if ok else 'FAIL'}] resume_fidelity {c.section:<15} " + f"contains {c.must_contain!r} ({c.note})") + for marker, section, note in RESUME_MUST_NOT: + ok = marker.lower() not in section_text(section).lower() + passed += ok + if verbose: + print(f" [{'PASS' if ok else 'FAIL'}] resume_fidelity {section:<15} " + f"excludes {marker!r} ({note})") + return passed / total + + +# ---- conflict detection / precision ---------------------------------------------------------------- + +# Planted live-live contradictions `memory_conflicts` MUST find (each tuple stays live: the value swap +# is provenance-protected from revision, the others are written with supersession off). +CONFLICT_PLANTS: tuple[tuple[str, str, str], ...] = ( + ("The primary database is PostgreSQL.", "The primary database is MySQL.", + "value swap on one slot (user-confirmed vs observed — revision correctly refused)"), + ("The API rate limit is 100 requests per minute.", "The API rate limit is 500 requests per minute.", + "numeric drift on the same fact"), + ("We support Internet Explorer in the dashboard.", + "We no longer support Internet Explorer in the dashboard.", "negation flip"), +) +# Benign look-alikes it must NOT flag. +CONFLICT_BENIGN: tuple[tuple[str, str, str], ...] = ( + ("Project Apollo launches on May 3.", "Project Artemis launches on June 9.", + "two facts about two subjects"), + ("Deploys run from the main branch.", "Deploys run from the main branch.", + "a restatement agrees — consolidation work, not a conflict"), + ("The design system uses Figma tokens.", "Standup happens on Tuesday mornings.", + "unrelated facts"), +) + + +def run_conflicts(verbose: bool = True) -> tuple[float, float]: + from midas.continuity import memory_conflicts + + found_planted = 0 + for a, b, note in CONFLICT_PLANTS: + mem = Memory(embedder=HashingEmbedder()) + mem.remember(a, kind="constraint", provenance="user_confirmation", actor="claude-code") + mem.remember(b, kind="constraint", actor="cursor") + ok = len(memory_conflicts(mem)) == 1 + found_planted += ok + if verbose: + print(f" [{'PASS' if ok else 'FAIL'}] conflict_detection ({note})") + + clean_benign = 0 + for a, b, note in CONFLICT_BENIGN: + mem = Memory(embedder=HashingEmbedder()) + mem.remember(a, kind="fact") + mem.remember(b, kind="fact") + ok = memory_conflicts(mem) == [] + clean_benign += ok + if verbose: + print(f" [{'PASS' if ok else 'FAIL'}] conflict_precision ({note})") + + return found_planted / len(CONFLICT_PLANTS), clean_benign / len(CONFLICT_BENIGN) + + def run(verbose: bool = True) -> dict[str, float]: mem = build_memory() safe = 0 @@ -189,10 +333,14 @@ def run(verbose: bool = True) -> dict[str, float]: print(f" [{'PASS' if ok else 'FAIL'}] repeated_mistake must_resurface=" f"{c.must_resurface!r} ({c.note})") + detection, precision = run_conflicts(verbose=verbose) scores = { "action_safety": safe / len(ACTION_CASES), "decision_adherence": adhered / len(ADHERENCE_CASES), "repeated_mistake": avoided / len(MISTAKE_CASES), + "resume_fidelity": run_resume(verbose=verbose), + "conflict_detection": detection, + "conflict_precision": precision, } if verbose: print(f"\n=== Agent Continuity Bench v0 ===") diff --git a/midas/continuity.py b/midas/continuity.py index a0971a4..dbea1b1 100644 --- a/midas/continuity.py +++ b/midas/continuity.py @@ -154,8 +154,9 @@ def memory_conflicts( hit = _contradiction(mem, record, other, score, nli, nli_threshold) if hit: signal, strength = hit - conflicts.append(Conflict(a=record, b=other, similarity=score, - signal=signal, strength=strength)) + # float() strips numpy scalars from store.search so every consumer can json-serialize + conflicts.append(Conflict(a=record, b=other, similarity=float(score), + signal=signal, strength=float(strength))) conflicts.sort(key=lambda c: c.strength, reverse=True) return conflicts[:limit] diff --git a/midas/inspector.py b/midas/inspector.py index eda454a..845c13a 100644 --- a/midas/inspector.py +++ b/midas/inspector.py @@ -113,6 +113,45 @@ def api_stats(mem: "Memory") -> dict[str, Any]: return {"total": len(recs), "live": sum(1 for r in recs if r.superseded_by is None), "kinds": kinds} +def api_conflicts(mem: "Memory", *, limit: int = 20) -> list[dict[str, Any]]: + """Live beliefs that contradict each other with neither superseding the other — the multi-agent + view. Ranked candidates for a HUMAN to resolve (forget one, or capture the corrected value).""" + from .continuity import memory_conflicts + + return [ + {"signal": c.signal, "similarity": round(c.similarity, 3), + "a": audit_record(c.a), "b": audit_record(c.b)} + for c in memory_conflicts(mem, limit=limit) + ] + + +def api_loops(mem: "Memory") -> list[dict[str, Any]]: + """Open commitments (promised work never closed), oldest — most overdue — first.""" + from .continuity import open_loops + + return [audit_record(r) for r in open_loops(mem)] + + +def api_close_loop(mem: "Memory", loop_id: str, resolution: str) -> dict[str, Any]: + """Close a commitment from the UI: records the resolution and supersedes the open loop.""" + from .continuity import close_loop + + try: + rec = close_loop(mem, loop_id, resolution or "closed via inspector", actor="inspector") + except ValueError as exc: + return {"ok": False, "error": str(exc)} + return {"ok": True, "closed_by": audit_record(rec)} + + +def api_audit_chain(mem: "Memory", *, limit: int = 50) -> dict[str, Any]: + """The tamper-evident mutation log: chain verification + the most recent entries (hashes only).""" + store = mem.store + if not hasattr(store, "verify_audit_log"): + return {"supported": False} + result = store.verify_audit_log() + return {"supported": True, **result, "tail": store.audit_log(limit=limit)} + + def api_overview(mem: "Memory") -> dict[str, Any]: """The memory-health dashboard a team/enterprise needs at a glance: counts, attributability (the compliance metric: fraction with both a source and an actor), revision activity, recency, and the @@ -262,7 +301,8 @@ def api_overview(mem: "Memory") -> dict[str, Any]: .bar .bt{flex:1;height:8px;background:rgba(255,255,255,.07);border-radius:999px;overflow:hidden} .bar .bt i{display:block;height:100%;border-radius:999px;background:linear-gradient(90deg,var(--gold2),var(--gold));box-shadow:0 0 10px rgba(255,200,0,.3)} .bar .bv{width:38px;text-align:right;font-size:12.5px;color:var(--text);font-variant-numeric:tabular-nums;font-weight:500} -@media(max-width:760px){.app{grid-template-columns:1fr}.side{position:static;height:auto;flex-direction:row;flex-wrap:wrap;align-items:center}.navlbl{display:none}.nav{display:flex;gap:2px;flex-wrap:wrap}.nav a.on::before{display:none}.foot{display:none}.main{padding:24px 20px}} +.cpair{display:grid;grid-template-columns:1fr 1fr;gap:10px;margin-top:8px} +@media(max-width:760px){.cpair{grid-template-columns:1fr}.app{grid-template-columns:1fr}.side{position:static;height:auto;flex-direction:row;flex-wrap:wrap;align-items:center}.navlbl{display:none}.nav{display:flex;gap:2px;flex-wrap:wrap}.nav a.on::before{display:none}.foot{display:none}.main{padding:24px 20px}}
@@ -349,7 +392,32 @@ def api_overview(mem: "Memory") -> dict[str, Any]:

Evidence · ${a.evidence.length}

`+(a.evidence.map(e=>card(e)).join('')||'
No supporting memory.
');}; q.onkeydown=e=>{if(e.key==='Enter')go.click()};}, + async conflicts(){main.innerHTML=head('Conflicts','Live beliefs that contradict each other — neither superseded the other. Resolve by forgetting the wrong one, or capture the corrected value.'); + const cs=await get('/api/conflicts'); + main.innerHTML+=cs.length?cs.map(c=>`
+ ${esc(c.signal)}similarity ${c.similarity}
+
${[c.a,c.b].map(r=>card(r, + ``)).join('')}
`).join('') + :'
No unresolved conflicts — every live belief agrees. ✓
';}, + async loops(){main.innerHTML=head('Open loops','Promised work that was never closed — oldest (most overdue) first.'); + const ls=await get('/api/loops'); + main.innerHTML+=ls.length?ls.map(r=>card(r, + ``)).join('') + :'
No open loops — every commitment is closed. ✓
';}, + async auditlog(){main.innerHTML=head('Audit log','The tamper-evident mutation chain — every write, revision, and deletion, hash-linked. No memory content, only hashes.'); + const a=await get('/api/audit_chain'); + if(!a.supported){main.innerHTML+='
This store has no audit chain (in-memory or audit=off).
';return;} + main.innerHTML+=`
${a.ok?'✓':'✕'}
+

${a.ok?'CHAIN INTACT':'CHAIN BROKEN'}

+
${a.entries} entries${a.ok?'':' · first invalid seq: '+a.first_invalid_seq+' — history after this point is untrusted'}
+

Most recent · ${a.tail.length}

` + +a.tail.slice().reverse().map(e=>`
#${e.seq} + ${esc(e.op)} + ${esc(e.record_id.slice(0,8))} · sha ${esc(e.content_sha.slice(0,14))}… + ${ago(e.at)}
`).join('');}, }; +async function closeLoop(id){const res=prompt('How was this loop closed? (recorded as the resolution)'); + if(res===null)return;await post('/api/close_loop',{id,resolution:res});V[tab]();} async function openProj(name){const d=await get('/api/project?name='+encodeURIComponent(name));const o=d.overview,g=d.governance; const stat=(l,v,s='')=>`
${v}
${l}
${s?`
${s}
`:''}
`; const bars=ob=>{const e=Object.entries(ob);if(!e.length)return '
—
';const mx=Math.max(...e.map(x=>x[1])); @@ -416,6 +484,12 @@ def do_GET(self) -> None: # noqa: N802 return self._json(api_diff(mem, hours=float(qs.get("hours", 24)))) if u.path == "/api/audit": return self._json(api_audit(mem, qs.get("query", ""), qs.get("use", "external_action"))) + if u.path == "/api/conflicts": + return self._json(api_conflicts(mem, limit=int(qs.get("limit", 20)))) + if u.path == "/api/loops": + return self._json(api_loops(mem)) + if u.path == "/api/audit_chain": + return self._json(api_audit_chain(mem, limit=int(qs.get("limit", 50)))) self._json({"error": "not found"}, 404) def do_POST(self) -> None: # noqa: N802 @@ -423,6 +497,8 @@ def do_POST(self) -> None: # noqa: N802 body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0))) or b"{}") if u.path == "/api/forget": return self._json(api_forget(mem, body.get("id", ""))) + if u.path == "/api/close_loop": + return self._json(api_close_loop(mem, body.get("id", ""), body.get("resolution", ""))) self._json({"error": "not found"}, 404) def log_message(self, *args: Any) -> None: # keep the console quiet diff --git a/tests/test_benches.py b/tests/test_benches.py index e276e0d..b426e5f 100644 --- a/tests/test_benches.py +++ b/tests/test_benches.py @@ -7,6 +7,8 @@ def test_bench_suite_all_green() -> None: out = run(verbose=False) - assert out["continuity"] == {"action_safety": 1.0, "decision_adherence": 1.0, "repeated_mistake": 1.0} + assert out["continuity"] == {"action_safety": 1.0, "decision_adherence": 1.0, + "repeated_mistake": 1.0, "resume_fidelity": 1.0, + "conflict_detection": 1.0, "conflict_precision": 1.0} assert out["memory_safety"] == {"ASR": 0.0, "benign_pass": 1.0} assert out["coding"] == {"decision_currency": 1.0, "repeated_mistake": 1.0, "forbidden_accuracy": 1.0} diff --git a/tests/test_inspector.py b/tests/test_inspector.py index 6e48361..7ed6dfe 100644 --- a/tests/test_inspector.py +++ b/tests/test_inspector.py @@ -92,3 +92,36 @@ def test_api_overview_health_metrics() -> None: assert 0.0 <= o["attributable"] <= 1.0 # only the first record has source + actor assert o["by_kind"]["constraint"] == 3 assert "by_provenance" in o and "projects" in o + + +def test_api_conflicts_and_loops() -> None: + from midas.continuity import remember_commitment + from midas.inspector import api_close_loop, api_conflicts, api_loops + + mem = _mem() + mem.remember("The primary database is PostgreSQL.", kind="constraint", actor="claude-code") + mem.remember("The primary database is MySQL.", kind="constraint", actor="cursor") + conflicts = api_conflicts(mem) + assert len(conflicts) == 1 and conflicts[0]["signal"] == "value" + assert {"a", "b", "similarity"} <= set(conflicts[0]) + + loop = remember_commitment(mem, "Migrate the sessions table to UUID keys.") + assert [r["id"] for r in api_loops(mem)] == [loop.id] + out = api_close_loop(mem, loop.id, "migrated in PR #42") + assert out["ok"] and api_loops(mem) == [] + assert api_close_loop(mem, "nope", "x")["ok"] is False + + +def test_api_audit_chain(tmp_path) -> None: + from midas.inspector import api_audit_chain + from midas.sqlite_store import SQLiteStore + + assert api_audit_chain(_mem()) == {"supported": False} # in-memory store has no chain + + mem = Memory(store=SQLiteStore(str(tmp_path / "m.sqlite3")), embedder=HashingEmbedder()) + mem.remember("The primary database is PostgreSQL.", kind="constraint") + out = api_audit_chain(mem) + assert out["supported"] and out["ok"] and out["entries"] == 1 + assert out["tail"][0]["op"] == "put" + import json as _json + assert "PostgreSQL" not in _json.dumps(out) # hashes only, never content From 0af1b9216c543cbcb07e9d851ab58340f5b52c9f Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 18:32:31 +0000 Subject: [PATCH 07/10] TS parity: continuity control-plane, cross-runtime audit chain, per-kind TTL MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - packages/midas-ts/src/continuity.ts — resume (one-call onboarding pack), memoryConflicts (heuristic tier: value swap / numeric drift / negation), and open loops (rememberCommitment / openLoops / closeLoop), exposed as MCP tools and SDK exports; "commitment" added to the TS MemoryKind. - SQLiteStore writes the same hash-chained audit_log as Python. The hash formula is canonicalised to integer microseconds on BOTH sides (float FORMATTING differs across runtimes; IEEE multiply+floor doesn't), verified bidirectionally: Python validates TS-written chains and vice versa. - Memory.forgetExpired + parseTtlSpec + maintain(ttl=…) / MIDAS_MCP_TTL. - 20/20 TS tests (6 new: conflicts, loop lifecycle, resume budget, chain tamper detection, audit opt-out). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- midas/sqlite_store.py | 5 +- packages/midas-ts/README.md | 11 +- packages/midas-ts/src/continuity.ts | 263 +++++++++++++++++++++ packages/midas-ts/src/index.ts | 13 +- packages/midas-ts/src/mcp.ts | 148 +++++++++++- packages/midas-ts/src/memory.ts | 24 +- packages/midas-ts/src/policy.ts | 12 + packages/midas-ts/src/store.ts | 91 ++++++- packages/midas-ts/src/types.ts | 3 +- packages/midas-ts/test/continuity.test.mjs | 121 ++++++++++ 10 files changed, 679 insertions(+), 12 deletions(-) create mode 100644 packages/midas-ts/src/continuity.ts create mode 100644 packages/midas-ts/test/continuity.test.mjs diff --git a/midas/sqlite_store.py b/midas/sqlite_store.py index ce9611f..710a802 100755 --- a/midas/sqlite_store.py +++ b/midas/sqlite_store.py @@ -43,8 +43,11 @@ def _audit_hash(seq: int, at: float, op: str, record_id: str, content_sha: str, prev_hash: str) -> str: + # `at` is canonicalised to integer microseconds (floor) — float FORMATTING differs across + # runtimes, but IEEE multiply+floor of the same stored double doesn't, so the TypeScript port + # verifies chains written by Python and vice versa. return hashlib.sha256( - f"{seq}|{at:.6f}|{op}|{record_id}|{content_sha}|{prev_hash}".encode() + f"{seq}|{int(at * 1_000_000)}|{op}|{record_id}|{content_sha}|{prev_hash}".encode() ).hexdigest() diff --git a/packages/midas-ts/README.md b/packages/midas-ts/README.md index 8fd09e6..3e6bef8 100644 --- a/packages/midas-ts/README.md +++ b/packages/midas-ts/README.md @@ -60,11 +60,18 @@ embedder and announces the fallback on stderr — recall still works, lexically. per store file**: vectors are model-specific, so don't point a bge-small process at a store written with the hashing embedder (or with Python's bge-base). +## Continuity control-plane & audit chain (ported) + +- **`resume`** (the one-call session-onboarding pack), **`memory_conflicts`** (live contradictions, + heuristic tier — the NLI gate stays Python-side), and **open loops** (`remember_commitment` / + `open_loops` / `close_loop`) are available as MCP tools and SDK functions. +- The SQLite store writes the same **hash-chained audit log** as Python — the hash formula is + canonicalised (integer microseconds), so **either runtime verifies chains written by the other**. +- Per-kind TTL retention: `Memory.forgetExpired`, `MIDAS_MCP_TTL`, `maintain(ttl=…)`. + ## Not ported (yet) - Local NLI contradiction gating, the cross-encoder reranker, and the eval harness. -- The continuity control-plane (`resume`, `memory_conflicts`, open loops) and the hash-chained - audit log — Python-only for now. ## Requirements diff --git a/packages/midas-ts/src/continuity.ts b/packages/midas-ts/src/continuity.ts new file mode 100644 index 0000000..6a7f84b --- /dev/null +++ b/packages/midas-ts/src/continuity.ts @@ -0,0 +1,263 @@ +/** Continuity control-plane (TypeScript port of `midas/continuity.py`) — resume a session in one + * call, surface live contradictions, and keep commitments across sessions. Deterministic, no LLM. + * The conflict detector is the heuristic tier only (the local NLI gate stays Python-side): same + * gates as the Python fallback — value swap on one slot, numeric drift, negation flip. */ +import { contentWords, properEntities, DURABLE_KINDS, Memory, approxTokens, formatRecord, fmtDate } from "./memory.js"; +import type { MemoryRecord } from "./types.js"; + +const NUM_RE = /\d+(?:[.,:]\d+)*/g; +const NEG_RE = + /\b(?:not|no longer|never|don'?t|doesn'?t|isn'?t|aren'?t|won'?t|can'?t|cannot|shouldn'?t|stop(?:ped)?|forbidden|banned|disallowed|deprecated)\b/i; +// Sentence-initial capitalised cue words that the entity regex picks up as noise. +const ENTITY_NOISE = new Set( + ("the a an this that these those we i you it they he she update updated now note reminder new " + + "also actually however meanwhile today yesterday tomorrow from after before once finally").split(" "), +); + +const entities = (text: string): Set => { + const out = new Set(); + for (const e of properEntities(text)) if (!ENTITY_NOISE.has(e)) out.add(e); + return out; +}; +const numbers = (text: string): Set => new Set(text.match(NUM_RE) ?? []); +const norm = (text: string): string => text.split(/\s+/).join(" ").toLowerCase(); +const diff = (a: Set, b: Set): Set => + new Set([...a].filter((x) => !b.has(x))); +const inter = (a: Set, b: Set): Set => + new Set([...a].filter((x) => b.has(x))); +const setEq = (a: Set, b: Set): boolean => + a.size === b.size && [...a].every((x) => b.has(x)); + +export interface Conflict { + a: MemoryRecord; + b: MemoryRecord; + similarity: number; + signal: "value" | "numeric" | "negation"; + strength: number; +} + +function contradiction(a: MemoryRecord, b: MemoryRecord, similarity: number): Conflict["signal"] | null { + if (norm(a.content) === norm(b.content)) return null; // restatements agree + const shared = inter(contentWords(a.content), contentWords(b.content)); + if (shared.size < 2) return null; + const na = numbers(a.content), nb = numbers(b.content); + const ea = entities(a.content), eb = entities(b.content); + const da = diff(ea, eb), db = diff(eb, ea); + const frameA = diff(diff(contentWords(a.content), ea), na); + const frameB = diff(diff(contentWords(b.content), eb), nb); + if (da.size === 1 && db.size === 1 && setEq(na, nb) && inter(frameA, frameB).size >= 2) { + return "value"; + } + if (da.size && db.size) return null; // each names its own thing → two facts, not one slot + if (na.size && nb.size && !setEq(na, nb)) return "numeric"; + if (NEG_RE.test(a.content) !== NEG_RE.test(b.content)) return "negation"; + return null; +} + +function scopeMatch(record: MemoryRecord, scope: Record | null | undefined): boolean { + if (!scope) return true; + return Object.entries(scope).every(([k, v]) => record.metadata?.[k] === v); +} + +export function memoryConflicts( + mem: Memory, + opts: { + scope?: Record | null; + minSimilarity?: number; + pool?: number; + neighbors?: number; + limit?: number; + } = {}, +): Conflict[] { + const minSimilarity = opts.minSimilarity ?? 0.35; + const live = mem.store + .all() + .filter( + (r) => + r.supersededBy === null && + r.embedding !== null && + (DURABLE_KINDS as readonly string[]).includes(r.kind) && + scopeMatch(r, opts.scope), + ) + .sort((a, b) => b.updatedAt - a.updatedAt) + .slice(0, opts.pool ?? 500); + const liveIds = new Set(live.map((r) => r.id)); + + const conflicts: Conflict[] = []; + const seen = new Set(); + for (const record of live) { + const near = mem.store.search(record.embedding!, { + limit: opts.neighbors ?? 10, + predicate: (r) => r.id !== record.id && liveIds.has(r.id), + }); + for (const [score, other] of near) { + if (score < minSimilarity) break; // sorted by similarity desc + const pair = record.id < other.id ? `${record.id}|${other.id}` : `${other.id}|${record.id}`; + if (seen.has(pair)) continue; + seen.add(pair); + const signal = contradiction(record, other, score); + if (signal) conflicts.push({ a: record, b: other, similarity: score, signal, strength: score }); + } + } + conflicts.sort((a, b) => b.strength - a.strength); + return conflicts.slice(0, opts.limit ?? 20); +} + +// ---- open loops (commitments) ------------------------------------------------------------------- + +export async function rememberCommitment( + mem: Memory, + content: string, + opts: { project?: string; due?: string; actor?: string; source?: string } = {}, +): Promise { + const metadata: Record = {}; + if (opts.project) metadata.project = opts.project; + if (opts.due) metadata.due = opts.due; + return mem.remember(content, { + kind: "commitment", + importance: 4, + provenance: "planning", + actor: opts.actor ?? null, + source: opts.source ?? null, + metadata, + }); +} + +export function openLoops( + mem: Memory, + opts: { scope?: Record | null; limit?: number } = {}, +): MemoryRecord[] { + const records = mem.store + .all() + .filter( + (r) => + r.supersededBy === null && + r.kind === "commitment" && + !r.metadata?.closes && // a resolution record is the close, not a new loop + scopeMatch(r, opts.scope), + ) + .sort((a, b) => a.createdAt - b.createdAt); // oldest = most overdue first + return opts.limit ? records.slice(0, opts.limit) : records; +} + +export async function closeLoop( + mem: Memory, + loopId: string, + resolution: string, + opts: { actor?: string } = {}, +): Promise { + const target = mem.store.get(loopId); + if (!target || target.kind !== "commitment") { + throw new Error(`no open commitment with id '${loopId}'`); + } + const metadata: Record = { closes: loopId }; + for (const k of ["project", "namespace"]) if (target.metadata?.[k]) metadata[k] = target.metadata[k]; + const closing = await mem.remember(`Done: ${resolution}`, { + kind: "commitment", + importance: 2, + provenance: "action", + actor: opts.actor ?? null, + metadata, + }); + target.supersededBy = closing.id; + target.metadata = { ...target.metadata, superseded_at: closing.createdAt }; + mem.store.put(target); + return closing; +} + +// ---- resume: the one-call session-onboarding pack -------------------------------------------------- + +export interface ResumePack { + generatedAt: number; + since: number; + pinned: MemoryRecord[]; + forbidden: MemoryRecord[]; + state: MemoryRecord[]; + added: MemoryRecord[]; + revised: Array<[MemoryRecord, MemoryRecord]>; + openLoops: MemoryRecord[]; + conflicts: Conflict[]; + context: string; + truncated: boolean; +} + +export function resume( + mem: Memory, + opts: { + scope?: Record | null; + project?: string; + since?: number; + tokenBudget?: number; + now?: number; + } = {}, +): ResumePack { + const now = opts.now ?? Date.now() / 1000; + const since = opts.since ?? now - 7 * 86_400; + const scope = opts.project ? { ...(opts.scope ?? {}), project: opts.project } : opts.scope ?? null; + const budget = opts.tokenBudget ?? 900; + + const all = mem.store.all(); + const live = all.filter((r) => r.supersededBy === null && scopeMatch(r, scope)); + const pinned = live + .filter((r) => r.metadata?.standing || r.kind === "mission") + .sort((a, b) => b.updatedAt - a.updatedAt); + const forbidden = live + .filter((r) => r.metadata?.code_kind === "forbidden_action") + .sort((a, b) => b.updatedAt - a.updatedAt); + const skip = new Set([...pinned, ...forbidden].map((r) => r.id)); + const state = live + .filter((r) => (DURABLE_KINDS as readonly string[]).includes(r.kind) && !skip.has(r.id)) + .sort((a, b) => b.updatedAt - a.updatedAt); + + // memory_diff: added = new current beliefs; revised = (old, new) pairs since `since`. + const byId = new Map(all.map((r) => [r.id, r])); + const supersedingIds = new Set(all.map((r) => r.supersededBy).filter((x): x is string => x !== null)); + const added: MemoryRecord[] = []; + const revised: Array<[MemoryRecord, MemoryRecord]> = []; + for (const r of all) { + if (!scopeMatch(r, scope)) continue; + if (r.supersededBy !== null) { + const nw = byId.get(r.supersededBy); + if (nw && nw.createdAt >= since) revised.push([r, nw]); + } else if (r.createdAt >= since && !supersedingIds.has(r.id)) { + added.push(r); + } + } + added.sort((a, b) => b.createdAt - a.createdAt); + revised.sort((a, b) => b[1].createdAt - a[1].createdAt); + + const loops = openLoops(mem, { scope }); + const conflicts = memoryConflicts(mem, { scope, limit: 5 }); + + const lines = [`RESUME (today is ${fmtDate(now)}; changes since ${fmtDate(since)})`]; + let used = approxTokens(lines[0]); + let truncated = false; + const emit = (header: string, rows: string[]): void => { + if (!rows.length) return; + for (const line of [header, ...rows]) { + const cost = approxTokens(line); + if (used + cost > budget) { + truncated = true; + return; + } + lines.push(line); + used += cost; + } + }; + const fmt = (r: MemoryRecord): string => formatRecord(r, { now }); + emit("PINNED:", pinned.map(fmt)); + emit("FORBIDDEN:", forbidden.map(fmt)); + emit("CHANGED (added):", added.map(fmt)); + emit("CHANGED (revised):", + revised.map(([o, n]) => `${fmt(o)}\n -> now: ${n.content.split(/\s+/).join(" ").slice(0, 300)}`)); + emit("CURRENT STATE:", state.map(fmt)); + emit("OPEN LOOPS:", loops.map(fmt)); + emit("CONFLICTS (unresolved — verify before relying on either):", + conflicts.map((c) => + `- [${c.signal}] "${c.a.content.split(/\s+/).join(" ").slice(0, 200)}" vs "${c.b.content.split(/\s+/).join(" ").slice(0, 200)}"`)); + + return { + generatedAt: now, since, pinned, forbidden, state, added, revised, + openLoops: loops, conflicts, context: lines.join("\n"), truncated, + }; +} diff --git a/packages/midas-ts/src/index.ts b/packages/midas-ts/src/index.ts index 31f3ec4..8c43d21 100644 --- a/packages/midas-ts/src/index.ts +++ b/packages/midas-ts/src/index.ts @@ -2,7 +2,7 @@ export * from "./types.js"; export { HashingEmbedder, LocalEmbedder, cosine, l2Normalize, tokenize, type Embedder } from "./embeddings.js"; export { contentImportance, structuralImportance, type ImportanceScorer } from "./importance.js"; export { BM25 } from "./bm25.js"; -export { InMemoryStore, SQLiteStore, type Predicate } from "./store.js"; +export { InMemoryStore, SQLiteStore, type AuditEntry, type Predicate } from "./store.js"; export { Memory, DEFAULT_POLICY, @@ -16,5 +16,14 @@ export { type RecallOptions, type RememberOptions, } from "./memory.js"; -export { AGENT_MEMORY_INSTRUCTIONS, policySummary } from "./policy.js"; +export { AGENT_MEMORY_INSTRUCTIONS, parseTtlSpec, policySummary } from "./policy.js"; +export { + closeLoop, + memoryConflicts, + openLoops, + rememberCommitment, + resume, + type Conflict, + type ResumePack, +} from "./continuity.js"; export { decideMemoryUse, MEMORY_USES, type GuardDecision, type MemoryUse } from "./guard.js"; diff --git a/packages/midas-ts/src/mcp.ts b/packages/midas-ts/src/mcp.ts index bcb3aaf..1228bf7 100644 --- a/packages/midas-ts/src/mcp.ts +++ b/packages/midas-ts/src/mcp.ts @@ -11,7 +11,14 @@ import { HashingEmbedder, LocalEmbedder, type Embedder } from "./embeddings.js"; import { decideMemoryUse, MEMORY_USES, type MemoryUse } from "./guard.js"; import { Memory, DEFAULT_POLICY } from "./memory.js"; import { structuralImportance } from "./importance.js"; -import { AGENT_MEMORY_INSTRUCTIONS, policySummary } from "./policy.js"; +import { AGENT_MEMORY_INSTRUCTIONS, parseTtlSpec, policySummary } from "./policy.js"; +import { + closeLoop as closeLoopImpl, + memoryConflicts, + openLoops as openLoopsView, + rememberCommitment as rememberCommitmentImpl, + resume as resumeView, +} from "./continuity.js"; import { InMemoryStore, SQLiteStore } from "./store.js"; import { MEMORY_PROVENANCE, type MemoryKind, type MemoryProvenance, type MemoryRecord } from "./types.js"; @@ -20,6 +27,7 @@ const MIN_IMPORTANCE = Number(process.env.MIDAS_MCP_MIN_IMPORTANCE ?? "2"); const ACTOR = process.env.MIDAS_MCP_ACTOR ?? "midas-mcp-ts"; const NAMESPACE = process.env.MIDAS_MCP_NAMESPACE ?? ""; const SUPERSEDE = process.env.MIDAS_MCP_SUPERSEDE !== "0"; +const TTL = parseTtlSpec(process.env.MIDAS_MCP_TTL ?? ""); async function buildEmbedder(): Promise { const choice = (process.env.MIDAS_MCP_EMBEDDER ?? "hashing").toLowerCase(); @@ -347,6 +355,138 @@ export function createServer(): McpServer { }, ); + server.registerTool( + "resume", + { + title: "Resume session", + description: + "START OF SESSION: everything needed to pick up where the last session left off, in ONE call — " + + "pinned directives, forbidden rules, what changed in the last `hours` (default: a week), current " + + "state, open commitments, and unresolved conflicts. `context` is prompt-ready and token-budgeted.", + inputSchema: { + project: z.string().default(""), + namespace: z.string().default(""), + hours: z.number().min(0).default(168), + token_budget: z.number().int().min(64).default(900), + }, + }, + async (args) => { + const pack = resumeView(mem, { + scope: nsFilter(args.namespace), + project: args.project || undefined, + since: Date.now() / 1000 - args.hours * 3600, + tokenBudget: args.token_budget, + }); + return json({ + context: pack.context, + truncated: pack.truncated, + counts: { + pinned: pack.pinned.length, forbidden: pack.forbidden.length, state: pack.state.length, + added: pack.added.length, revised: pack.revised.length, + open_loops: pack.openLoops.length, conflicts: pack.conflicts.length, + }, + open_loops: pack.openLoops.map(serializeRecord), + conflicts: pack.conflicts.map((c) => ({ + signal: c.signal, similarity: round3(c.similarity), + a: serializeRecord(c.a), b: serializeRecord(c.b), + })), + }); + }, + ); + + server.registerTool( + "memory_conflicts", + { + title: "Memory conflicts", + description: + "Live beliefs that CONTRADICT each other with neither superseding the other — the multi-agent " + + "failure mode. Ranked candidate pairs (same-slot heuristic: value swap / numbers disagree / one " + + "side negates). Verify with the user, then forget the wrong one or capture the corrected value.", + inputSchema: { namespace: z.string().default(""), limit: z.number().int().min(1).default(10) }, + }, + async (args) => { + const found = memoryConflicts(mem, { scope: nsFilter(args.namespace), limit: args.limit }); + return json({ + count: found.length, + conflicts: found.map((c) => ({ + signal: c.signal, similarity: round3(c.similarity), strength: round3(c.strength), + a: serializeRecord(c.a), b: serializeRecord(c.b), + })), + }); + }, + ); + + server.registerTool( + "open_loops", + { + title: "Open loops", + description: + "Unresolved commitments — work someone said WOULD be done and never closed — oldest (most " + + "overdue) first. Record one with remember_commitment; close it with close_loop.", + inputSchema: { + project: z.string().default(""), + namespace: z.string().default(""), + limit: z.number().int().min(1).default(20), + }, + }, + async (args) => { + const scope: Record = { ...(nsFilter(args.namespace) ?? {}) }; + if (args.project) scope.project = args.project; + const records = openLoopsView(mem, { + scope: Object.keys(scope).length ? scope : null, + limit: args.limit, + }); + return json({ count: records.length, open_loops: records.map(serializeRecord) }); + }, + ); + + server.registerTool( + "remember_commitment", + { + title: "Remember commitment", + description: + "Record a commitment (an OPEN LOOP): promised work that stays visible in open_loops/resume " + + "until closed with close_loop, so promises survive across sessions.", + inputSchema: { + content: z.string(), + project: z.string().default(""), + due: z.string().default(""), + session: z.string().default("default"), + namespace: z.string().default(""), + }, + }, + async (args) => { + const rec = await rememberCommitmentImpl(mem, args.content, { + project: args.project || undefined, + due: args.due || undefined, + actor: ACTOR, + source: `mcp:${args.session}`, + }); + rec.metadata = { ...rec.metadata, ...nsMetadata(args.namespace, args.session) }; + mem.store.put(rec); + return text(`commitment recorded (${rec.id}) — close it with close_loop when done`); + }, + ); + + server.registerTool( + "close_loop", + { + title: "Close loop", + description: + "Close an open commitment: records the resolution and supersedes the open loop, keeping the " + + "promise -> resolution history auditable. Get loop_id from open_loops.", + inputSchema: { loop_id: z.string(), resolution: z.string() }, + }, + async (args) => { + try { + const rec = await closeLoopImpl(mem, args.loop_id, args.resolution, { actor: ACTOR }); + return text(`loop closed (${args.loop_id} -> ${rec.id})`); + } catch (err) { + return text((err as Error).message); + } + }, + ); + server.registerTool( "maintain", { @@ -357,10 +497,13 @@ export function createServer(): McpServer { inputSchema: { max_records: z.number().int().min(0).default(0), min_value: z.number().min(0).default(0), + ttl: z.string().default(""), }, }, async (args) => { const before = mem.store.all().length; + const ttlMap = args.ttl ? parseTtlSpec(args.ttl) : TTL; + const expired = Object.keys(ttlMap).length ? mem.forgetExpired(ttlMap) : []; const forgotten = args.max_records || args.min_value ? mem.forgetDecayed({ @@ -371,8 +514,9 @@ export function createServer(): McpServer { return json({ before, remaining: mem.store.all().length, + expired: expired.length, forgotten: forgotten.length, - removed_ids: forgotten, + removed_ids: [...expired, ...forgotten], }); }, ); diff --git a/packages/midas-ts/src/memory.ts b/packages/midas-ts/src/memory.ts index 7a5de6d..24a6e63 100644 --- a/packages/midas-ts/src/memory.ts +++ b/packages/midas-ts/src/memory.ts @@ -40,7 +40,7 @@ const PROVENANCE_RANK: Record = { user_confirmation: 3, }; -function contentWords(text: string): Set { +export function contentWords(text: string): Set { const out = new Set(); for (const w of text.split(/\s+/)) { const lw = w.toLowerCase(); @@ -49,7 +49,7 @@ function contentWords(text: string): Set { return out; } -function properEntities(text: string): Set { +export function properEntities(text: string): Set { const out = new Set(); for (const m of text.matchAll(PROPER_ENTITY)) { const lw = m[0].toLowerCase(); @@ -509,6 +509,26 @@ export class Memory { } return toDrop.filter((rid) => this.store.delete(rid)); } + + /** Age-based retention: delete records whose kind outlived its TTL (kind -> days). Mirrors the + * Python rules: user-confirmed, standing, and supersession-chain records never expire silently. */ + forgetExpired(ttlByKind: Record, opts: { now?: number } = {}): string[] { + const now = opts.now ?? this.now(); + const records = this.store.all(); + const pointedTo = new Set(records.map((r) => r.supersededBy).filter((x): x is string => x !== null)); + const expired = records + .filter( + (r) => + ttlByKind[r.kind] !== undefined && + now - r.createdAt > ttlByKind[r.kind] * 86_400 && + r.provenance !== "user_confirmation" && + !r.metadata?.standing && + r.supersededBy === null && + !pointedTo.has(r.id), + ) + .sort((a, b) => a.createdAt - b.createdAt); + return expired.filter((r) => this.store.delete(r.id)).map((r) => r.id); + } } function normalize(text: string): string { diff --git a/packages/midas-ts/src/policy.ts b/packages/midas-ts/src/policy.ts index 6a0fbc0..49a80e5 100644 --- a/packages/midas-ts/src/policy.ts +++ b/packages/midas-ts/src/policy.ts @@ -27,3 +27,15 @@ export function policySummary(policy: MemoryPolicy): string { "guard external/destructive actions to user_confirmation provenance" ); } + +/** Parse a retention spec like "chat=30,note=90" (kind -> days) — the MIDAS_MCP_TTL format. + * Malformed entries are skipped rather than crashing the server. */ +export function parseTtlSpec(spec: string): Record { + const ttl: Record = {}; + for (const part of (spec ?? "").split(",")) { + const [kind, days] = part.split("="); + const d = Number(days); + if (kind?.trim() && Number.isFinite(d) && d > 0) ttl[kind.trim()] = d; + } + return ttl; +} diff --git a/packages/midas-ts/src/store.ts b/packages/midas-ts/src/store.ts index 3bcffcb..f4bdbfe 100644 --- a/packages/midas-ts/src/store.ts +++ b/packages/midas-ts/src/store.ts @@ -3,6 +3,7 @@ * SQLiteStore uses the SAME table schema and float32-blob encoding as the Python * `midas/sqlite_store.py`, and the same `PRAGMA data_version` staleness probe — so a TypeScript * MCP server and a Python one can share a single memory file, live, in both directions. */ +import { createHash } from "node:crypto"; import { DatabaseSync } from "node:sqlite"; import type { MemoryRecord } from "./types.js"; @@ -69,12 +70,40 @@ interface Row { embedding: Uint8Array | null; } +const AUDIT_GENESIS = "0".repeat(64); + +function auditHash(seq: number, at: number, op: string, recordId: string, contentSha: string, prevHash: string): string { + // `at` is canonicalised to integer microseconds (floor) — identical to the Python formula, so + // either runtime verifies chains written by the other. + return createHash("sha256") + .update(`${seq}|${Math.floor(at * 1_000_000)}|${op}|${recordId}|${contentSha}|${prevHash}`, "utf8") + .digest("hex"); +} + +function contentSha(record: MemoryRecord): string { + return createHash("sha256") + .update(`${record.id}\x00${record.content}\x00${record.supersededBy ?? ""}`, "utf8") + .digest("hex"); +} + +export interface AuditEntry { + seq: number; + at: number; + op: string; + record_id: string; + content_sha: string; + prev_hash: string; + hash: string; +} + export class SQLiteStore extends InMemoryStore { private db: DatabaseSync; private dataVersion: number; + private auditEnabled: boolean; - constructor(path: string) { + constructor(path: string, opts: { audit?: boolean } = {}) { super(); + this.auditEnabled = opts.audit ?? true; this.db = new DatabaseSync(path); this.db.exec("PRAGMA journal_mode=WAL"); this.db.exec(` @@ -93,10 +122,62 @@ export class SQLiteStore extends InMemoryStore { embedding BLOB ) `); + // Tamper-evident mutation log — same additive table and hash chain as the Python store. + this.db.exec(` + CREATE TABLE IF NOT EXISTS audit_log ( + seq INTEGER PRIMARY KEY, + at REAL NOT NULL, + op TEXT NOT NULL, + record_id TEXT NOT NULL, + content_sha TEXT NOT NULL, + prev_hash TEXT NOT NULL, + hash TEXT NOT NULL + ) + `); this.load(); this.dataVersion = this.currentDataVersion(); } + private audit(op: string, recordId: string, sha: string): void { + const row = this.db.prepare("SELECT seq, hash FROM audit_log ORDER BY seq DESC LIMIT 1").get() as + | { seq: number; hash: string } + | undefined; + const [prevSeq, prevHash] = row ? [Number(row.seq), String(row.hash)] : [0, AUDIT_GENESIS]; + const seq = prevSeq + 1; + const at = Date.now() / 1000; + this.db + .prepare( + "INSERT INTO audit_log (seq, at, op, record_id, content_sha, prev_hash, hash) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run(seq, at, op, recordId, sha, prevHash, auditHash(seq, at, op, recordId, sha, prevHash)); + } + + /** The mutation history (append-only): hashes only, never memory content. */ + auditLog(opts: { limit?: number } = {}): AuditEntry[] { + const rows = this.db + .prepare("SELECT seq, at, op, record_id, content_sha, prev_hash, hash FROM audit_log ORDER BY seq") + .all() as unknown as AuditEntry[]; + const entries = rows.map((r) => ({ ...r, seq: Number(r.seq), at: Number(r.at) })); + return opts.limit ? entries.slice(-opts.limit) : entries; + } + + /** Walk the whole chain recomputing every hash — any edit/removal/reorder breaks it. */ + verifyAuditLog(): { ok: boolean; entries: number; first_invalid_seq: number | null } { + const entries = this.auditLog(); + let prevHash = AUDIT_GENESIS; + let expected = 1; + for (const e of entries) { + const good = + e.seq === expected && + e.prev_hash === prevHash && + e.hash === auditHash(e.seq, e.at, e.op, e.record_id, e.content_sha, e.prev_hash); + if (!good) return { ok: false, entries: entries.length, first_invalid_seq: e.seq }; + prevHash = e.hash; + expected += 1; + } + return { ok: true, entries: entries.length, first_invalid_seq: null }; + } + private currentDataVersion(): number { const row = this.db.prepare("PRAGMA data_version").get() as { data_version: number }; return Number(row.data_version); @@ -156,6 +237,7 @@ export class SQLiteStore extends InMemoryStore { record.supersededBy, blob, ); + if (this.auditEnabled) this.audit("put", record.id, contentSha(record)); } override get(recordId: string): MemoryRecord | null { @@ -178,14 +260,19 @@ export class SQLiteStore extends InMemoryStore { override delete(recordId: string): boolean { this.refreshIfStale(); + const target = super.get(recordId); const existed = super.delete(recordId); - if (existed) this.db.prepare("DELETE FROM memories WHERE id = ?").run(recordId); + if (existed) { + this.db.prepare("DELETE FROM memories WHERE id = ?").run(recordId); + if (this.auditEnabled) this.audit("delete", recordId, target ? contentSha(target) : AUDIT_GENESIS); + } return existed; } override clear(): void { super.clear(); this.db.exec("DELETE FROM memories"); + if (this.auditEnabled) this.audit("clear", "*", AUDIT_GENESIS); } close(): void { diff --git a/packages/midas-ts/src/types.ts b/packages/midas-ts/src/types.ts index 3f685a9..3cc063f 100644 --- a/packages/midas-ts/src/types.ts +++ b/packages/midas-ts/src/types.ts @@ -1,7 +1,7 @@ /** Midas memory types — mirrors `midas/types.py` so records stay portable across the * Python/TypeScript split (including sharing one SQLite file between both servers). */ -export type MemoryKind = "note" | "chat" | "mission" | "fact" | "preference" | "constraint"; +export type MemoryKind = "note" | "chat" | "mission" | "fact" | "preference" | "constraint" | "commitment"; export const MEMORY_KINDS: readonly MemoryKind[] = [ "note", "chat", @@ -9,6 +9,7 @@ export const MEMORY_KINDS: readonly MemoryKind[] = [ "fact", "preference", "constraint", + "commitment", // an open loop: work someone said WILL be done (close via continuity.closeLoop) ]; export type MemoryProvenance = "planning" | "action" | "observation" | "user_confirmation"; diff --git a/packages/midas-ts/test/continuity.test.mjs b/packages/midas-ts/test/continuity.test.mjs new file mode 100644 index 0000000..25c8635 --- /dev/null +++ b/packages/midas-ts/test/continuity.test.mjs @@ -0,0 +1,121 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { + Memory, + SQLiteStore, + closeLoop, + memoryConflicts, + openLoops, + rememberCommitment, + resume, +} from "../dist/index.js"; + +test("memoryConflicts finds the value swap, numeric drift, and negation flip", async () => { + const mem = new Memory(); + await mem.remember("The primary database is PostgreSQL.", { kind: "constraint", actor: "claude-code" }); + await mem.remember("The primary database is MySQL.", { kind: "constraint", actor: "cursor" }); + const value = memoryConflicts(mem); + assert.equal(value.length, 1); + assert.equal(value[0].signal, "value"); + assert.deepEqual(new Set([value[0].a.actor, value[0].b.actor]), new Set(["claude-code", "cursor"])); + + const mem2 = new Memory(); + await mem2.remember("The API rate limit is 100 requests per minute.", { kind: "fact" }); + await mem2.remember("The API rate limit is 500 requests per minute.", { kind: "fact" }); + assert.equal(memoryConflicts(mem2)[0].signal, "numeric"); + + const mem3 = new Memory(); + await mem3.remember("We support Internet Explorer in the dashboard.", { kind: "constraint" }); + await mem3.remember("We no longer support Internet Explorer in the dashboard.", { kind: "constraint" }); + assert.equal(memoryConflicts(mem3)[0].signal, "negation"); +}); + +test("memoryConflicts ignores benign look-alikes and superseded pairs", async () => { + const mem = new Memory(); + await mem.remember("Project Apollo launches on May 3.", { kind: "fact" }); + await mem.remember("Project Artemis launches on June 9.", { kind: "fact" }); + await mem.remember("Deploys run from the main branch.", { kind: "fact" }); + await mem.remember("Deploys run from the main branch.", { kind: "fact" }); + assert.deepEqual(memoryConflicts(mem), []); + + const mem2 = new Memory(); + const old = await mem2.remember("The launch date is September 14.", { kind: "fact" }); + const nw = await mem2.remember("The launch date is September 20.", { kind: "fact" }); + assert.equal(memoryConflicts(mem2).length, 1); // both live → real conflict + old.supersededBy = nw.id; + mem2.store.put(old); + assert.deepEqual(memoryConflicts(mem2), []); // revision resolves it +}); + +test("open loops lifecycle: record, list oldest-first, close with auditable resolution", async () => { + const mem = new Memory(); + const loop = await rememberCommitment(mem, "Migrate the sessions table to UUID keys.", { project: "apollo" }); + assert.equal(loop.kind, "commitment"); + assert.deepEqual(openLoops(mem).map((r) => r.id), [loop.id]); + assert.equal(openLoops(mem, { scope: { project: "apollo" } })[0].id, loop.id); + + const done = await closeLoop(mem, loop.id, "migrated in PR #42"); + assert.deepEqual(openLoops(mem), []); + assert.equal(mem.store.get(loop.id).supersededBy, done.id); + assert.equal(done.metadata.closes, loop.id); + await assert.rejects(closeLoop(mem, "nope", "x"), /no open commitment/); +}); + +test("resume bundles pinned/state/changes/loops/conflicts under a token budget", async () => { + const mem = new Memory(); + await mem.remember("From now on, always answer with metric units.", { + kind: "mission", importance: 5, metadata: { standing: true }, + }); + await mem.remember("Decision: the primary database is PostgreSQL.", { kind: "constraint", importance: 5 }); + await rememberCommitment(mem, "Add the audit log table."); + const pack = resume(mem); + assert.match(pack.context, /PINNED:/); + assert.match(pack.context, /metric units/); + assert.match(pack.context, /CURRENT STATE:/); + assert.match(pack.context, /PostgreSQL/); + assert.match(pack.context, /OPEN LOOPS:/); + assert.equal(pack.truncated, false); + + for (let i = 0; i < 60; i++) { + await mem.remember(`Architecture decision number ${i} about service ${i} routing.`, { + kind: "constraint", importance: 4, + }); + } + assert.equal(resume(mem, { tokenBudget: 80 }).truncated, true); +}); + +test("SQLite audit chain: chained entries, verification, tamper detection", async () => { + const dir = mkdtempSync(join(tmpdir(), "midas-ts-audit-")); + const db = join(dir, "m.db"); + const store = new SQLiteStore(db); + const mem = new Memory({ store }); + const a = await mem.remember("The primary database is PostgreSQL.", { kind: "constraint" }); + await mem.remember("The launch date is September 14.", { kind: "fact" }); + mem.forget(a.id); + + const log = store.auditLog(); + assert.deepEqual(log.map((e) => e.op), ["put", "put", "delete"]); + assert.equal(log[1].prev_hash, log[0].hash); + assert.ok(!JSON.stringify(log).includes("PostgreSQL")); // hashes only, never content + assert.deepEqual(store.verifyAuditLog(), { ok: true, entries: 3, first_invalid_seq: null }); + + // an attacker rewrites history directly → the chain breaks at that entry + const { DatabaseSync } = await import("node:sqlite"); + const raw = new DatabaseSync(db); + raw.exec("UPDATE audit_log SET op = 'delete' WHERE seq = 2"); + raw.close(); + const tampered = new SQLiteStore(db).verifyAuditLog(); + assert.equal(tampered.ok, false); + assert.equal(tampered.first_invalid_seq, 2); +}); + +test("audit can be disabled for perf-sensitive paths", async () => { + const dir = mkdtempSync(join(tmpdir(), "midas-ts-audit-")); + const store = new SQLiteStore(join(dir, "m.db"), { audit: false }); + await new Memory({ store }).remember("perf-path write", { kind: "note" }); + assert.deepEqual(store.auditLog(), []); +}); From 9ac048d7636aff368b80958771a1cedd33fa4e69 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 5 Jul 2026 18:37:35 +0000 Subject: [PATCH 08/10] Adoption + trust closers: Claude Code hook, Mem0/Zep import, SQLCipher at rest; release 0.2.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - `midas init --claude-hook` installs a Claude Code SessionEnd hook (`midas hook capture-session`): each finished session's user turns are OFFERED to capture — the no-LLM policy still decides what's kept. Fail-open (a memory hook must never break a session), idempotent install/uninstall into ~/.claude/settings.json with backup; removed by `midas uninstall`. - `midas import --from mem0|zep`: Mem0 distilled memories → facts (original ids/user/agent attribution preserved in metadata); Zep facts/edges → facts, messages → chat turns. Tolerant of the wrapped and bare-list export shapes. - Encryption at rest (opt-in): SQLiteStore(key=…) / MIDAS_MCP_KEY via the new [encrypted] extra (SQLCipher). Ciphertext on disk (verified: no plaintext, no SQLite header), wrong/no key can't read, audit chain works under encryption, and a set key without the extra FAILS CLOSED. - Release prep 0.2.0: version bump (pyproject, __init__, npm package + MCP handshake), dated CHANGELOG covering the whole batch; Linux CI now installs [encrypted] so those tests run (they skip where wheels are absent). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- .github/workflows/ci.yml | 4 +- CHANGELOG.md | 26 +++++- README.md | 17 +++- midas/__init__.py | 2 +- midas/cli.py | 57 +++++++++++- midas/hooks.py | 138 ++++++++++++++++++++++++++++ midas/importers.py | 61 ++++++++++++ midas/mcp_server.py | 2 + midas/sqlite_store.py | 20 +++- packages/midas-ts/package-lock.json | 4 +- packages/midas-ts/package.json | 2 +- packages/midas-ts/src/mcp.ts | 2 +- pyproject.toml | 3 +- tests/test_encrypted_store.py | 44 +++++++++ tests/test_hooks.py | 110 ++++++++++++++++++++++ tests/test_importers.py | 52 +++++++++++ 16 files changed, 525 insertions(+), 19 deletions(-) create mode 100644 midas/hooks.py create mode 100644 tests/test_encrypted_store.py create mode 100644 tests/test_hooks.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2f0e453..9c3b31b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,8 +16,8 @@ jobs: - uses: actions/setup-python@v5 with: python-version: ${{ matrix.python }} - - name: Install (all extras + dev) - run: pip install ".[all,dev]" + - name: Install (all extras + dev + encrypted) + run: pip install ".[all,dev,encrypted]" - name: Run the suite run: python -m pytest -q diff --git a/CHANGELOG.md b/CHANGELOG.md index 4e67651..aa806f0 100755 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,7 +3,7 @@ Notable changes to Midas. Pre-1.0 — the API may change. Format loosely follows [Keep a Changelog](https://keepachangelog.com/). -## [Unreleased] +## [0.2.0] — 2026-07-05 ### Added - **The continuity control-plane** (`midas/continuity.py` + MCP tools): **`resume`** — everything an @@ -36,6 +36,30 @@ Notable changes to Midas. Pre-1.0 — the API may change. Format loosely follows (`remember`/`capture`/`recall`/`buildContext`/`forgetMatching` return promises). - **CI on Windows and macOS** (core + MCP): the client-wiring code is full of per-OS paths that only Linux exercised. +- **The Agent Continuity Bench measures the new control-plane**: `resume_fidelity` (the one-call pack + contains live state / forbidden rules / open loops and never presents superseded values or closed + loops as live), `conflict_detection`, and `conflict_precision` — all → 1.00, wired into + `midas bench` with the same all-green verdict rule. +- **Inspector views** for the control-plane: Conflicts (ranked contradiction pairs with per-side + forget), Open loops (close with a recorded resolution), and Audit log (chain verification + a + hash-only tail). +- **TypeScript parity for the control-plane and audit chain**: `resume`, `memoryConflicts` + (heuristic tier), open loops, `forgetExpired`/TTL, and the same hash-chained `audit_log` — the hash + is canonicalised to integer microseconds so **either runtime verifies chains written by the other** + (validated bidirectionally). +- **`midas init --claude-hook`**: a Claude Code SessionEnd hook (`midas hook capture-session`) that + offers each finished session's user turns to `capture` — memory even when the agent never calls the + tool, with the same no-LLM keep/skip policy, fail-open by construction, removed by + `midas uninstall`. +- **`midas import --from mem0 | zep`**: migrate Mem0 exports (distilled memories → facts with + original ids/attribution preserved) and Zep exports (facts/edges → facts, messages → chat turns). +- **Encryption at rest (opt-in)**: `SQLiteStore(key=…)` / `MIDAS_MCP_KEY` with the new `[encrypted]` + extra (SQLCipher). The file on disk is ciphertext; opening without the key fails; if the key is set + but the extra is missing Midas **fails closed** instead of silently writing plaintext. + +### Changed +- The TypeScript `Memory` API is async (`remember`/`capture`/`recall`/`buildContext`/ + `forgetMatching` return promises) to support the optional ONNX embedder. ### Added - **`midas init --json` / `midas status --json` — a machine-readable client wiring receipt** (#15). diff --git a/README.md b/README.md index 77495e0..9cc97ba 100755 --- a/README.md +++ b/README.md @@ -135,8 +135,12 @@ midas serve --http --token # require `Authorization: Bearer ` Keep Midas current with **`midas update`**. See your memory anytime with **`midas inspect`**. Already carrying agent memory in files? **`midas import --from claude-md CLAUDE.md`** (or -`--from cursorrules .cursorrules`, `--from jsonl`) turns those rules into first-class, recallable, -governable memories — tagged with where they came from, idempotent on re-run. +`--from cursorrules`, `--from jsonl`, `--from mem0`, `--from zep`) turns those rules and exports into +first-class, recallable, governable memories — tagged with where they came from, idempotent on re-run. + +Want memory even when the agent never calls `capture`? **`midas init --claude-hook`** installs a +Claude Code SessionEnd hook that offers each session's user turns to memory — Midas's no-LLM policy +still decides what is actually kept.
Manual setup — any client, or to customize (click to expand) @@ -209,7 +213,8 @@ audit), `stats`, `forget` (chain-safe), `forget_matching` (topic-level erasure, `MIDAS_MCP_MAX_RECORDS` · `MIDAS_MCP_MIN_IMPORTANCE` · `MIDAS_MCP_NAMESPACE` (`=auto` → per-project scope) · `MIDAS_MCP_ANN=1` (sub-linear IVF for huge stores) · `MIDAS_MCP_SUPERSEDE` · `MIDAS_MCP_NLI=1` (NLI-gated revision) · `MIDAS_MCP_AUTO_MAINTAIN=` (idle-time upkeep) · `MIDAS_MCP_PINNED` (pin standing directives) · -`MIDAS_MCP_TTL` (per-kind retention, e.g. `chat=30,note=90`) · `MIDAS_MCP_TOKEN` (HTTP bearer auth). +`MIDAS_MCP_TTL` (per-kind retention, e.g. `chat=30,note=90`) · `MIDAS_MCP_TOKEN` (HTTP bearer auth) · +`MIDAS_MCP_KEY` (SQLCipher encryption at rest — `pip install "midas-memory[encrypted]"`).
@@ -338,5 +343,7 @@ python -m eval.continuity # Local-first: every memory lives in a SQLite file on your machine, recall returns the exact stored text, and capture/recall/forget make **no network calls**. No account, API key, or telemetry. The only outbound -traffic is a one-time embedding-model download (for the `local` backend) and the package install. Full -details in [`PRIVACY.md`](PRIVACY.md) · [Apache-2.0](LICENSE). +traffic is a one-time embedding-model download (for the `local` backend) and the package install. +Optional **encryption at rest**: set `MIDAS_MCP_KEY` with the `[encrypted]` extra and the store is a +SQLCipher database — unreadable without the key (and Midas fails closed rather than silently writing +plaintext). Full details in [`PRIVACY.md`](PRIVACY.md) · [Apache-2.0](LICENSE). diff --git a/midas/__init__.py b/midas/__init__.py index 847b960..5ac6d4d 100755 --- a/midas/__init__.py +++ b/midas/__init__.py @@ -116,4 +116,4 @@ "IVFIndex", "IVFStore", ] -__version__ = "0.1.1" +__version__ = "0.2.0" diff --git a/midas/cli.py b/midas/cli.py index 0f95bd8..ee48535 100644 --- a/midas/cli.py +++ b/midas/cli.py @@ -234,6 +234,12 @@ def already_wired(path: Path) -> bool: record(name, cfg_path, detected=True, wired=already_wired(cfg_path) if dry else ok, changed=ok and not dry, reason=msg if (dry or not ok) else None) + if getattr(args, "claude_hook", False): + from midas.hooks import install_claude_hook + + msg = install_claude_hook(Path.home() / ".claude" / "settings.json", dry=dry) + results.append(("Claude Code hook", msg)) + # Clients configured by a JSON file (only the ones already present, unless --all): for name, client_id, path, key, shape in _json_clients(): if not (path.exists() or args.all): @@ -480,6 +486,26 @@ def check(good: bool, label: str, hint: str = "") -> None: return 0 if ok else 1 +# ---- hook (Claude Code SessionEnd auto-capture) ------------------------------------------------- + +def cmd_hook(args: argparse.Namespace) -> int: + """Entry point Claude Code's SessionEnd hook runs: read the hook payload from stdin, offer the + session's user turns to memory (the capture policy decides what's kept). ALWAYS exits 0 — a + memory hook must never break the user's session.""" + from midas.hooks import capture_session + + if args.action != "capture-session": + print(f"unknown hook action '{args.action}'", file=sys.stderr) + return 0 + try: + payload = json.loads(sys.stdin.read() or "{}") + except Exception: + payload = {} + result = capture_session(payload, db=args.db) + print(json.dumps(result), file=sys.stderr) # visible in hook logs, invisible to the agent + return 0 + + # ---- audit ------------------------------------------------------------------------------------ def cmd_audit(args: argparse.Namespace) -> int: @@ -586,6 +612,16 @@ def _import_rules(args: argparse.Namespace, source: str) -> int: "source": f"import:{path.name}", "metadata": {**(d.get("metadata") or {}), "imported_from": path.name}, }) + elif source in ("mem0", "zep"): + from midas.importers import parse_mem0_export, parse_zep_export + + try: + data = json.loads(text) + except Exception: + print(f"{path} is not valid JSON (expected a {source} export)", file=sys.stderr) + return 1 + parse = parse_mem0_export if source == "mem0" else parse_zep_export + dicts = parse(data, source_file=path.name) else: items = parse_markdown_rules(text) dicts = to_record_dicts(items, source_file=path.name, @@ -607,7 +643,7 @@ def _import_rules(args: argparse.Namespace, source: str) -> int: added += 1 note = f" ({skipped} already present, skipped)" if skipped else "" print(f"imported {added} rules from {path} into {db}{note}") - if source != "jsonl" and not getattr(args, "confirmed", False): + if source in ("claude-md", "cursorrules") and not getattr(args, "confirmed", False): print(" provenance=observation — imported rules inform recall/planning but cannot authorize " "guarded actions (re-run with --confirmed to vouch for them).") return 0 @@ -652,6 +688,11 @@ def cmd_uninstall(args: argparse.Namespace) -> int: r = _cli_remove("codex", ["codex", "mcp", "remove", "midas"]) if r: results.append(("Codex", r)) + from midas.hooks import uninstall_claude_hook + + r = uninstall_claude_hook(Path.home() / ".claude" / "settings.json") + if r != "not installed": + results.append(("Claude Code hook", r)) for name, _cid, path, key, _shape in _json_clients(): if not path.exists(): continue @@ -762,6 +803,9 @@ def main() -> None: help="memory auto-separates per project (git repo / cwd) instead of one shared pool") pi.add_argument("--json", action="store_true", help="print a machine-readable client wiring receipt instead of text") + pi.add_argument("--claude-hook", action="store_true", + help="also install a Claude Code SessionEnd hook that auto-captures each " + "session's user turns (the capture policy decides what's kept)") pi.set_defaults(func=cmd_init) ps = sub.add_parser("serve", help="run the MCP server (stdio, or --http for an MCP URL)") @@ -798,6 +842,12 @@ def main() -> None: help="print a machine-readable diagnosis instead of text") pd.set_defaults(func=cmd_doctor) + ph = sub.add_parser("hook", help="hook entry points (used by `midas init --claude-hook`)") + ph.add_argument("action", help="capture-session: read a SessionEnd payload from stdin and " + "offer the session's user turns to memory") + ph.add_argument("--db") + ph.set_defaults(func=cmd_hook) + pa = sub.add_parser("audit", help="show/verify the tamper-evident mutation log (hash chain)") pa.add_argument("--db") pa.add_argument("--limit", type=int, default=10, help="how many recent entries to show") @@ -813,9 +863,10 @@ def main() -> None: pm = sub.add_parser("import", help="import memory (midas export JSON, CLAUDE.md, .cursorrules…)") pm.add_argument("file", help="the file to import") pm.add_argument("--from", dest="source", default="midas", - choices=("midas", "claude-md", "cursorrules", "jsonl"), + choices=("midas", "claude-md", "cursorrules", "jsonl", "mem0", "zep"), help="what the file is: a midas export (default), a rules markdown " - "(CLAUDE.md/AGENTS.md), a .cursorrules, or generic JSONL records") + "(CLAUDE.md/AGENTS.md), a .cursorrules, generic JSONL records, or a " + "Mem0/Zep JSON export") pm.add_argument("--project", help="tag imported rules with this project scope") pm.add_argument("--confirmed", action="store_true", help="stamp imported rules as user-confirmed (they may then authorize " diff --git a/midas/hooks.py b/midas/hooks.py new file mode 100644 index 0000000..3e1d8d5 --- /dev/null +++ b/midas/hooks.py @@ -0,0 +1,138 @@ +"""Claude Code session hook — "install Midas and it starts remembering", even when the agent never +calls `capture`. + +`midas init --claude-hook` registers a SessionEnd hook in `~/.claude/settings.json` that runs +`midas hook capture-session`. When a Claude Code session ends, the hook receives the session payload +on stdin (session id, transcript path, cwd), reads the transcript JSONL, and OFFERS each user turn to +`Memory.capture` — Midas's no-LLM policy then decides what is actually kept (importance floor, dedup), +exactly as if the agent had called `capture` itself. User turns only in v1: they carry the durable +preferences/facts/decisions; assistant prose is mostly restatement. + +Fail-open by design: a hook must never break the user's session, so every error path returns a result +instead of raising, and the CLI always exits 0. +""" +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + + +def _text_of(content: Any) -> str: + """A message's text: transcripts store content as a plain string or a list of typed blocks.""" + if isinstance(content, str): + return content.strip() + if isinstance(content, list): + return " ".join( + b.get("text", "").strip() for b in content + if isinstance(b, dict) and b.get("type") == "text" + ).strip() + return "" + + +def user_turns(transcript_path: str | Path) -> list[str]: + """The user-authored texts in a Claude Code transcript (JSONL), in order. Tool results, meta + lines, and unparseable rows are skipped — a hook must tolerate format drift.""" + turns: list[str] = [] + try: + lines = Path(transcript_path).read_text().splitlines() + except Exception: + return [] + for line in lines: + try: + row = json.loads(line) + except Exception: + continue + if not isinstance(row, dict) or row.get("type") != "user" or row.get("isMeta"): + continue + message = row.get("message") or {} + if message.get("role") != "user": + continue + text = _text_of(message.get("content")) + if text and not text.startswith("<"): # skip tool_result / system-reminder style payloads + turns.append(text) + return turns + + +def capture_session(payload: dict, *, db: str | None = None) -> dict: + """Offer a finished session's user turns to memory. `payload` is the hook's stdin JSON + (session_id, transcript_path, cwd). Returns {offered, stored, skipped} — never raises.""" + try: + from midas import Memory + from midas.sqlite_store import SQLiteStore + + transcript = payload.get("transcript_path") or "" + turns = user_turns(transcript) + if not turns: + return {"offered": 0, "stored": 0, "skipped": 0} + + import os + path = db or os.getenv("MIDAS_MCP_DB") or str(Path.home() / ".midas" / "memory.sqlite3") + mem = Memory(store=SQLiteStore(path)) + session = payload.get("session_id") or "session" + cwd = payload.get("cwd") or "" + stored = 0 + for text in turns: + result = mem.capture( + text, kind="chat", actor="claude-code", + source=f"hook:claude-code:{session}", + metadata={"session": session} | ({"origin": {"cwd": cwd}} if cwd else {}), + ) + stored += bool(result.stored) + return {"offered": len(turns), "stored": stored, "skipped": len(turns) - stored} + except Exception as exc: # fail-open: a memory hook must never break the session + return {"offered": 0, "stored": 0, "skipped": 0, "error": str(exc)[:200]} + + +# The hook block `midas init --claude-hook` merges into ~/.claude/settings.json. +HOOK_COMMAND = "midas hook capture-session" + + +def hook_settings_block() -> dict: + return {"type": "command", "command": HOOK_COMMAND} + + +def install_claude_hook(settings_path: Path, *, dry: bool = False) -> str: + """Register the SessionEnd hook in Claude Code's settings.json, non-destructively (merge + + backup; idempotent — re-running never duplicates the entry).""" + import shutil + + cfg: dict = {} + if settings_path.exists(): + try: + cfg = json.loads(settings_path.read_text() or "{}") + except Exception: + return f"⚠ {settings_path} is not valid JSON — add the hook by hand" + matchers = cfg.setdefault("hooks", {}).setdefault("SessionEnd", []) + for m in matchers: + if any(h.get("command") == HOOK_COMMAND for h in m.get("hooks", [])): + return "already installed" + if dry: + return f"would add SessionEnd hook → {settings_path}" + if settings_path.exists(): + shutil.copy(settings_path, str(settings_path) + ".midas-bak") + matchers.append({"hooks": [hook_settings_block()]}) + settings_path.parent.mkdir(parents=True, exist_ok=True) + settings_path.write_text(json.dumps(cfg, indent=2) + "\n") + return f"SessionEnd auto-capture hook added → {settings_path}" + + +def uninstall_claude_hook(settings_path: Path) -> str: + """Remove the SessionEnd hook (for `midas uninstall`).""" + if not settings_path.exists(): + return "not installed" + try: + cfg = json.loads(settings_path.read_text() or "{}") + except Exception: + return "left as-is (unparseable)" + matchers = cfg.get("hooks", {}).get("SessionEnd", []) + kept = [m for m in matchers + if not any(h.get("command") == HOOK_COMMAND for h in m.get("hooks", []))] + if len(kept) == len(matchers): + return "not installed" + import shutil + + shutil.copy(settings_path, str(settings_path) + ".midas-bak") + cfg["hooks"]["SessionEnd"] = kept + settings_path.write_text(json.dumps(cfg, indent=2) + "\n") + return "removed" diff --git a/midas/importers.py b/midas/importers.py index 03fb868..80236f7 100644 --- a/midas/importers.py +++ b/midas/importers.py @@ -100,3 +100,64 @@ def to_record_dicts( "metadata": meta, }) return out + + +def _mem0_items(data: object) -> list[dict]: + """Mem0 exports arrive as a list of memory objects, or wrapped in {"results": […]} / + {"memories": […]}. The distilled text lives in "memory" (older exports: "text").""" + if isinstance(data, dict): + data = data.get("results") or data.get("memories") or [] + return [d for d in data if isinstance(d, dict)] if isinstance(data, list) else [] + + +def parse_mem0_export(data: object, *, source_file: str) -> list[dict]: + """Mem0 → Midas: each distilled memory becomes a fact (importance 3), keeping the original id, + user/agent attribution, and timestamps in metadata so nothing silently loses its origin.""" + out: list[dict] = [] + for d in _mem0_items(data): + content = (d.get("memory") or d.get("text") or "").strip() + if not content: + continue + meta = {"imported_from": source_file, "mem0_id": d.get("id")} + for k in ("user_id", "agent_id", "run_id"): + if d.get(k): + meta[k] = d[k] + if isinstance(d.get("metadata"), dict): + meta.update({k: v for k, v in d["metadata"].items() if k not in meta}) + out.append({"content": content, "kind": "fact", "importance": 3, + "provenance": "observation", "source": f"import:{source_file}", + "actor": d.get("agent_id") or None, "metadata": meta}) + return out + + +def parse_zep_export(data: object, *, source_file: str) -> list[dict]: + """Zep → Midas: facts (graph edges / session facts) become facts; raw messages become chat turns. + Accepts {"facts": […]}, {"edges": […]}, {"messages": […]}, or a bare list of either shape.""" + facts: list[dict] = [] + messages: list[dict] = [] + if isinstance(data, dict): + facts = [d for d in (data.get("facts") or data.get("edges") or []) if isinstance(d, dict)] + messages = [d for d in (data.get("messages") or []) if isinstance(d, dict)] + elif isinstance(data, list): + for d in data: + if not isinstance(d, dict): + continue + (facts if ("fact" in d) else messages).append(d) + + out: list[dict] = [] + for d in facts: + content = (d.get("fact") or "").strip() + if not content: + continue + out.append({"content": content, "kind": "fact", "importance": 3, + "provenance": "observation", "source": f"import:{source_file}", + "metadata": {"imported_from": source_file, "zep_uuid": d.get("uuid")}}) + for d in messages: + content = (d.get("content") or "").strip() + if not content: + continue + out.append({"content": content, "kind": "chat", "importance": 2, + "provenance": "observation", "source": f"import:{source_file}", + "actor": d.get("role") or None, + "metadata": {"imported_from": source_file, "zep_uuid": d.get("uuid")}}) + return out diff --git a/midas/mcp_server.py b/midas/mcp_server.py index 50b5195..4ed8c1a 100755 --- a/midas/mcp_server.py +++ b/midas/mcp_server.py @@ -37,6 +37,8 @@ supersession chains never expire (default: none) MIDAS_MCP_TOKEN = require `Authorization: Bearer ` on the HTTP transport — without it any local process can read/write the shared store (default: off) + MIDAS_MCP_KEY = encrypt the store at rest with SQLCipher (`pip install + "midas-memory[encrypted]"`); fails closed if the extra is missing """ from __future__ import annotations diff --git a/midas/sqlite_store.py b/midas/sqlite_store.py index 710a802..3fd4804 100755 --- a/midas/sqlite_store.py +++ b/midas/sqlite_store.py @@ -25,6 +25,7 @@ import hashlib import json +import os import sqlite3 import struct import threading @@ -63,7 +64,7 @@ class SQLiteStore(InMemoryStore): def __init__( self, path: str | Path, *, ann_threshold: int | None = None, ann_nprobe: int = 16, - audit: bool = True, + audit: bool = True, key: str | None = None, ) -> None: super().__init__(ann_threshold=ann_threshold, ann_nprobe=ann_nprobe) self._audit_enabled = audit @@ -73,7 +74,22 @@ def __init__( # One connection shared across threads, guarded by the lock below: MCP servers execute # sync tools in worker threads, so the default same-thread check would crash mid-session. self._lock = threading.RLock() - self._conn = sqlite3.connect(str(self._path), check_same_thread=False) + # Encryption at rest (opt-in): with `key` (or MIDAS_MCP_KEY), the file is a SQLCipher + # database — unreadable without the key. FAILS CLOSED: if the user asked for encryption and + # SQLCipher isn't installed, refuse rather than silently writing plaintext. + key = key if key is not None else (os.getenv("MIDAS_MCP_KEY") or None) + if key: + try: + from sqlcipher3 import dbapi2 as _sqlcipher + except ImportError as exc: # pragma: no cover - exercised only without the extra + raise RuntimeError( + 'MIDAS_MCP_KEY is set but SQLCipher is not installed — ' + 'pip install "midas-memory[encrypted]"' + ) from exc + self._conn = _sqlcipher.connect(str(self._path), check_same_thread=False) + self._conn.execute("PRAGMA key = '{}'".format(key.replace("'", "''"))) + else: + self._conn = sqlite3.connect(str(self._path), check_same_thread=False) self._conn.execute("PRAGMA journal_mode=WAL") self._conn.execute( """ diff --git a/packages/midas-ts/package-lock.json b/packages/midas-ts/package-lock.json index 915b6e6..0e572f5 100644 --- a/packages/midas-ts/package-lock.json +++ b/packages/midas-ts/package-lock.json @@ -1,12 +1,12 @@ { "name": "midas-memory-mcp", - "version": "0.1.1", + "version": "0.2.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "midas-memory-mcp", - "version": "0.1.1", + "version": "0.2.0", "license": "Apache-2.0", "dependencies": { "@modelcontextprotocol/sdk": "^1.0.0", diff --git a/packages/midas-ts/package.json b/packages/midas-ts/package.json index 0ce24c3..caf4347 100644 --- a/packages/midas-ts/package.json +++ b/packages/midas-ts/package.json @@ -1,6 +1,6 @@ { "name": "midas-memory-mcp", - "version": "0.1.1", + "version": "0.2.0", "description": "Midas — local-first, source-traceable agent memory over MCP (TypeScript port, experimental). No LLM at ingest or query.", "license": "Apache-2.0", "type": "module", diff --git a/packages/midas-ts/src/mcp.ts b/packages/midas-ts/src/mcp.ts index 1228bf7..1c19584 100644 --- a/packages/midas-ts/src/mcp.ts +++ b/packages/midas-ts/src/mcp.ts @@ -112,7 +112,7 @@ function text(data: string) { export function createServer(): McpServer { const server = new McpServer( - { name: "midas-memory", version: "0.0.4" }, // keep in sync with package.json on each release + { name: "midas-memory", version: "0.2.0" }, // keep in sync with package.json on each release { instructions: AGENT_MEMORY_INSTRUCTIONS }, ); diff --git a/pyproject.toml b/pyproject.toml index 022ec76..1c39718 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "midas-memory" -version = "0.1.1" +version = "0.2.0" description = "Local-first, eval-first memory for long-horizon AI agents — no LLM at ingest" readme = "README.md" license = "Apache-2.0" @@ -23,6 +23,7 @@ local = ["fastembed>=0.7"] turbovec = ["turbovec>=0.8"] model2vec = ["model2vec>=0.8"] mcp = ["mcp>=1.0"] +encrypted = ["sqlcipher3-binary>=0.5"] langgraph = ["langgraph>=0.2"] openai = ["openai>=1.0"] dev = ["pytest>=8", "tiktoken"] diff --git a/tests/test_encrypted_store.py b/tests/test_encrypted_store.py new file mode 100644 index 0000000..0f23b97 --- /dev/null +++ b/tests/test_encrypted_store.py @@ -0,0 +1,44 @@ +"""Encryption at rest (opt-in SQLCipher): the file is unreadable without the key, everything else — +search, audit chain, persistence — behaves identically. Skipped when the `encrypted` extra is absent.""" +from __future__ import annotations + +import sqlite3 + +import pytest + +pytest.importorskip("sqlcipher3") + +from midas import HashingEmbedder, Memory +from midas.sqlite_store import SQLiteStore + + +def test_encrypted_store_roundtrip_and_ciphertext(tmp_path) -> None: + db = str(tmp_path / "m.sqlite3") + mem = Memory(store=SQLiteStore(db, key="s3cret"), embedder=HashingEmbedder()) + mem.remember("The primary database is PostgreSQL.", kind="constraint", importance=5) + + # plaintext never touches disk + raw = (tmp_path / "m.sqlite3").read_bytes() + assert b"PostgreSQL" not in raw and not raw.startswith(b"SQLite format 3") + + # right key: reopen and recall; audit chain intact + again = SQLiteStore(db, key="s3cret") + assert any("PostgreSQL" in r.content for r in again.all()) + assert again.verify_audit_log()["ok"] + + # plain sqlite (no key) cannot read the file + with pytest.raises(sqlite3.DatabaseError): + sqlite3.connect(db).execute("SELECT count(*) FROM memories").fetchone() + + # wrong key fails too (surfaced as a database error on open) + with pytest.raises(Exception): + SQLiteStore(db, key="wrong") + + +def test_key_from_env(tmp_path, monkeypatch) -> None: + db = str(tmp_path / "env.sqlite3") + monkeypatch.setenv("MIDAS_MCP_KEY", "env-key") + SQLiteStore(db) # picks the key up from the environment + monkeypatch.delenv("MIDAS_MCP_KEY") + with pytest.raises(Exception): + SQLiteStore(db).all() # without the key it is not a readable database diff --git a/tests/test_hooks.py b/tests/test_hooks.py new file mode 100644 index 0000000..1307e8c --- /dev/null +++ b/tests/test_hooks.py @@ -0,0 +1,110 @@ +"""The Claude Code SessionEnd auto-capture hook: transcript parsing, policy-gated capture, +fail-open behaviour, and idempotent install/uninstall into settings.json.""" +from __future__ import annotations + +import argparse +import io +import json + +from midas.hooks import ( + HOOK_COMMAND, + capture_session, + install_claude_hook, + uninstall_claude_hook, + user_turns, +) + + +def _transcript(tmp_path, rows) -> str: + p = tmp_path / "transcript.jsonl" + p.write_text("\n".join(json.dumps(r) for r in rows)) + return str(p) + + +def test_user_turns_parses_both_content_shapes_and_skips_noise(tmp_path) -> None: + path = _transcript(tmp_path, [ + {"type": "user", "message": {"role": "user", "content": "plain string turn"}}, + {"type": "user", "message": {"role": "user", + "content": [{"type": "text", "text": "block turn"}]}}, + {"type": "assistant", "message": {"role": "assistant", "content": "assistant prose"}}, + {"type": "user", "isMeta": True, "message": {"role": "user", "content": "meta noise"}}, + {"type": "user", "message": {"role": "user", + "content": [{"type": "tool_result", "content": "x"}]}}, + {"type": "user", "message": {"role": "user", "content": "injected"}}, + ]) + assert user_turns(path) == ["plain string turn", "block turn"] + assert user_turns(tmp_path / "missing.jsonl") == [] # fail-open: no transcript, no crash + + +def test_capture_session_stores_signal_and_skips_trivia(tmp_path) -> None: + from midas.sqlite_store import SQLiteStore + + db = str(tmp_path / "m.sqlite3") + path = _transcript(tmp_path, [ + {"type": "user", "message": {"role": "user", + "content": "The primary database for Apollo is PostgreSQL 16."}}, + {"type": "user", "message": {"role": "user", "content": "lol ok cool"}}, + ]) + result = capture_session( + {"session_id": "s1", "transcript_path": path, "cwd": "/repo"}, db=db) + assert result["offered"] == 2 and result["stored"] == 1 and result["skipped"] == 1 + recs = SQLiteStore(db).all() + assert len(recs) == 1 and "PostgreSQL" in recs[0].content + assert recs[0].actor == "claude-code" and recs[0].source == "hook:claude-code:s1" + + +def test_capture_session_never_raises() -> None: + out = capture_session({"transcript_path": "/nonexistent/x.jsonl"}, db=":memory:!!bad//path") + assert out["stored"] == 0 # fail-open, whatever went wrong + + +def test_install_and_uninstall_hook_idempotent(tmp_path) -> None: + settings = tmp_path / ".claude" / "settings.json" + settings.parent.mkdir(parents=True) + settings.write_text(json.dumps({"model": "opus", "hooks": {"PreToolUse": []}})) + + assert "would add" in install_claude_hook(settings, dry=True) + assert json.loads(settings.read_text()).get("hooks", {}).get("SessionEnd") is None # dry: untouched + + assert "added" in install_claude_hook(settings) + cfg = json.loads(settings.read_text()) + assert cfg["model"] == "opus" and "PreToolUse" in cfg["hooks"] # merge, never clobber + assert cfg["hooks"]["SessionEnd"][0]["hooks"][0]["command"] == HOOK_COMMAND + + assert install_claude_hook(settings) == "already installed" # idempotent + assert len(json.loads(settings.read_text())["hooks"]["SessionEnd"]) == 1 + + assert uninstall_claude_hook(settings) == "removed" + assert json.loads(settings.read_text())["hooks"]["SessionEnd"] == [] + assert uninstall_claude_hook(settings) == "not installed" + + +def test_cli_hook_command_reads_stdin_and_exits_zero(tmp_path, capsys, monkeypatch) -> None: + from midas import cli + + db = str(tmp_path / "m.sqlite3") + path = _transcript(tmp_path, [ + {"type": "user", "message": {"role": "user", + "content": "My API rate limit is 5000 requests per hour."}}, + ]) + monkeypatch.setattr("sys.stdin", + io.StringIO(json.dumps({"session_id": "s2", "transcript_path": path}))) + rc = cli.cmd_hook(argparse.Namespace(action="capture-session", db=db)) + assert rc == 0 + assert json.loads(capsys.readouterr().err)["stored"] == 1 + + monkeypatch.setattr("sys.stdin", io.StringIO("not json")) + assert cli.cmd_hook(argparse.Namespace(action="capture-session", db=db)) == 0 # fail-open + + +def test_init_claude_hook_flag(tmp_path, capsys, monkeypatch) -> None: + from midas import cli + from tests.test_cli import _sandbox_client_paths + + monkeypatch.setattr(cli.shutil, "which", lambda exe: None) + _sandbox_client_paths(monkeypatch, tmp_path) + rc = cli.cmd_init(argparse.Namespace(db=str(tmp_path / "m.sqlite3"), dry_run=False, all=False, + claude_hook=True)) + assert rc == 0 and "Claude Code hook" in capsys.readouterr().out + cfg = json.loads((tmp_path / ".claude" / "settings.json").read_text()) + assert cfg["hooks"]["SessionEnd"][0]["hooks"][0]["command"] == HOOK_COMMAND diff --git a/tests/test_importers.py b/tests/test_importers.py index c67e16c..925a5f7 100644 --- a/tests/test_importers.py +++ b/tests/test_importers.py @@ -100,3 +100,55 @@ def test_cli_import_empty_file_fails_cleanly(tmp_path, capsys) -> None: project=None, confirmed=False, overwrite=False) assert cli.cmd_import(ns) == 1 assert "nothing importable" in capsys.readouterr().err + + +def test_parse_mem0_export() -> None: + from midas.importers import parse_mem0_export + + data = {"results": [ + {"id": "m1", "memory": "User prefers dark mode.", "user_id": "u1", + "metadata": {"topic": "ui"}}, + {"id": "m2", "text": "Launch is September 14.", "agent_id": "coder"}, + {"id": "m3", "memory": ""}, # empty → skipped + ]} + out = parse_mem0_export(data, source_file="mem0.json") + assert [d["content"] for d in out] == ["User prefers dark mode.", "Launch is September 14."] + assert out[0]["metadata"]["mem0_id"] == "m1" and out[0]["metadata"]["user_id"] == "u1" + assert out[0]["metadata"]["topic"] == "ui" + assert out[1]["actor"] == "coder" + assert all(d["kind"] == "fact" and d["provenance"] == "observation" for d in out) + # bare-list shape works too + assert len(parse_mem0_export(data["results"], source_file="x")) == 2 + + +def test_parse_zep_export() -> None: + from midas.importers import parse_zep_export + + data = {"facts": [{"uuid": "f1", "fact": "The primary database is PostgreSQL."}], + "messages": [{"uuid": "z1", "role": "user", "content": "hola equipo"}]} + out = parse_zep_export(data, source_file="zep.json") + assert out[0]["kind"] == "fact" and out[0]["metadata"]["zep_uuid"] == "f1" + assert out[1]["kind"] == "chat" and out[1]["actor"] == "user" + # bare list: rows with a "fact" key are facts, the rest messages + mixed = parse_zep_export([{"fact": "a stable fact here"}, {"content": "a chat line"}], + source_file="x") + assert [d["kind"] for d in mixed] == ["fact", "chat"] + + +def test_cli_import_mem0(tmp_path, capsys) -> None: + from midas.sqlite_store import SQLiteStore + + f = tmp_path / "mem0.json" + f.write_text(json.dumps([{"id": "m1", "memory": "User prefers dark mode."}])) + ns = argparse.Namespace(file=str(f), source="mem0", db=str(tmp_path / "m.sqlite3"), + project=None, confirmed=False, overwrite=False) + assert cli.cmd_import(ns) == 0 + recs = SQLiteStore(str(tmp_path / "m.sqlite3")).all() + assert len(recs) == 1 and recs[0].metadata["mem0_id"] == "m1" + + bad = tmp_path / "bad.json" + bad.write_text("not json") + ns2 = argparse.Namespace(file=str(bad), source="zep", db=str(tmp_path / "m.sqlite3"), + project=None, confirmed=False, overwrite=False) + assert cli.cmd_import(ns2) == 1 + assert "not valid JSON" in capsys.readouterr().err From 6dd523f1ba08d0c89f378c2fc532764c4fdb81ad Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 6 Jul 2026 04:19:06 +0000 Subject: [PATCH 09/10] Redesign midas inspect: light+dark themes, real charts, fixed color system MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The inspector had one problem underneath its polish: every kind, every provenance, every signal rendered in the same monochrome gold accent, so a glass-box tool for governed memory couldn't be visually parsed at a glance. And it was dark-only, had no real charts, used native confirm()/prompt(), and its hash router never listened for hashchange (back/forward and direct links silently did nothing). - Fixed categorical color system: kind and provenance each get a consistent, validated color (colorblind-safe contrast/separation checked in both themes) used identically across every view — Overview bars, Browse tags, Project governance, Conflicts. Status color (good/warning/ critical) reserved separately for verdicts/chain-integrity/conflicts, so a revision reads as "normal lifecycle" (amber) not "error" (red). - A real second theme: light, deliberately designed (not an inverted dark), toggle + system-preference default + persisted choice. - Real charts: a 30-day activity line/area chart with a hover crosshair + tooltip, and a recency (short/medium/long) stacked bar — backed by two new pure API functions, api_timeseries and api_meta (plus by_tier added to api_overview). - Native confirm()/prompt() replaced with in-app modal + toast components; grouped nav (Memory/Coding/Governance) with live badge counts; a command palette (Cmd/Ctrl-K) and "/" search shortcut; responsive down to phone width with a horizontal pill nav. - Fixed the hashchange bug (the router only ran once, at load) and two real API-shape bugs the fixes surfaced under interaction testing: api_conflicts/api_loops return bare lists (the code assumed a {count,...} envelope) and the forget receipt's key is content_sha256 (the code read content_sha). Verified with Playwright end-to-end (not just screenshots): forget with modal confirm, close-loop with a resolution textarea, theme toggle, command-palette keyboard nav, "/" focus, and browser back/forward through the hash router — all against a seeded demo store, in both themes and a mobile viewport, zero console/page errors. 336 tests pass, ruff clean. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- CHANGELOG.md | 16 + README.md | 11 +- midas/inspector.py | 866 +++++++++++++++++++++++++++++++--------- tests/test_inspector.py | 46 +++ 4 files changed, 741 insertions(+), 198 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index aa806f0..9f054ab 100755 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,22 @@ Notable changes to Midas. Pre-1.0 — the API may change. Format loosely follows [Keep a Changelog](https://keepachangelog.com/). +## [Unreleased] + +### Changed +- **`midas inspect` redesigned end to end.** A real light theme (not an inverted dark one) alongside the + existing dark brand theme, toggle + system-preference default + persisted choice; a fixed, validated + categorical color system so every `kind` and `provenance` value keeps the *same* color everywhere in + the app (Overview bars, Browse tags, Project governance) instead of one monochrome accent for + everything; a real 30-day activity chart (SVG line + area, hover crosshair/tooltip) and a recency + (short/medium/long) stacked bar on Overview; grouped, badge-annotated navigation (Memory / Coding / + Governance) with a command palette (**⌘K**) and a **/** search shortcut; native `confirm()`/`prompt()` + replaced with in-app modal/toast components; a responsive layout down to phone width with a horizontal + pill nav. New endpoints backing it: `api_timeseries` (daily capture counts), `api_meta` (version/db/ + embedder/audit-chain identity), and `by_tier` added to `api_overview`. Fixed a real routing bug in the + process: the old build never listened for `hashchange`, so browser back/forward and direct links to a + view silently did nothing. + ## [0.2.0] — 2026-07-05 ### Added diff --git a/README.md b/README.md index 9cc97ba..fc1ddd1 100755 --- a/README.md +++ b/README.md @@ -280,10 +280,19 @@ midas inspect --db ~/.midas/memory.sqlite3 # opens http://localhost:7777 # before install: python -m midas.inspector --db --embedder hashing ``` -- **Browse + search** every memory (verbatim, with provenance + source). +- **Overview** — counts, attributability, a 30-day activity chart, and kind/provenance/recency breakdowns, + each kind and provenance color-coded *consistently across every view* (a fixed categorical palette, + validated for colorblind-safe contrast in both themes — never color-only, every value keeps its label). +- **Browse + search** every memory (verbatim, with provenance + source), filterable by kind, provenance, + and sort order. - **Belief history + time-travel** — what you believed, what it superseded, and when. - **Project state** (decisions / bugs / forbidden) and **what changed** since a date. - **Governance** — would memory authorize an action, and why (the audit trail); **forget** with a receipt. +- **Conflicts** and **Open loops** — the same control-plane views from `memory_conflicts`/`open_loops`, + with one-click resolve/close from the UI. +- **Audit log** — the hash chain's verification status and its most recent entries. +- Light + dark themes (a real second theme, not an inverted dark one), keyboard shortcuts (**⌘K** to jump + anywhere or search, **/** to focus search), and a responsive layout down to phone width. And every mutation (write / revise / forget) appends to a **tamper-evident, hash-chained audit log** inside the store — hashes only, never content. `midas audit` shows it; `midas audit --json` verifies diff --git a/midas/inspector.py b/midas/inspector.py index 845c13a..3b4b234 100644 --- a/midas/inspector.py +++ b/midas/inspector.py @@ -155,17 +155,19 @@ def api_audit_chain(mem: "Memory", *, limit: int = 50) -> dict[str, Any]: def api_overview(mem: "Memory") -> dict[str, Any]: """The memory-health dashboard a team/enterprise needs at a glance: counts, attributability (the compliance metric: fraction with both a source and an actor), revision activity, recency, and the - distribution by kind / provenance / project. All computed — no fabricated governance counters.""" + distribution by kind / provenance / project / recency tier. All computed — no fabricated counters.""" recs = list(mem.store.all()) total = len(recs) now = time.time() by_kind: dict[str, int] = {} by_prov: dict[str, int] = {} + by_tier: dict[str, int] = {"short": 0, "medium": 0, "long": 0} projects: dict[str, int] = {} imp_sum = added_24h = added_7d = 0 for r in recs: by_kind[r.kind] = by_kind.get(r.kind, 0) + 1 by_prov[r.provenance] = by_prov.get(r.provenance, 0) + 1 + by_tier[mem.tier(r, now=now)] += 1 imp_sum += r.importance proj = (r.metadata or {}).get("project") if proj: @@ -186,268 +188,732 @@ def api_overview(mem: "Memory") -> dict[str, Any]: "added_7d": added_7d, "by_kind": rank(by_kind), "by_provenance": rank(by_prov), + "by_tier": by_tier, "projects": rank(projects), } +def api_timeseries(mem: "Memory", *, days: int = 30) -> dict[str, Any]: + """Daily capture volume for the last `days` days (UTC-bucketed) — the activity trend a static + count can't show. Zero-filled so gaps in captures show as true zeros, not missing days.""" + import datetime as _dt + + days = max(1, int(days)) + today = _dt.datetime.now(tz=_dt.timezone.utc).date() + buckets: dict[str, int] = { + (today - _dt.timedelta(days=i)).isoformat(): 0 for i in range(days - 1, -1, -1) + } + cutoff = time.time() - days * 86_400.0 + for r in mem.store.all(): + if r.created_at < cutoff: + continue + day = _dt.datetime.fromtimestamp(r.created_at, tz=_dt.timezone.utc).date().isoformat() + if day in buckets: + buckets[day] += 1 + return {"days": days, "buckets": [{"date": d, "count": c} for d, c in buckets.items()]} + + +def api_meta(mem: "Memory", *, db: str, embedder_name: str) -> dict[str, Any]: + """Which store/version/embedder this Inspector instance is actually looking at — so a user + juggling several stores (or a bug report) doesn't have to guess. No memory content.""" + from midas import __version__ + + store = mem.store + return { + "version": __version__, + "db": db, + "embedder": embedder_name, + "records": len(store.all()), + "schema_version": store.schema_version() if hasattr(store, "schema_version") else None, + "audit_chain": hasattr(store, "verify_audit_log"), + } + + # --- The embedded UI --------------------------------------------------------------------------- -INDEX_HTML = r""" -Midas Inspector +.conflict-head{display:flex;align-items:center;gap:9px;margin-bottom:4px;flex-wrap:wrap} +@media(max-width:760px){.cpair{grid-template-columns:1fr}.app{grid-template-columns:1fr} +.side{position:static;height:auto;flex-direction:column}.side>.nav,.side>.navlbl{display:none} +.side{padding:14px}.brand{padding-bottom:10px} +.mobnav{display:flex!important} +.foot{display:none}.main{padding:22px 18px}.vhead{flex-direction:column;gap:8px}.vhead .side-stat{text-align:left}} +.mobnav{display:none;gap:6px;overflow-x:auto;padding-bottom:4px;-webkit-overflow-scrolling:touch} +.mobnav a{flex-shrink:0;display:flex;align-items:center;gap:6px;padding:7px 12px;border-radius:999px;border:1px solid var(--line2); +color:var(--steel);font-size:12.5px;white-space:nowrap;cursor:pointer} +.mobnav a.on{color:var(--gold);border-color:var(--gline);background:var(--gsoft)} +.mobnav svg{width:14px;height:14px} +/* chart: sparkline */ +.chartwrap{position:relative} +.spark{width:100%;height:120px;display:block;overflow:visible} +.spark .fill{fill:url(#sparkgrad)} +.spark .line{fill:none;stroke:var(--gold);stroke-width:2;stroke-linecap:round;stroke-linejoin:round} +.spark .hit{fill:transparent} +.spark .cursor{stroke:var(--line2);stroke-width:1} +.spark .dot{fill:var(--gold);stroke:var(--bg);stroke-width:2} +.charttip{position:absolute;pointer-events:none;background:var(--surf2);border:1px solid var(--line2);border-radius:9px; +padding:6px 10px;font-size:11.5px;box-shadow:var(--shadow);white-space:nowrap;opacity:0;transition:opacity .1s;z-index:5;transform:translate(-50%,-100%)} +.charttip.on{opacity:1} +.charttip b{font-variant-numeric:tabular-nums} +/* toasts */ +#toasts{position:fixed;right:18px;bottom:18px;display:flex;flex-direction:column;gap:8px;z-index:200;max-width:340px} +.toast{background:var(--surf2);border:1px solid var(--line2);border-radius:12px;padding:11px 14px;box-shadow:var(--shadow); +font-size:13px;display:flex;align-items:flex-start;gap:10px;animation:pop .22s cubic-bezier(.16,1,.3,1) both} +.toast svg{width:16px;height:16px;flex-shrink:0;margin-top:1px} +.toast.ok svg{color:var(--good)}.toast.err svg{color:var(--critical)} +.toast .tx{flex:1;line-height:1.4}.toast .tx small{display:block;color:var(--steel);font-size:11px;margin-top:2px;font-family:ui-monospace,monospace} +/* modal */ +#modalwrap{position:fixed;inset:0;background:rgba(5,5,10,.55);backdrop-filter:blur(2px);display:none;place-items:center;z-index:150;padding:20px} +:root[data-theme="light"] #modalwrap{background:rgba(30,28,20,.35)} +#modalwrap.on{display:grid} +.modal{background:var(--plane);border:1px solid var(--line2);border-radius:18px;padding:22px 24px;width:100%;max-width:440px; +box-shadow:0 20px 60px rgba(0,0,0,.4);animation:pop .2s cubic-bezier(.16,1,.3,1) both} +.modal h3{margin:0 0 8px;font-size:16.5px} +.modal p{margin:0 0 16px;color:var(--steel);font-size:13.5px;line-height:1.5} +.modal textarea{width:100%;min-height:80px;resize:vertical;margin-bottom:14px} +.modal .mbtns{display:flex;justify-content:flex-end;gap:9px} +/* command palette */ +#palwrap{position:fixed;inset:0;background:rgba(5,5,10,.5);display:none;place-items:flex-start;justify-content:center;z-index:160;padding-top:14vh} +:root[data-theme="light"] #palwrap{background:rgba(30,28,20,.32)} +#palwrap.on{display:grid} +.pal{width:100%;max-width:520px;background:var(--plane);border:1px solid var(--line2);border-radius:16px;box-shadow:0 24px 70px rgba(0,0,0,.45); +overflow:hidden;animation:pop .16s cubic-bezier(.16,1,.3,1) both} +.pal input{width:100%;border:none;border-radius:0;border-bottom:1px solid var(--line);padding:16px 18px;font-size:15px;background:transparent} +.pal input:focus{box-shadow:none} +.pal ul{list-style:none;margin:0;padding:8px;max-height:340px;overflow-y:auto} +.pal li{display:flex;align-items:center;gap:11px;padding:10px 12px;border-radius:10px;cursor:pointer;font-size:13.5px;color:var(--steel)} +.pal li.sel{background:var(--gsoft);color:var(--gold)} +.pal li svg{width:15px;height:15px;flex-shrink:0} +.pal li .k{margin-left:auto;font-size:10.5px;color:var(--muted)} +.skiplink{position:absolute;left:-9999px;top:0;background:var(--gold);color:#15151f;padding:10px 16px;border-radius:0 0 10px 0;font-weight:600;z-index:300} +.skiplink:focus{left:0} + + +
-
+
+
+
+
+
    """ +

    Trust · by provenance

    ${barList(Object.entries(g.by_provenance),provColor)}
    +

    Contributors · by actor

    ${barList(Object.entries(g.by_actor),null,{labelFmt:actorLabel})}
    ` + +Object.entries(d.state).map(([k,rs])=>`

    ${esc(k.replace(/_/g,' '))} ${rs.length}

    `+rs.map(r=>card(r)).join('')).join('') + +`

    Recent ${d.recent.length}

    `+(d.recent.length?d.recent.map(r=>card(r)).join(''):emptyState('—')); + document.getElementById('backp').onclick=()=>{location.hash='project';}; +} + +// ---- routing (hash-driven; hashchange IS wired, unlike the old build) ---- +function setActive(t){document.querySelectorAll('.side nav a,#mobnav a').forEach(a=>a.classList.toggle('on',a.dataset.t===t));} +function route(){ + const h=location.hash.slice(1); + if(h.startsWith('project=')){tab='project';setActive('project');openProj(decodeURIComponent(h.slice(8)));return;} + const s=V[h]?h:'overview'; + tab=s;setActive(s);V[s](); +} +function navigateTo(t){if(location.hash.slice(1)===t){route();}else{location.hash=t;}} +window.addEventListener('hashchange',route); +document.querySelectorAll('.side nav a[data-t]').forEach(a=>a.onclick=()=>navigateTo(a.dataset.t)); + +function buildMobileNav(){ + const items=[...document.querySelectorAll('.side nav a[data-t]')]; + document.getElementById('mobnav').innerHTML=items.map(a=> + `${a.querySelector('svg').outerHTML}${a.querySelector('.lbl').outerHTML}`).join(''); + document.querySelectorAll('#mobnav a').forEach(a=>a.onclick=()=>navigateTo(a.dataset.t)); +} + +// ---- command palette (Cmd/Ctrl+K) ---- +let palSel=0,palMatches=[]; +function palItems(){return [...document.querySelectorAll('.side nav a[data-t]')].map(a=>({ + type:'nav',t:a.dataset.t,label:a.querySelector('.lbl').textContent,icon:a.querySelector('svg').outerHTML}));} +function openPalette(){ + const wrap=document.getElementById('palwrap'),input=document.getElementById('palinput'); + wrap.classList.add('on');input.value='';input.focus();renderPalette(''); +} +function closePalette(){document.getElementById('palwrap').classList.remove('on');} +function renderPalette(q){ + const items=palItems(),ql=q.trim().toLowerCase(); + palMatches=ql?items.filter(i=>i.label.toLowerCase().includes(ql)):items; + if(ql)palMatches=palMatches.concat([{type:'search',label:`Search memory for "${q}"`,q}]); + palSel=0; + document.getElementById('pallist').innerHTML=palMatches.length?palMatches.map((m,i)=> + `
  • ${m.icon||ICON_SEARCH_SM}${esc(m.label)}${m.type==='nav'?'↵':''}
  • ` + ).join(''):`
  • No matches
  • `; + document.querySelectorAll('#pallist li[data-i]').forEach(li=>li.onclick=()=>choosePalette(+li.dataset.i)); +} +function highlightPal(){document.querySelectorAll('#pallist li[data-i]').forEach(li=>li.classList.toggle('sel',+li.dataset.i===palSel));} +function choosePalette(i){ + const m=palMatches[i]; if(!m)return; closePalette(); + if(m.type==='nav')navigateTo(m.t); + else{navigateTo('browse');setTimeout(()=>{const qi=document.getElementById('q');if(qi){qi.value=m.q;document.getElementById('go').click();}},70);} +} +document.getElementById('palhint').onclick=openPalette; +document.getElementById('palinput').oninput=e=>renderPalette(e.target.value); +document.getElementById('palwrap').onclick=e=>{if(e.target.id==='palwrap')closePalette();}; +document.addEventListener('keydown',e=>{ + const mod=e.metaKey||e.ctrlKey; + if(mod&&e.key.toLowerCase()==='k'){e.preventDefault();openPalette();return;} + const wrap=document.getElementById('palwrap'); + if(wrap.classList.contains('on')){ + if(e.key==='Escape')closePalette(); + else if(e.key==='ArrowDown'){e.preventDefault();palSel=Math.min(palMatches.length-1,palSel+1);highlightPal();} + else if(e.key==='ArrowUp'){e.preventDefault();palSel=Math.max(0,palSel-1);highlightPal();} + else if(e.key==='Enter'){e.preventDefault();choosePalette(palSel);} + return; + } + if(e.key==='/'&&!['INPUT','TEXTAREA'].includes(document.activeElement.tagName)){ + const qi=document.getElementById('q'); if(qi){e.preventDefault();qi.focus();} + } +}); + +// ---- theme toggle ---- +function applyThemeIcon(){ + const t=document.documentElement.dataset.theme; + document.getElementById('themebtn').innerHTML=t==='light' + ?'' + :''; +} +document.getElementById('themebtn').onclick=()=>{ + const next=document.documentElement.dataset.theme==='light'?'dark':'light'; + document.documentElement.dataset.theme=next; + try{localStorage.setItem('midas-theme',next);}catch(e){} + applyThemeIcon(); +}; +applyThemeIcon(); + +// ---- boot ---- +buildMobileNav(); +loadFoot(); +route(); + +""" # --- HTTP layer (thin router) ------------------------------------------------------------------ -def _make_handler(mem: "Memory"): +def _make_handler(mem: "Memory", *, db: str = "", embedder_name: str = ""): class Handler(BaseHTTPRequestHandler): def _send(self, body: bytes, ctype: str, code: int = 200) -> None: self.send_response(code) @@ -490,6 +956,10 @@ def do_GET(self) -> None: # noqa: N802 return self._json(api_loops(mem)) if u.path == "/api/audit_chain": return self._json(api_audit_chain(mem, limit=int(qs.get("limit", 50)))) + if u.path == "/api/timeseries": + return self._json(api_timeseries(mem, days=int(qs.get("days", 30)))) + if u.path == "/api/meta": + return self._json(api_meta(mem, db=db, embedder_name=embedder_name)) self._json({"error": "not found"}, 404) def do_POST(self) -> None: # noqa: N802 @@ -512,11 +982,13 @@ def serve(db: str, *, host: str = "127.0.0.1", port: int = 7777, embedder: str = from .memory import Memory from .sqlite_store import SQLiteStore + embedder_name = "hashing" if embedder == "local": try: from .embeddings import LocalEmbedder emb = LocalEmbedder() + embedder_name = "local" except Exception: from .embeddings import HashingEmbedder @@ -527,7 +999,7 @@ def serve(db: str, *, host: str = "127.0.0.1", port: int = 7777, embedder: str = emb = HashingEmbedder() mem = Memory(store=SQLiteStore(db), embedder=emb) - httpd = ThreadingHTTPServer((host, port), _make_handler(mem)) + httpd = ThreadingHTTPServer((host, port), _make_handler(mem, db=db, embedder_name=embedder_name)) url = f"http://{host}:{port}" print(f"Midas Inspector — {db}\n → {url} (local only, zero egress; Ctrl-C to stop)") if open_browser: diff --git a/tests/test_inspector.py b/tests/test_inspector.py index 7ed6dfe..1e8795f 100644 --- a/tests/test_inspector.py +++ b/tests/test_inspector.py @@ -125,3 +125,49 @@ def test_api_audit_chain(tmp_path) -> None: assert out["tail"][0]["op"] == "put" import json as _json assert "PostgreSQL" not in _json.dumps(out) # hashes only, never content + + +def test_api_overview_includes_recency_tiers() -> None: + from midas.inspector import api_overview + + mem = _mem() + now = time.time() + mem.remember("fresh fact", kind="fact", created_at=now) + mem.remember("a week-old fact", kind="fact", created_at=now - 3 * 86400) + mem.remember("an old fact", kind="fact", created_at=now - 40 * 86400) + o = api_overview(mem) + assert o["by_tier"] == {"short": 1, "medium": 1, "long": 1} + + +def test_api_timeseries_buckets_are_zero_filled() -> None: + from midas.inspector import api_timeseries + + mem = _mem() + now = time.time() + mem.remember("today x2 one", kind="note", created_at=now) + mem.remember("today x2 two", kind="note", created_at=now) + mem.remember("three days ago", kind="note", created_at=now - 3 * 86400) + out = api_timeseries(mem, days=7) + assert out["days"] == 7 and len(out["buckets"]) == 7 + counts = {b["date"]: b["count"] for b in out["buckets"]} + assert sum(counts.values()) == 3 + assert list(counts.values()).count(0) == 5 # the other 5 days are zero-filled, not missing + # buckets are in chronological order (oldest first, today last) + dates = [b["date"] for b in out["buckets"]] + assert dates == sorted(dates) + + +def test_api_meta_reports_store_identity() -> None: + from midas.inspector import api_meta + from midas.sqlite_store import SQLiteStore + + mem = _mem() + out = api_meta(mem, db=":memory:", embedder_name="hashing") + assert out["db"] == ":memory:" and out["embedder"] == "hashing" + assert out["records"] == 0 and out["audit_chain"] is False # InMemoryStore has no chain + + mem2 = Memory(store=SQLiteStore(":memory:"), embedder=HashingEmbedder()) + mem2.remember("a fact", kind="fact") + out2 = api_meta(mem2, db="/tmp/x.sqlite3", embedder_name="local") + assert out2["records"] == 1 and out2["audit_chain"] is True and out2["schema_version"] == 1 + assert "version" in out2 From 5dbe633fac2fb8ed981407c1e1ce28fb3d157b7b Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 6 Jul 2026 17:36:56 +0000 Subject: [PATCH 10/10] Renumber the batch into 1.0.0 (fold [1.1.0] into the single 1.0.0 release) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1.0.0 hasn't been published yet, so instead of shipping this batch as a separate 1.1.0 on top, fold it into the one comprehensive 1.0.0 release. - All own-version references back to 1.0.0 (pyproject, __init__, midas-ts package.json + lock own-version lines, mcp.ts handshake, mcpb manifest, server.json); the lockfile's `forwarded` dependency version is untouched. - CHANGELOG: [1.1.0] folded into a single [1.0.0] — 2026-07-06; the contract paragraph reworded (this release now genuinely adds the control-plane, audit chain, and tooling, not just consolidates), Added/Changed/Security merged with nothing dropped or duplicated. Merged tree: 336 passed, 5 skipped; ruff clean; midas.__version__ == 1.0.0. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01Deray2qcPRy1hZVQnboHp4 --- CHANGELOG.md | 44 +++++++++++++---------------- mcpb/manifest.json | 2 +- midas/__init__.py | 2 +- packages/midas-ts/package-lock.json | 4 +-- packages/midas-ts/package.json | 2 +- packages/midas-ts/src/mcp.ts | 2 +- pyproject.toml | 2 +- server.json | 4 +-- 8 files changed, 29 insertions(+), 33 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e79d1e4..4a295ca 100755 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,7 +8,12 @@ breaking changes only land in a major. Format loosely follows ## [Unreleased] -## [1.1.0] — 2026-07-06 +## [1.0.0] — 2026-07-06 + +The 1.0 contract: the public surface — the `midas.Memory` SDK, the guard, the coding `is_forbidden` +gate, the MCP tools, and the `midas` CLI — is now locked under semver. This release consolidates and +hardens everything since 0.1.1 (the retrieval core, the governance guard, the control-plane, the audit +chain, the client tooling, and the TypeScript port) and freezes the API. ### Added - **The continuity control-plane** (`midas/continuity.py` + MCP tools): **`resume`** — everything an @@ -61,29 +66,6 @@ breaking changes only land in a major. Format loosely follows - **Encryption at rest (opt-in)**: `SQLiteStore(key=…)` / `MIDAS_MCP_KEY` with the new `[encrypted]` extra (SQLCipher). The file on disk is ciphertext; opening without the key fails; if the key is set but the extra is missing Midas **fails closed** instead of silently writing plaintext. - -### Changed -- **`midas inspect` redesigned end to end.** A real light theme (not an inverted dark one) alongside the - existing dark brand theme, toggle + system-preference default + persisted choice; a fixed, validated - categorical color system so every `kind` and `provenance` value keeps the *same* color everywhere in - the app (Overview bars, Browse tags, Project governance) instead of one monochrome accent for - everything; a real 30-day activity chart (SVG line + area, hover crosshair/tooltip) and a recency - (short/medium/long) stacked bar on Overview; grouped, badge-annotated navigation (Memory / Coding / - Governance) with a command palette (**⌘K**) and a **/** search shortcut; native `confirm()`/`prompt()` - replaced with in-app modal/toast components; a responsive layout down to phone width with a horizontal - pill nav. New endpoints backing it: `api_timeseries` (daily capture counts), `api_meta` (version/db/ - embedder/audit-chain identity), and `by_tier` added to `api_overview`. Fixed a real routing bug in the - process: the old build never listened for `hashchange`, so browser back/forward and direct links to a - view silently did nothing. -- The TypeScript `Memory` API is async (`remember`/`capture`/`recall`/`buildContext`/ - `forgetMatching` return promises) to support the optional ONNX embedder. - -## [1.0.0] — 2026-07-05 - -The 1.0 contract: the API that exists today is the API, locked under semver. No new features gate -this release — it consolidates what shipped and hardened since 0.1.1 and freezes the surface. - -### Added - **`midas init --json` / `midas status --json` — a machine-readable client wiring receipt** (#15). One command now yields a compact, pasteable proof of what was wired: per client its config path, detected/wired/changed state, backup path, and skip reason; plus the memory DB, scope mode @@ -101,6 +83,20 @@ this release — it consolidates what shipped and hardened since 0.1.1 and freez recall finding the signal among 1,500 noise records. ### Changed +- **`midas inspect` redesigned end to end.** A real light theme (not an inverted dark one) alongside the + existing dark brand theme, toggle + system-preference default + persisted choice; a fixed, validated + categorical color system so every `kind` and `provenance` value keeps the *same* color everywhere in + the app (Overview bars, Browse tags, Project governance) instead of one monochrome accent for + everything; a real 30-day activity chart (SVG line + area, hover crosshair/tooltip) and a recency + (short/medium/long) stacked bar on Overview; grouped, badge-annotated navigation (Memory / Coding / + Governance) with a command palette (**⌘K**) and a **/** search shortcut; native `confirm()`/`prompt()` + replaced with in-app modal/toast components; a responsive layout down to phone width with a horizontal + pill nav. New endpoints backing it: `api_timeseries` (daily capture counts), `api_meta` (version/db/ + embedder/audit-chain identity), and `by_tier` added to `api_overview`. Fixed a real routing bug in the + process: the old build never listened for `hashchange`, so browser back/forward and direct links to a + view silently did nothing. +- The TypeScript `Memory` API is async (`remember`/`capture`/`recall`/`buildContext`/ + `forgetMatching` return promises) to support the optional ONNX embedder. - **A bare `Memory()` now auto-selects real semantic recall.** With the `[local]` extra installed, `Memory()` upgrades from the offline hashing embedder to `LocalEmbedder` automatically (override with `MIDAS_EMBEDDER=hashing|local`; default `auto`). Fixes the out-of-the-box recall weakness an diff --git a/mcpb/manifest.json b/mcpb/manifest.json index a4b8297..41842b8 100644 --- a/mcpb/manifest.json +++ b/mcpb/manifest.json @@ -2,7 +2,7 @@ "manifest_version": "0.3", "name": "midas-memory", "display_name": "Midas — Local Agent Memory", - "version": "1.1.0", + "version": "1.0.0", "description": "Local-first, source-traceable memory for Claude — no LLM at ingest, fully offline.", "long_description": "Midas gives Claude a persistent, **local** memory. Everything is stored on your own machine in a SQLite file, recall returns the **exact source text** (no LLM rewriting at ingest or query, so every memory is auditable), and capture/forget run with **no network calls** — your memories never leave your computer. Midas auto-scores what is worth keeping (facts, decisions, preferences, constraints, corrections) and drops chit-chat and duplicates, and it keeps the store bounded by forgetting low-value memories without an LLM. Install it and Claude starts remembering across sessions on its own via the recall → work → capture loop.", "author": { diff --git a/midas/__init__.py b/midas/__init__.py index 9132716..bbba783 100755 --- a/midas/__init__.py +++ b/midas/__init__.py @@ -116,4 +116,4 @@ "IVFIndex", "IVFStore", ] -__version__ = "1.1.0" +__version__ = "1.0.0" diff --git a/packages/midas-ts/package-lock.json b/packages/midas-ts/package-lock.json index c161a1b..9ddf92a 100644 --- a/packages/midas-ts/package-lock.json +++ b/packages/midas-ts/package-lock.json @@ -1,12 +1,12 @@ { "name": "midas-memory-mcp", - "version": "1.1.0", + "version": "1.0.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "midas-memory-mcp", - "version": "1.1.0", + "version": "1.0.0", "license": "Apache-2.0", "dependencies": { "@modelcontextprotocol/sdk": "^1.0.0", diff --git a/packages/midas-ts/package.json b/packages/midas-ts/package.json index 20d27d1..c140e1c 100644 --- a/packages/midas-ts/package.json +++ b/packages/midas-ts/package.json @@ -1,6 +1,6 @@ { "name": "midas-memory-mcp", - "version": "1.1.0", + "version": "1.0.0", "description": "Midas — local-first, source-traceable agent memory over MCP (TypeScript port, experimental). No LLM at ingest or query.", "license": "Apache-2.0", "type": "module", diff --git a/packages/midas-ts/src/mcp.ts b/packages/midas-ts/src/mcp.ts index 4e88a49..c1d0e3b 100644 --- a/packages/midas-ts/src/mcp.ts +++ b/packages/midas-ts/src/mcp.ts @@ -112,7 +112,7 @@ function text(data: string) { export function createServer(): McpServer { const server = new McpServer( - { name: "midas-memory", version: "1.1.0" }, // keep in sync with package.json on each release + { name: "midas-memory", version: "1.0.0" }, // keep in sync with package.json on each release { instructions: AGENT_MEMORY_INSTRUCTIONS }, ); diff --git a/pyproject.toml b/pyproject.toml index 3ecdd14..7afdaee 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "midas-memory" -version = "1.1.0" +version = "1.0.0" description = "Local-first, eval-first memory for long-horizon AI agents — no LLM at ingest" readme = "README.md" license = "Apache-2.0" diff --git a/server.json b/server.json index c7fc275..7e49ff2 100644 --- a/server.json +++ b/server.json @@ -7,13 +7,13 @@ "url": "https://github.com/vornicx/Midas", "source": "github" }, - "version": "1.1.0", + "version": "1.0.0", "packages": [ { "registryType": "pypi", "registryBaseUrl": "https://pypi.org", "identifier": "midas-memory-mcp", - "version": "1.1.0", + "version": "1.0.0", "runtimeHint": "uvx", "transport": { "type": "stdio"