From bcd310b8c1445e2ef3b7034523f83d42dd001e60 Mon Sep 17 00:00:00 2001 From: RV <263877755+xfznprojects@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:16:38 +0200 Subject: [PATCH 1/5] Add retrieval and tool-layer evaluations evals/ scores retrieval against 30 hand-labeled questions over the library scripts/seed_demo.py generates, and scores the tool layer separately against 7 computable questions. The corpus is built in a throwaway directory and the runner refuses to start unless the resolved upload root is inside it, so it cannot touch a real library. Labels follow what the analyzers actually detect rather than what the synthesizer was asked to produce; those differ. The weight ablation holds component scores fixed and varies only the blend, and asserts first that re-ranking under the production weights reproduces search() exactly, so the ablation cannot drift from the code it describes. --- evals/__init__.py | 1 + evals/golden_set.py | 191 ++++++++++++++++++++++++++++ evals/harness.py | 298 ++++++++++++++++++++++++++++++++++++++++++++ evals/metrics.py | 121 ++++++++++++++++++ evals/run.py | 246 ++++++++++++++++++++++++++++++++++++ evals/tool_set.py | 101 +++++++++++++++ 6 files changed, 958 insertions(+) create mode 100644 evals/__init__.py create mode 100644 evals/golden_set.py create mode 100644 evals/harness.py create mode 100644 evals/metrics.py create mode 100644 evals/run.py create mode 100644 evals/tool_set.py diff --git a/evals/__init__.py b/evals/__init__.py new file mode 100644 index 0000000..683a697 --- /dev/null +++ b/evals/__init__.py @@ -0,0 +1 @@ +"""Retrieval evaluation for SessionIQ.""" diff --git a/evals/golden_set.py b/evals/golden_set.py new file mode 100644 index 0000000..ab39d32 --- /dev/null +++ b/evals/golden_set.py @@ -0,0 +1,191 @@ +"""Labeled questions over the generated demo library. + +Ground truth is keyed on ``/`` because the demo corpus +contains several files that share a name (four ``cover.png`` artwork files). + +Every expected value here was read off a real ingestion run of +``scripts/seed_demo.py`` — the detected tempo, key, loudness and centroid in the +labels are what the analyzers actually produce, not the values the synthesizer +was asked for. Those two differ: ``cue_draft.wav`` is synthesized on D and +detected in A, for example. Labels follow the analyzers, because the analyzers +are what retrieval indexes. + +Re-generate the corpus after changing ``seed_demo.py`` and re-check the labels; +``tests/test_evals.py`` asserts that every label still names a real file. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +# Every asset scripts/seed_demo.py produces, project-qualified. +DEMO_CORPUS: tuple[str, ...] = ( + "Neon Horizon/Intro/intro.wav", + "Neon Horizon/Intro/cover.png", + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Midnight Drive/mix_notes.txt", + "Neon Horizon/Midnight Drive/cover.png", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Afterglow/cover.png", + "Lo-Fi Study/lofi_sketch.wav", + "Lo-Fi Study/chords.mid", + "Lo-Fi Study/ideas.txt", + "Lo-Fi Study/cover.png", + "Client Cue 03/cue_draft.wav", + "Client Cue 03/reference.wav", + "Client Cue 03/brief.txt", +) + +# Detected values, for readers of this file. +# intro.wav 89.1 bpm key G centroid 837.8 -18.8 LUFS +# midnight_drive.wav 117.5 bpm key E centroid 1218.1 -18.3 LUFS +# afterglow.wav 117.5 bpm key E centroid 1202.1 -18.4 LUFS +# afterglow_master.wav 117.5 bpm key E centroid 1208.0 -18.4 LUFS +# afterglow_reference.wav 117.5 bpm key E centroid 1190.4 -25.0 LUFS (quiet bounce) +# lofi_sketch.wav 76.0 bpm key C centroid 813.6 -18.9 LUFS +# cue_draft.wav 99.4 bpm key A centroid 871.0 -19.9 LUFS +# reference.wav 99.4 bpm key A centroid 858.2 -15.5 LUFS (loudest) + +FASTEST_TRACKS = ( + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", +) + +KEY_OF_E = FASTEST_TRACKS + +ARTWORK = ( + "Neon Horizon/Intro/cover.png", + "Neon Horizon/Midnight Drive/cover.png", + "Neon Horizon/Afterglow/cover.png", + "Lo-Fi Study/cover.png", +) + +# Notes whose text literally contains a TODO marker. +TODO_NOTES = ( + "Neon Horizon/Midnight Drive/mix_notes.txt", + "Lo-Fi Study/ideas.txt", + "Client Cue 03/brief.txt", +) + + +@dataclass(frozen=True) +class GoldenCase: + question: str + expected: tuple[str, ...] + category: str + project: str | None = None + + +GOLDEN_SET: tuple[GoldenCase, ...] = ( + # --- acoustic superlatives ------------------------------------------------ + GoldenCase("which track is the loudest?", ("Client Cue 03/reference.wav",), "superlative"), + GoldenCase( + "which track is the quietest?", + ("Neon Horizon/Afterglow/afterglow_reference.wav",), + "superlative", + ), + GoldenCase("which track is the slowest?", ("Lo-Fi Study/lofi_sketch.wav",), "superlative"), + GoldenCase("which tracks are the fastest?", FASTEST_TRACKS, "superlative"), + GoldenCase( + "which track is the brightest?", + ("Neon Horizon/Midnight Drive/midnight_drive.wav",), + "superlative", + ), + GoldenCase("which track is the darkest?", ("Lo-Fi Study/lofi_sketch.wav",), "superlative"), + # Same intent, natural wording that has to reach the metadata vocabulary + # through the synonym map rather than by literal token overlap. + GoldenCase("which song is the loudest?", ("Client Cue 03/reference.wav",), "synonym"), + GoldenCase("show me the fastest tune", FASTEST_TRACKS, "synonym"), + GoldenCase("which track has the highest volume?", ("Client Cue 03/reference.wav",), "synonym"), + # --- key ------------------------------------------------------------------ + GoldenCase("which track is in the key of G?", ("Neon Horizon/Intro/intro.wav",), "key"), + GoldenCase( + "which tracks are in the key of C?", + ("Lo-Fi Study/lofi_sketch.wav", "Lo-Fi Study/chords.mid"), + "key", + ), + GoldenCase("which audio files are in the key of E?", KEY_OF_E, "key"), + # --- note content --------------------------------------------------------- + GoldenCase("which notes mention a dusty piano loop?", ("Lo-Fi Study/ideas.txt",), "content"), + GoldenCase("which notes still have todo items?", TODO_NOTES, "content"), + GoldenCase( + "which notes mention sidechaining the pads?", + ("Neon Horizon/Midnight Drive/mix_notes.txt",), + "content", + ), + GoldenCase("what did the client ask for?", ("Client Cue 03/brief.txt",), "content"), + GoldenCase("which notes mention tape saturation?", ("Lo-Fi Study/ideas.txt",), "content"), + GoldenCase( + "where is the note about checking the low end on headphones?", + ("Neon Horizon/Afterglow/notes.txt",), + "content", + ), + # --- status --------------------------------------------------------------- + GoldenCase("what still needs work?", ("Client Cue 03/cue_draft.wav",), "status"), + GoldenCase( + "which files are ready?", + ("Neon Horizon/Afterglow/afterglow_master.wav",), + "status", + ), + GoldenCase( + "what is in progress?", + ("Neon Horizon/Midnight Drive/midnight_drive.wav",), + "status", + ), + # --- tags ----------------------------------------------------------------- + GoldenCase( + "which track is tagged synthwave?", + ("Neon Horizon/Midnight Drive/midnight_drive.wav",), + "tag", + ), + GoldenCase("which track is my favorite?", ("Neon Horizon/Afterglow/afterglow.wav",), "tag"), + GoldenCase("which track is chill?", ("Lo-Fi Study/lofi_sketch.wav",), "tag"), + GoldenCase( + "which track is marked as a single?", + ("Neon Horizon/Midnight Drive/midnight_drive.wav",), + "tag", + ), + # --- format --------------------------------------------------------------- + GoldenCase("which project has a midi file?", ("Lo-Fi Study/chords.mid",), "format"), + GoldenCase("where is the artwork?", ARTWORK, "format"), + # --- project-scoped ------------------------------------------------------- + GoldenCase( + "what files are in this project?", + ( + "Lo-Fi Study/lofi_sketch.wav", + "Lo-Fi Study/chords.mid", + "Lo-Fi Study/ideas.txt", + "Lo-Fi Study/cover.png", + ), + "project", + project="Lo-Fi Study", + ), + GoldenCase( + "which file is the reference?", + ("Client Cue 03/reference.wav",), + "project", + project="Client Cue 03", + ), + # --- aggregate ------------------------------------------------------------ + GoldenCase( + "list all the audio tracks", + ( + "Neon Horizon/Intro/intro.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Lo-Fi Study/lofi_sketch.wav", + "Client Cue 03/cue_draft.wav", + "Client Cue 03/reference.wav", + ), + "aggregate", + ), +) + +CATEGORIES: tuple[str, ...] = tuple(sorted({case.category for case in GOLDEN_SET})) diff --git a/evals/harness.py b/evals/harness.py new file mode 100644 index 0000000..a5d9797 --- /dev/null +++ b/evals/harness.py @@ -0,0 +1,298 @@ +"""Retrieval evaluation harness. + +Loads the generated demo library into a retriever, runs every question in the +golden set, and reports retrieval quality plus latency. + +The weight ablation re-scores candidates from the same component values the +production retriever computes, rather than calling ``search()`` once per +configuration. Before trusting those numbers the harness re-ranks the library +under the production weights and asserts the order matches what ``search()`` +actually returns, so the ablation cannot silently drift away from the code it +claims to describe. +""" + +from __future__ import annotations + +import json +import time +from collections import Counter, defaultdict +from dataclasses import dataclass, field + +from evals.golden_set import GOLDEN_SET, GoldenCase +from evals.metrics import CaseResult, MetricSummary, percentile, summarize +from evals.tool_set import TOOL_SET +from sessioniq.models import ProjectAsset +from sessioniq.retrieval import ( + METADATA_NORM, + InMemoryRetriever, + _inverse_document_frequency, # noqa: PLC2701 - internal harness, not shipped API + _metadata_score, + _weighted_cosine, + content_tokens, + expand_query_tokens, + is_aggregate_query, +) +from sessioniq.tools import LibraryToolbox + +K_VALUES = (1, 3, 5) +RANK_LIMIT = 10 + +# (name, lexical, metadata, vector) +WeightConfig = tuple[str, float, float, float] + +BASELINE_LEXICAL_METADATA: WeightConfig = ("production (lexical+metadata)", 0.70, 0.30, 0.0) +BASELINE_WITH_VECTOR: WeightConfig = ("production (with vector)", 0.55, 0.25, 0.20) + +ABLATION: tuple[WeightConfig, ...] = ( + ("lexical only", 1.00, 0.00, 0.00), + ("metadata only", 0.00, 1.00, 0.00), + ("lexical heavy", 0.85, 0.15, 0.00), + BASELINE_LEXICAL_METADATA, + ("balanced", 0.60, 0.40, 0.00), + ("metadata heavy", 0.40, 0.60, 0.00), +) + + +def qualified(asset: ProjectAsset) -> str: + """Project-qualified file name, unique across the demo corpus.""" + return f"{asset.project_name}/{asset.file_name}" + + +@dataclass +class AblationRow: + name: str + summary: MetricSummary + case_results: list[CaseResult] = field(default_factory=list) + + +@dataclass +class ToolCaseResult: + question: str + field: str + op: str + expected: tuple[str, ...] + actual: str | None + passed: bool + + +@dataclass +class EvalReport: + corpus_size: int + projects: int + vector_enabled: bool + baseline: MetricSummary + by_category: dict[str, MetricSummary] + ablation: list[AblationRow] + latency_ms: dict[str, float] + weak_cases: list[dict] + unretrieved: int + tool_results: list[ToolCaseResult] + faithfulness_checked: int + faithfulness_mismatches: list[str] + + +def _components( + retriever: InMemoryRetriever, + query: str, + project_name: str | None, +) -> tuple[list[ProjectAsset], dict[str, tuple[float, float, float]]]: + """Per-candidate (lexical, metadata, vector) values, as production computes them.""" + candidates = [ + asset + for asset in retriever.assets + if project_name is None + or asset.project_name == project_name + or asset.project_name.startswith(project_name + "/") + ] + if not candidates: + return [], {} + + query_counts = expand_query_tokens(content_tokens(query)) + doc_counts = { + asset.id: Counter(content_tokens(asset.cached_search_text())) for asset in candidates + } + idf = _inverse_document_frequency(doc_counts.values()) + semantic = retriever._semantic_scores(query, candidates) + + values = { + asset.id: ( + _weighted_cosine(query_counts, doc_counts[asset.id], idf), + min(_metadata_score(query, asset) / METADATA_NORM, 1.0), + semantic.get(asset.id, 0.0), + ) + for asset in candidates + } + return candidates, values + + +def _rank( + candidates: list[ProjectAsset], + values: dict[str, tuple[float, float, float]], + weights: tuple[float, float, float], + limit: int, +) -> list[tuple[ProjectAsset, float]]: + lexical_weight, metadata_weight, vector_weight = weights + scored = [] + for asset in candidates: + lexical, metadata, vector = values[asset.id] + score = ( + lexical_weight * lexical + metadata_weight * metadata + vector_weight * vector + ) + if score > 0: + scored.append((asset, score)) + scored.sort(key=lambda item: item[1], reverse=True) + return scored[:limit] + + +def _effective_limit(query: str, candidates: list[ProjectAsset], limit: int) -> int: + # Aggregate questions intentionally widen the candidate window in production. + return max(limit, len(candidates)) if is_aggregate_query(query) else limit + + +def _run_case( + retriever: InMemoryRetriever, + case: GoldenCase, + weights: tuple[float, float, float], + limit: int, +) -> CaseResult: + candidates, values = _components(retriever, case.question, case.project) + effective = _effective_limit(case.question, candidates, limit) + ranked = _rank(candidates, values, weights, effective) + return CaseResult( + question=case.question, + category=case.category, + expected=case.expected, + retrieved=[qualified(asset) for asset, _ in ranked], + top_scores=[score for _, score in ranked], + ) + + +def _check_faithfulness( + retriever: InMemoryRetriever, weights: tuple[float, float, float] +) -> tuple[int, list[str]]: + """The re-ranked order must match what search() returns under these weights.""" + checked = 0 + mismatches: list[str] = [] + for case in GOLDEN_SET: + candidates, values = _components(retriever, case.question, case.project) + if not candidates: + continue + effective = _effective_limit(case.question, candidates, RANK_LIMIT) + mine = [qualified(asset) for asset, _ in _rank(candidates, values, weights, effective)] + actual = [ + qualified(source.asset) + for source in retriever.search( + case.question, limit=RANK_LIMIT, project_name=case.project + ) + if source.score > 0 + ][:effective] + if mine: + checked += 1 + if mine != actual: + mismatches.append(f"{case.question!r}: harness={mine[:3]} search={actual[:3]}") + return checked, mismatches + + +def _latency(retriever: InMemoryRetriever, repeats: int = 5) -> dict[str, float]: + samples: list[float] = [] + for _ in range(repeats): + for case in GOLDEN_SET: + started = time.perf_counter() + retriever.search(case.question, limit=RANK_LIMIT, project_name=case.project) + samples.append((time.perf_counter() - started) * 1000) + return { + "calls": len(samples), + "mean": round(sum(samples) / len(samples), 3), + "p50": round(percentile(samples, 0.50), 3), + "p95": round(percentile(samples, 0.95), 3), + "max": round(max(samples), 3), + } + + +def run_tool_eval(assets: list[ProjectAsset]) -> list[ToolCaseResult]: + """Execute each computable question's tool call and check the winner.""" + toolbox = LibraryToolbox(assets) + by_id = {asset.id: qualified(asset) for asset in assets} + results: list[ToolCaseResult] = [] + for case in TOOL_SET: + payload = json.loads(toolbox.execute(case.tool, dict(case.arguments))) + winner = (payload.get("asset") or {}).get("asset_id") + actual = by_id.get(winner) if winner else None + results.append( + ToolCaseResult( + question=case.question, + field=case.field, + op=case.op, + expected=case.expected_any, + actual=actual, + passed=actual in case.expected_any, + ) + ) + return results + + +def run_eval(retriever: InMemoryRetriever, vector_enabled: bool) -> EvalReport: + """Evaluate an already-populated retriever against the golden set.""" + baseline_weights = BASELINE_WITH_VECTOR if vector_enabled else BASELINE_LEXICAL_METADATA + + case_results = [ + _run_case(retriever, case, baseline_weights[1:], RANK_LIMIT) for case in GOLDEN_SET + ] + baseline = summarize(case_results, K_VALUES) + + grouped: dict[str, list[CaseResult]] = defaultdict(list) + for result in case_results: + grouped[result.category].append(result) + by_category = {name: summarize(items, K_VALUES) for name, items in sorted(grouped.items())} + + rows: list[AblationRow] = [] + configs = list(ABLATION) + if vector_enabled: + configs.insert(0, BASELINE_WITH_VECTOR) + for name, *weights in configs: + if not vector_enabled and weights[2]: + continue + results = [_run_case(retriever, case, tuple(weights), RANK_LIMIT) for case in GOLDEN_SET] + rows.append( + AblationRow(name=name, summary=summarize(results, K_VALUES), case_results=results) + ) + + weak_cases = [] + unretrieved = 0 + for result in case_results: + expected = set(result.expected) + rank = next( + (i for i, name in enumerate(result.retrieved, start=1) if name in expected), + None, + ) + if rank == 1: + continue + if rank is None: + unretrieved += 1 + weak_cases.append( + { + "question": result.question, + "category": result.category, + "expected": list(result.expected), + "retrieved": result.retrieved[:5], + "first_relevant_rank": rank, + "scores": [round(score, 4) for score in result.top_scores[:5]], + } + ) + + checked, mismatches = _check_faithfulness(retriever, baseline_weights[1:]) + + return EvalReport( + corpus_size=len(retriever.assets), + projects=len({asset.project_name for asset in retriever.assets}), + vector_enabled=vector_enabled, + baseline=baseline, + by_category=by_category, + ablation=rows, + latency_ms=_latency(retriever), + weak_cases=weak_cases, + unretrieved=unretrieved, + tool_results=run_tool_eval(retriever.assets), + faithfulness_checked=checked, + faithfulness_mismatches=mismatches, + ) diff --git a/evals/metrics.py b/evals/metrics.py new file mode 100644 index 0000000..6ba5f4a --- /dev/null +++ b/evals/metrics.py @@ -0,0 +1,121 @@ +"""Retrieval metrics. + +Every function takes a ranked list of retrieved file names and the set of file +names that are actually relevant for that question. Ranking is by descending +retrieval score; anything the retriever scored at zero is not counted as +retrieved at all. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from statistics import mean + + +def _top_k(retrieved: list[str], k: int) -> list[str]: + return retrieved[: max(k, 0)] + + +def hit_at_k(retrieved: list[str], relevant: set[str], k: int) -> float: + """1.0 when at least one relevant file is in the top k, else 0.0.""" + if not relevant: + return 0.0 + return 1.0 if set(_top_k(retrieved, k)) & relevant else 0.0 + + +def recall_at_k(retrieved: list[str], relevant: set[str], k: int) -> float: + """Share of the relevant files that appear in the top k.""" + if not relevant: + return 0.0 + return len(set(_top_k(retrieved, k)) & relevant) / len(relevant) + + +def precision_at_k(retrieved: list[str], relevant: set[str], k: int) -> float: + """Share of the top k slots that hold a relevant file. + + Divided by k rather than by the number of results actually returned, so the + figure stays comparable between configurations that return different counts. + """ + if k <= 0: + return 0.0 + return len(set(_top_k(retrieved, k)) & relevant) / k + + +def reciprocal_rank(retrieved: list[str], relevant: set[str]) -> float: + """1 / rank of the first relevant file; 0.0 when none is retrieved.""" + if not relevant: + return 0.0 + for rank, name in enumerate(retrieved, start=1): + if name in relevant: + return 1.0 / rank + return 0.0 + + +def percentile(values: list[float], fraction: float) -> float: + """Linear-interpolated percentile of a sample, in the 0..1 range.""" + if not values: + return 0.0 + ordered = sorted(values) + if len(ordered) == 1: + return ordered[0] + position = max(0.0, min(1.0, fraction)) * (len(ordered) - 1) + lower = int(position) + upper = min(lower + 1, len(ordered) - 1) + weight = position - lower + return ordered[lower] * (1 - weight) + ordered[upper] * weight + + +@dataclass +class CaseResult: + question: str + category: str + expected: tuple[str, ...] + retrieved: list[str] + top_scores: list[float] = field(default_factory=list) + + def as_dict(self) -> dict: + return { + "question": self.question, + "category": self.category, + "expected": list(self.expected), + "retrieved": self.retrieved, + "top_scores": [round(score, 6) for score in self.top_scores], + } + + +@dataclass +class MetricSummary: + cases: int + hit: dict[int, float] + recall: dict[int, float] + precision: dict[int, float] + mrr: float + + def as_dict(self) -> dict: + return { + "cases": self.cases, + "hit_at_k": {str(k): round(v, 4) for k, v in self.hit.items()}, + "recall_at_k": {str(k): round(v, 4) for k, v in self.recall.items()}, + "precision_at_k": {str(k): round(v, 4) for k, v in self.precision.items()}, + "mrr": round(self.mrr, 4), + } + + +def summarize(results: list[CaseResult], k_values: tuple[int, ...] = (1, 3, 5)) -> MetricSummary: + """Average each metric across cases.""" + if not results: + return MetricSummary(0, dict.fromkeys(k_values, 0.0), {}, {}, 0.0) # type: ignore[arg-type] + return MetricSummary( + cases=len(results), + hit={ + k: mean(hit_at_k(r.retrieved, set(r.expected), k) for r in results) for k in k_values + }, + recall={ + k: mean(recall_at_k(r.retrieved, set(r.expected), k) for r in results) for k in k_values + }, + precision={ + k: mean(precision_at_k(r.retrieved, set(r.expected), k) for r in results) + for k in k_values + }, + mrr=mean(reciprocal_rank(r.retrieved, set(r.expected)) for r in results), + ) diff --git a/evals/run.py b/evals/run.py new file mode 100644 index 0000000..0f4a211 --- /dev/null +++ b/evals/run.py @@ -0,0 +1,246 @@ +"""Run the retrieval evaluation. + + python evals/run.py # offline lexical + metadata retrieval + python evals/run.py --vector # include the ChromaDB vector layer + python evals/run.py --out evals/results + +The demo library is generated into a throwaway directory and deleted +afterwards. This script refuses to run against any upload root other than the +one it created, so it can never purge a real library. +""" + +from __future__ import annotations + +import argparse +import json +import os +import shutil +import sys +import tempfile +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + + +def _build_assets(work: Path) -> list: + """Generate the demo corpus inside ``work`` and load it as ProjectAssets.""" + sys.path.insert(0, str(ROOT / "scripts")) + import seed_demo # noqa: PLC0415 - must load after SESSIONIQ_UPLOAD_ROOT is set + + from sessioniq.models import ProjectAsset # noqa: PLC0415 + from sessioniq.project_workspace import UPLOAD_ROOT # noqa: PLC0415 + + if not str(UPLOAD_ROOT.resolve()).startswith(str(work.resolve())): + raise SystemExit( + f"refusing to run: upload root {UPLOAD_ROOT} is outside the temporary " + f"directory {work}" + ) + + seed_demo.purge() + seed_demo.build() + index = json.loads((work / "library-index.json").read_text(encoding="utf-8")) + return [ProjectAsset.model_validate(entry) for entry in index["assets"]] + + +def _markdown(report, generated_at: str) -> str: + lines: list[str] = [] + add = lines.append + + add("# Retrieval evaluation") + add("") + add(f"Generated {generated_at} by `python evals/run.py`.") + add("") + add(f"- Corpus: **{report.corpus_size} assets** across **{report.projects} projects**, " + "generated by `scripts/seed_demo.py`") + add(f"- Queries: **{report.baseline.cases}** labeled questions in `evals/golden_set.py`") + add(f"- Vector layer: **{'enabled' if report.vector_enabled else 'disabled'}**") + add(f"- Ranking fidelity: re-ranked order matched `search()` on " + f"**{report.faithfulness_checked}** queries " + f"({len(report.faithfulness_mismatches)} mismatches)") + add("") + + summary = report.baseline + add("## Headline") + add("") + add("| metric | @1 | @3 | @5 |") + add("| --- | --- | --- | --- |") + add(f"| hit rate | {summary.hit[1]:.3f} | {summary.hit[3]:.3f} | {summary.hit[5]:.3f} |") + add(f"| recall | {summary.recall[1]:.3f} | {summary.recall[3]:.3f} | {summary.recall[5]:.3f} |") + add(f"| precision | {summary.precision[1]:.3f} | {summary.precision[3]:.3f} | " + f"{summary.precision[5]:.3f} |") + add(f"| MRR | {summary.mrr:.3f} | | |") + add("") + + lat = report.latency_ms + add("## Latency per query") + add("") + add(f"Measured over {lat['calls']} calls (p50 **{lat['p50']} ms**, " + f"p95 **{lat['p95']} ms**, max {lat['max']} ms, mean {lat['mean']} ms).") + add("") + + add("## By question type") + add("") + add("| category | n | hit@1 | hit@3 | hit@5 | MRR |") + add("| --- | --- | --- | --- | --- | --- |") + for name, cat in report.by_category.items(): + add(f"| {name} | {cat.cases} | {cat.hit[1]:.3f} | {cat.hit[3]:.3f} | " + f"{cat.hit[5]:.3f} | {cat.mrr:.3f} |") + add("") + + tool_passed = sum(1 for r in report.tool_results if r.passed) + add("## Computable questions (tool layer)") + add("") + add("Superlative questions ask for a comparison, not for a similar file, so every audio " + "file matches the query words about equally well and ranking cannot pick a winner. " + "`compute_stat` compares the real numbers instead.") + add("") + add(f"**{tool_passed}/{len(report.tool_results)}** returned the correct asset.") + add("") + add("| question | field | op | expected | tool returned | |") + add("| --- | --- | --- | --- | --- | --- |") + for result in report.tool_results: + mark = "ok" if result.passed else "**wrong**" + expected = ( + result.expected[0] + if len(result.expected) == 1 + else f"one of {len(result.expected)} tied" + ) + add(f"| {result.question} | {result.field} | {result.op} | {expected} | " + f"{result.actual or '(nothing)'} | {mark} |") + add("") + + add("## Blend weight ablation") + add("") + add("Component scores are held fixed and only the blend weights change, so each row " + "isolates what the weighting is worth.") + add("") + add("| configuration | hit@1 | hit@3 | hit@5 | recall@5 | precision@5 | MRR |") + add("| --- | --- | --- | --- | --- | --- | --- |") + for row in report.ablation: + s = row.summary + add(f"| {row.name} | {s.hit[1]:.3f} | {s.hit[3]:.3f} | {s.hit[5]:.3f} | " + f"{s.recall[5]:.3f} | {s.precision[5]:.3f} | {s.mrr:.3f} |") + add("") + + add("## Cases not ranked first") + add("") + add(f"{len(report.weak_cases)} of {report.baseline.cases} questions did not put an expected " + f"file at rank 1; {report.unretrieved} of those retrieved no expected file at all.") + add("") + if report.weak_cases: + add("| question | category | first relevant rank | top result |") + add("| --- | --- | --- | --- |") + for case in report.weak_cases: + rank = case["first_relevant_rank"] or "not retrieved" + top = case["retrieved"][0] if case["retrieved"] else "(nothing)" + add(f"| {case['question']} | {case['category']} | {rank} | {top} |") + add("") + add("The superlative questions dominate this list by design — they are the ones the " + "tool layer answers exactly, above.") + add("") + return "\n".join(lines) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--out", default=str(ROOT / "evals" / "results"), + help="directory for the JSON and Markdown reports") + parser.add_argument("--vector", action="store_true", + help="enable the ChromaDB vector layer (downloads a model)") + parser.add_argument("--keep", action="store_true", + help="keep the generated corpus for inspection") + args = parser.parse_args() + + work = Path(tempfile.mkdtemp(prefix="sessioniq-eval-")) + os.environ["SESSIONIQ_UPLOAD_ROOT"] = str(work / "uploads") + if not args.vector: + os.environ["SESSIONIQ_DISABLE_VECTOR"] = "1" + + try: + from evals.harness import run_eval # noqa: PLC0415 + from sessioniq.retrieval import HybridRetriever, InMemoryRetriever # noqa: PLC0415 + + print(f"building demo corpus in {work} ...") + assets = _build_assets(work) + + retriever = ( + HybridRetriever(persist_directory=work / "chroma") + if args.vector + else InMemoryRetriever() + ) + retriever.add_assets(assets) + vector_on = bool(getattr(retriever, "vector_enabled", False)) + print(f"vector layer: {'enabled' if vector_on else 'disabled'}") + + report = run_eval(retriever, vector_enabled=vector_on) + + if report.faithfulness_mismatches: + print("\nRANKING FIDELITY FAILED — the harness does not reproduce search():") + for mismatch in report.faithfulness_mismatches: + print(f" {mismatch}") + return 1 + + from datetime import UTC, datetime # noqa: PLC0415 + + generated = datetime.now(UTC).strftime("%Y-%m-%d") + out_dir = Path(args.out) + out_dir.mkdir(parents=True, exist_ok=True) + tool_passed = sum(1 for r in report.tool_results if r.passed) + + payload = { + "generated": generated, + "corpus_size": report.corpus_size, + "projects": report.projects, + "vector_enabled": report.vector_enabled, + "questions": report.baseline.cases, + "baseline": report.baseline.as_dict(), + "by_category": {k: v.as_dict() for k, v in report.by_category.items()}, + "ablation": [{"name": r.name, **r.summary.as_dict()} for r in report.ablation], + "latency_ms": report.latency_ms, + "tool_accuracy": { + "passed": tool_passed, + "total": len(report.tool_results), + "cases": [ + { + "question": r.question, + "field": r.field, + "op": r.op, + "expected": list(r.expected), + "actual": r.actual, + "passed": r.passed, + } + for r in report.tool_results + ], + }, + "weak_cases": report.weak_cases, + "unretrieved": report.unretrieved, + "ranking_fidelity": { + "checked": report.faithfulness_checked, + "mismatches": report.faithfulness_mismatches, + }, + } + stem = "retrieval-report-vector" if report.vector_enabled else "retrieval-report" + (out_dir / f"{stem}.json").write_text( + json.dumps(payload, indent=2), encoding="utf-8" + ) + (out_dir / f"{stem}.md").write_text( + _markdown(report, generated), encoding="utf-8" + ) + + print(f"\nwrote {out_dir / f'{stem}.md'}") + print(f"wrote {out_dir / f'{stem}.json'}") + print(f"\nhit@1 {report.baseline.hit[1]:.3f} hit@3 {report.baseline.hit[3]:.3f} " + f"hit@5 {report.baseline.hit[5]:.3f} MRR {report.baseline.mrr:.3f} " + f"p95 {report.latency_ms['p95']} ms") + print(f"tool accuracy {tool_passed}/{len(report.tool_results)}") + return 0 + finally: + if args.keep: + print(f"corpus kept at {work}") + else: + shutil.rmtree(work, ignore_errors=True) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/tool_set.py b/evals/tool_set.py new file mode 100644 index 0000000..087c492 --- /dev/null +++ b/evals/tool_set.py @@ -0,0 +1,101 @@ +"""Questions that must be computed rather than retrieved. + +"Which track is the loudest" is not a similarity question: every audio file +matches the words *loud* and *track* about equally well, so ranking by text +overlap cannot pick a winner. These questions are answered by the tool layer +(``compute_stat``), which compares real numbers. + +Each case names the tool call the assistant is expected to make and the asset +that call must return. The two tie cases accept any member of the tied set, +because which one wins depends on snapshot order rather than on the comparison. +""" + +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class ToolCase: + question: str + tool: str + arguments: dict + expected_any: tuple[str, ...] + field: str + op: str + + +# Detected extremes in the demo corpus, for reference: +# rms_db max -14.89 (Client Cue 03/reference.wav) min -24.57 (afterglow_reference.wav) +# bpm max 117.5 (four tied) min 76.0 (lofi_sketch.wav) +# centroid max 1218.1 (midnight_drive.wav) min 813.6 (lofi_sketch.wav) +# duration max 8.0s (four tied) + +TOOL_SET: tuple[ToolCase, ...] = ( + ToolCase( + "which track is the loudest?", + "compute_stat", + {"field": "rms_db", "op": "max"}, + ("Client Cue 03/reference.wav",), + "rms_db", + "max", + ), + ToolCase( + "which track is the quietest?", + "compute_stat", + {"field": "rms_db", "op": "min"}, + ("Neon Horizon/Afterglow/afterglow_reference.wav",), + "rms_db", + "min", + ), + ToolCase( + "which track is the slowest?", + "compute_stat", + {"field": "bpm", "op": "min"}, + ("Lo-Fi Study/lofi_sketch.wav",), + "bpm", + "min", + ), + ToolCase( + "which tracks are the fastest?", + "compute_stat", + {"field": "bpm", "op": "max"}, + ( + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + ), + "bpm", + "max", + ), + ToolCase( + "which track is the brightest?", + "compute_stat", + {"field": "brightness", "op": "max"}, + ("Neon Horizon/Midnight Drive/midnight_drive.wav",), + "brightness", + "max", + ), + ToolCase( + "which track is the darkest?", + "compute_stat", + {"field": "brightness", "op": "min"}, + ("Lo-Fi Study/lofi_sketch.wav",), + "brightness", + "min", + ), + ToolCase( + "which track is the longest?", + "compute_stat", + {"field": "duration", "op": "max"}, + ( + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + ), + "duration", + "max", + ), +) From c82450f9319363c3ff1a527c5fa834b457180a4a Mon Sep 17 00:00:00 2001 From: RV <263877755+xfznprojects@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:16:41 +0200 Subject: [PATCH 2/5] Add tests for the evaluation harness MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Covers the metric arithmetic, the aggregate-limit rule, and the integrity of both labeled sets: every expected file must exist in the demo corpus, and every tool case must name a field the toolbox actually exposes. The corpus itself is not built here — that needs audio analysis, so the full evaluation stays a separate command. --- pyproject.toml | 2 +- tests/test_evals.py | 161 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 162 insertions(+), 1 deletion(-) create mode 100644 tests/test_evals.py diff --git a/pyproject.toml b/pyproject.toml index 41f22c0..5cc36eb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -41,7 +41,7 @@ where = ["src"] [tool.pytest.ini_options] testpaths = ["tests"] -pythonpath = ["src"] +pythonpath = ["src", "."] [tool.ruff] line-length = 100 diff --git a/tests/test_evals.py b/tests/test_evals.py new file mode 100644 index 0000000..b968160 --- /dev/null +++ b/tests/test_evals.py @@ -0,0 +1,161 @@ +"""Tests for the retrieval evaluation harness. + +These cover the metric arithmetic and the integrity of the labeled sets. The +full evaluation needs a generated audio corpus, so it runs as a separate +command (``python evals/run.py``) rather than inside the unit suite. +""" + +from __future__ import annotations + +import pytest + +from evals.golden_set import DEMO_CORPUS, GOLDEN_SET +from evals.harness import _effective_limit +from evals.metrics import ( + CaseResult, + hit_at_k, + percentile, + precision_at_k, + recall_at_k, + reciprocal_rank, + summarize, +) +from evals.tool_set import TOOL_SET +from sessioniq.tools import FIELD_ACCESSORS + +RETRIEVED = ["a.wav", "b.wav", "c.wav", "d.wav"] + + +class TestRankingMetrics: + def test_hit_at_k_respects_the_window(self): + assert hit_at_k(RETRIEVED, {"b.wav"}, 1) == 0.0 + assert hit_at_k(RETRIEVED, {"b.wav"}, 2) == 1.0 + assert hit_at_k(RETRIEVED, {"b.wav"}, 4) == 1.0 + + def test_hit_is_zero_when_nothing_expected_is_retrieved(self): + assert hit_at_k(RETRIEVED, {"z.wav"}, 4) == 0.0 + + def test_recall_is_the_share_of_relevant_files_found(self): + assert recall_at_k(RETRIEVED, {"a.wav", "z.wav"}, 4) == 0.5 + assert recall_at_k(RETRIEVED, {"a.wav", "b.wav"}, 2) == 1.0 + assert recall_at_k(RETRIEVED, {"c.wav", "d.wav"}, 1) == 0.0 + + def test_precision_divides_by_k_not_by_result_count(self): + # Only two results returned, one relevant, but k is 4. + assert precision_at_k(["a.wav", "z.wav"], {"a.wav"}, 4) == 0.25 + + def test_reciprocal_rank_uses_the_first_relevant_position(self): + assert reciprocal_rank(RETRIEVED, {"a.wav"}) == 1.0 + assert reciprocal_rank(RETRIEVED, {"c.wav"}) == pytest.approx(1 / 3) + + def test_reciprocal_rank_is_zero_when_absent(self): + assert reciprocal_rank(RETRIEVED, {"z.wav"}) == 0.0 + + def test_empty_expectation_never_scores(self): + assert hit_at_k(RETRIEVED, set(), 4) == 0.0 + assert recall_at_k(RETRIEVED, set(), 4) == 0.0 + assert reciprocal_rank(RETRIEVED, set()) == 0.0 + + +class TestPercentile: + def test_interpolates_between_neighbours(self): + assert percentile([1, 2, 3, 4], 0.5) == pytest.approx(2.5) + + def test_returns_edges_for_extremes(self): + assert percentile([1, 2, 3, 4], 0.0) == 1 + assert percentile([1, 2, 3, 4], 1.0) == 4 + + def test_single_value_and_empty_input(self): + assert percentile([7.0], 0.95) == 7.0 + assert percentile([], 0.5) == 0.0 + + +class TestSummarize: + def _case(self, retrieved, expected, category="x"): + return CaseResult( + question="q", category=category, expected=tuple(expected), retrieved=list(retrieved) + ) + + def test_averages_across_cases(self): + results = [ + self._case(["a.wav"], ["a.wav"]), + self._case(["z.wav"], ["a.wav"]), + ] + summary = summarize(results) + assert summary.cases == 2 + assert summary.hit[1] == pytest.approx(0.5) + assert summary.mrr == pytest.approx(0.5) + + def test_empty_result_set_summarizes_to_zero(self): + summary = summarize([]) + assert summary.cases == 0 + assert summary.mrr == 0.0 + assert all(value == 0.0 for value in summary.hit.values()) + + +class TestAggregateLimitExpansion: + """`is_aggregate_query` widens the candidate window for whole-library questions.""" + + def _candidates(self) -> list[str]: + return ["a", "b", "c", "d", "e"] + + def test_aggregate_questions_widen_the_window(self): + assert _effective_limit("list all the audio tracks", self._candidates(), 4) == 5 + + def test_interrogatives_are_treated_as_aggregate(self): + # "which" and "how" are aggregate hints, so ordinary-looking questions + # also return every candidate that scored above zero. + for question in ("which track is the loudest?", "how many tracks are there?"): + assert _effective_limit(question, self._candidates(), 4) == 5 + + def test_questions_without_hint_words_keep_the_requested_limit(self): + assert _effective_limit("describe the sidechain on the pads", self._candidates(), 4) == 4 + + +class TestGoldenSetIntegrity: + def test_every_expected_file_is_in_the_demo_corpus(self): + known = set(DEMO_CORPUS) + unknown = { + name for case in GOLDEN_SET for name in case.expected if name not in known + } + assert not unknown, f"golden set names files the demo corpus does not create: {unknown}" + + def test_questions_are_unique(self): + questions = [case.question for case in GOLDEN_SET] + assert len(questions) == len(set(questions)) + + def test_every_case_has_an_expectation_and_a_category(self): + for case in GOLDEN_SET: + assert case.expected, f"{case.question!r} has no expected files" + assert case.category + + def test_project_scoped_cases_name_a_project_in_the_corpus(self): + projects = {name.split("/")[0] for name in DEMO_CORPUS} + for case in GOLDEN_SET: + if case.project: + assert case.project in projects, f"{case.question!r} targets {case.project}" + + def test_corpus_has_no_duplicate_entries(self): + assert len(DEMO_CORPUS) == len(set(DEMO_CORPUS)) + + +class TestToolSetIntegrity: + def test_every_expected_file_is_in_the_demo_corpus(self): + known = set(DEMO_CORPUS) + unknown = { + name for case in TOOL_SET for name in case.expected_any if name not in known + } + assert not unknown, f"tool set names files the demo corpus does not create: {unknown}" + + def test_fields_are_ones_the_toolbox_exposes(self): + for case in TOOL_SET: + assert case.field in FIELD_ACCESSORS + + def test_ops_are_comparable(self): + for case in TOOL_SET: + assert case.op in {"min", "max"} + + def test_arguments_match_the_case_metadata(self): + for case in TOOL_SET: + assert case.arguments["field"] == case.field + assert case.arguments["op"] == case.op From 56ee512ac079e845c37ed52053cc24b329ca2db4 Mon Sep 17 00:00:00 2001 From: RV <263877755+xfznprojects@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:16:48 +0200 Subject: [PATCH 3/5] Publish the measured results and the methodology The README claimed trustworthy citations with nothing measured behind it. It now reports what retrieval and the tool layer actually score, on which questions, at what latency, and what the evaluation does not cover. Both configurations are checked in so the numbers can be read without running anything: lexical plus metadata, and the same with the vector layer enabled. --- README.md | 23 + evals/README.md | 126 +++++ evals/results/retrieval-report-vector.json | 591 ++++++++++++++++++++ evals/results/retrieval-report-vector.md | 81 +++ evals/results/retrieval-report.json | 593 +++++++++++++++++++++ evals/results/retrieval-report.md | 81 +++ 6 files changed, 1495 insertions(+) create mode 100644 evals/README.md create mode 100644 evals/results/retrieval-report-vector.json create mode 100644 evals/results/retrieval-report-vector.md create mode 100644 evals/results/retrieval-report.json create mode 100644 evals/results/retrieval-report.md diff --git a/README.md b/README.md index 0a17b7d..723976d 100644 --- a/README.md +++ b/README.md @@ -214,6 +214,29 @@ A persistent bottom player continues across views and remembers the last track a +## 📊 Measured results + +`evals/` holds a labeled evaluation: 30 questions over a generated 17-file corpus, each with the +files that should answer it, plus 7 computable questions checked against the tool layer. + +| configuration | hit@1 | hit@3 | hit@5 | recall@5 | MRR | p95 latency | +| --- | --- | --- | --- | --- | --- | --- | +| lexical + metadata (offline default) | 0.733 | 0.867 | 0.900 | 0.871 | 0.806 | 1.9 ms | +| + vector search (ChromaDB) | 0.767 | 0.867 | 0.900 | 0.879 | 0.822 | 851 ms | + +Every expected file was retrieved for all 30 questions. Topical questions — key, status, tags, +project scope, note content — put the right file first essentially every time. + +Superlative questions do not, and are not meant to. *"Which track is the loudest"* asks for a +comparison, not a similar file: every audio file matches the words about equally well, so ranking +cannot pick a winner. Those go through the tool layer instead, which answered **7 of 7** correctly +by comparing real numbers. + +The trade-off is the honest one: vector search buys 3 points of hit@1 and costs roughly two orders +of magnitude in latency. That is why it is optional. + +Methodology, metric definitions and known limits: [`evals/README.md`](evals/README.md). + ## 🏗️ Architecture ```mermaid diff --git a/evals/README.md b/evals/README.md new file mode 100644 index 0000000..6da1714 --- /dev/null +++ b/evals/README.md @@ -0,0 +1,126 @@ +# Retrieval evaluation + +SessionIQ's claim is that answers are grounded in the files they cite. That claim is only worth +something if retrieval actually finds the right files, so this directory measures whether it does — +and measures it against known answers rather than against a feeling. + +``` +python evals/run.py # offline: lexical + metadata +python evals/run.py --vector # adds the ChromaDB vector layer +``` + +Both runs write a Markdown and JSON report to `evals/results/`. The JSON is the machine-readable +form; the Markdown is what the numbers in the main README come from. + +## Results + +Measured 2026-09-17 over 30 labeled questions on a 17-file corpus. + +| configuration | hit@1 | hit@3 | hit@5 | recall@5 | MRR | p50 | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | +| lexical + metadata | 0.733 | 0.867 | 0.900 | 0.871 | 0.806 | 1.0 ms | 1.9 ms | +| + vector (ChromaDB) | 0.767 | 0.867 | 0.900 | 0.879 | 0.822 | 322 ms | 851 ms | + +Every expected file was retrieved somewhere in all 30 questions. What separates the two +configurations is how often the right file comes *first*, and what that costs in latency. + +### Where retrieval is strong, and where it isn't + +| question type | n | hit@1 | hit@5 | MRR | +| --- | --- | --- | --- | --- | +| key | 3 | 1.000 | 1.000 | 1.000 | +| status | 3 | 1.000 | 1.000 | 1.000 | +| tag | 4 | 1.000 | 1.000 | 1.000 | +| format | 2 | 1.000 | 1.000 | 1.000 | +| project-scoped | 2 | 1.000 | 1.000 | 1.000 | +| content | 6 | 0.833 | 1.000 | 0.917 | +| synonym-expanded | 3 | 0.333 | 1.000 | 0.556 | +| superlative | 6 | 0.167 | 0.500 | 0.336 | + +Topical questions — which files are in this key, which are ready, which carry a tag, what a note +says — are answered essentially perfectly. Superlative questions are not, and that is by design +rather than by accident: *"which track is the loudest"* is a comparison, not a similarity search. +Every audio file contains the words *loud*, *track* and *bpm* about equally often, so ranking by +text overlap has no basis on which to pick a winner. It surfaces the right neighbourhood and stops +there. + +That is what the tool layer is for, so this harness measures it separately. + +## Computable questions (tool layer) + +`evals/tool_set.py` lists seven superlative questions and the tool call that should answer each. +The harness executes the call and checks the asset it names. + +**7 / 7 returned the correct asset.** Each one is an exact comparison over real numbers — the +loudest file, the slowest tempo, the brightest spectral centroid — rather than a guess about which +file sounds relevant. + +Reading the two tables together is the point: retrieval finds candidates, tools compute answers. +Neither replaces the other, and a single end-to-end "accuracy" number would hide that. + +## How the corpus works + +The evaluation runs against the library that `scripts/seed_demo.py` generates, not against a real +one. That generator synthesizes every file in code, so the corpus is identical on every machine and +in CI, and no licensing question arises from committing it (nothing is committed — it is rebuilt +each run). + +Ground truth is labeled against what the analyzers actually report, not against what the +synthesizer was asked for. These differ: `cue_draft.wav` is synthesized on D and detected in A. +Labels follow the analyzers, because the analyzers are what retrieval indexes. + +The corpus lives in a temporary directory. `evals/run.py` refuses to start unless the resolved +upload root sits inside the directory it created, so it cannot purge a real library. + +## What is measured + +**Ranking quality** over the rank-ordered result list, at k = 1, 3, 5: + +- **hit@k** — did any expected file appear in the top k +- **recall@k** — what share of the expected files appeared in the top k +- **precision@k** — what share of the top k slots held an expected file, divided by k rather than + by the number of results returned, so configurations that return different counts stay comparable +- **MRR** — 1 / rank of the first expected file + +Results the retriever scored at zero are not counted as retrieved; the retriever returns +recently-analyzed files as a fallback when nothing matches, and those are not candidate answers. + +**Latency** as p50, p95 and max over 150 calls. + +**Blend weights** by holding the component scores fixed and varying only the weights, so each row +isolates what the weighting contributes. Before any ablation number is trusted, the harness re-ranks +the library under the production weights and asserts the order matches what `search()` really +returns. It passes on all 30 questions; if it ever stops passing, the harness has drifted from the +code it describes and the evaluation exits non-zero. + +## Limits + +- **One corpus, one domain.** Thirty questions over seventeen synthesized audio, MIDI and note + files. It is large enough to separate retrieval configurations and far too small to be a + benchmark. +- **Synthetic audio.** The loops are tonal and rhythmic enough for librosa, but they are not real + music, and real libraries have messier filenames and metadata. +- **Generated answers are not scored.** This measures retrieval and the tool layer. It does not + score the wording of an answer, and it does not measure hallucination. With no model configured + the offline engine answers deterministically, and scoring it would measure that engine rather + than a model. +- **Vector latency is machine-dependent.** The 322 ms p50 above includes embedding the query with + ChromaDB's default local model; repeated runs on one machine varied between roughly 320 ms and + 851 ms p95 depending on load. Read it as "hundreds of milliseconds", not as a precise figure. +- **The vector configuration is not perfectly repeatable.** Across runs its MRR moved between 0.821 + and 0.822 and its recall@5 between 0.879 and 0.887, while hit@1, hit@3 and hit@5 held steady. The + offline configuration reproduced identically every time. Differences that small are within noise + at 30 questions — treat the vector rows as approximate. +- **Labels are hand-written.** They were read off a real ingestion run, and + `tests/test_evals.py` checks that every label still names a file the corpus creates, but a + mistaken label would quietly depress the score rather than announce itself. + +## Reproducing + +``` +python evals/run.py --keep # leave the generated corpus on disk for inspection +python -m pytest tests/test_evals.py +``` + +Regenerate the corpus and re-check the labels after changing `seed_demo.py`; the detected tempo, +key and centroid values the labels depend on are recorded at the top of `evals/golden_set.py`. diff --git a/evals/results/retrieval-report-vector.json b/evals/results/retrieval-report-vector.json new file mode 100644 index 0000000..e5ee3bf --- /dev/null +++ b/evals/results/retrieval-report-vector.json @@ -0,0 +1,591 @@ +{ + "generated": "2026-09-17", + "corpus_size": 17, + "projects": 5, + "vector_enabled": true, + "questions": 30, + "baseline": { + "cases": 30, + "hit_at_k": { + "1": 0.7667, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5736, + "3": 0.7847, + "5": 0.8792 + }, + "precision_at_k": { + "1": 0.7667, + "3": 0.4333, + "5": 0.32 + }, + "mrr": 0.8224 + }, + "by_category": { + "aggregate": { + "cases": 1, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.125, + "3": 0.375, + "5": 0.625 + }, + "precision_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "mrr": 1.0 + }, + "content": { + "cases": 6, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.8889, + "3": 0.9444, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.3889, + "5": 0.2667 + }, + "mrr": 1.0 + }, + "format": { + "cases": 2, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.625, + "3": 0.875, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.6667, + "5": 0.5 + }, + "mrr": 1.0 + }, + "key": { + "cases": 3, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.5833, + "3": 0.9167, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.6667, + "5": 0.4667 + }, + "mrr": 1.0 + }, + "project": { + "cases": 2, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.625, + "3": 0.875, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.6667, + "5": 0.5 + }, + "mrr": 1.0 + }, + "status": { + "cases": 3, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.3333, + "5": 0.2 + }, + "mrr": 1.0 + }, + "superlative": { + "cases": 6, + "hit_at_k": { + "1": 0.1667, + "3": 0.3333, + "5": 0.5 + }, + "recall_at_k": { + "1": 0.0417, + "3": 0.2917, + "5": 0.5 + }, + "precision_at_k": { + "1": 0.1667, + "3": 0.2222, + "5": 0.2 + }, + "mrr": 0.334 + }, + "synonym": { + "cases": 3, + "hit_at_k": { + "1": 0.3333, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.0833, + "3": 0.8333, + "5": 0.9167 + }, + "precision_at_k": { + "1": 0.3333, + "3": 0.4444, + "5": 0.3333 + }, + "mrr": 0.5556 + }, + "tag": { + "cases": 4, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.3333, + "5": 0.2 + }, + "mrr": 1.0 + } + }, + "ablation": [ + { + "name": "production (with vector)", + "cases": 30, + "hit_at_k": { + "1": 0.7667, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5736, + "3": 0.7847, + "5": 0.8792 + }, + "precision_at_k": { + "1": 0.7667, + "3": 0.4333, + "5": 0.32 + }, + "mrr": 0.8224 + }, + { + "name": "lexical only", + "cases": 30, + "hit_at_k": { + "1": 0.7, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5069, + "3": 0.7708, + "5": 0.8542 + }, + "precision_at_k": { + "1": 0.7, + "3": 0.4222, + "5": 0.3067 + }, + "mrr": 0.7784 + }, + { + "name": "metadata only", + "cases": 30, + "hit_at_k": { + "1": 0.3667, + "3": 0.6, + "5": 0.7 + }, + "recall_at_k": { + "1": 0.2736, + "3": 0.5319, + "5": 0.6875 + }, + "precision_at_k": { + "1": 0.3667, + "3": 0.2667, + "5": 0.2267 + }, + "mrr": 0.5247 + }, + { + "name": "lexical heavy", + "cases": 30, + "hit_at_k": { + "1": 0.7333, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5403, + "3": 0.7597, + "5": 0.8542 + }, + "precision_at_k": { + "1": 0.7333, + "3": 0.4111, + "5": 0.3067 + }, + "mrr": 0.8062 + }, + { + "name": "production (lexical+metadata)", + "cases": 30, + "hit_at_k": { + "1": 0.7333, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5403, + "3": 0.7597, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.7333, + "3": 0.4111, + "5": 0.3133 + }, + "mrr": 0.8062 + }, + { + "name": "balanced", + "cases": 30, + "hit_at_k": { + "1": 0.7, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5069, + "3": 0.7764, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.7, + "3": 0.4222, + "5": 0.3133 + }, + "mrr": 0.7847 + }, + { + "name": "metadata heavy", + "cases": 30, + "hit_at_k": { + "1": 0.6667, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.4736, + "3": 0.7764, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.6667, + "3": 0.4222, + "5": 0.3133 + }, + "mrr": 0.7631 + } + ], + "latency_ms": { + "calls": 150, + "mean": 407.779, + "p50": 321.887, + "p95": 850.876, + "max": 1256.44 + }, + "tool_accuracy": { + "passed": 7, + "total": 7, + "cases": [ + { + "question": "which track is the loudest?", + "field": "rms_db", + "op": "max", + "expected": [ + "Client Cue 03/reference.wav" + ], + "actual": "Client Cue 03/reference.wav", + "passed": true + }, + { + "question": "which track is the quietest?", + "field": "rms_db", + "op": "min", + "expected": [ + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Afterglow/afterglow_reference.wav", + "passed": true + }, + { + "question": "which track is the slowest?", + "field": "bpm", + "op": "min", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "actual": "Lo-Fi Study/lofi_sketch.wav", + "passed": true + }, + { + "question": "which tracks are the fastest?", + "field": "bpm", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + }, + { + "question": "which track is the brightest?", + "field": "brightness", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + }, + { + "question": "which track is the darkest?", + "field": "brightness", + "op": "min", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "actual": "Lo-Fi Study/lofi_sketch.wav", + "passed": true + }, + { + "question": "which track is the longest?", + "field": "duration", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + } + ] + }, + "weak_cases": [ + { + "question": "which track is the loudest?", + "category": "superlative", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1895, + 0.1892, + 0.1891, + 0.1879, + 0.1862 + ] + }, + { + "question": "which track is the quietest?", + "category": "superlative", + "expected": [ + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "first_relevant_rank": 4, + "scores": [ + 0.1878, + 0.1876, + 0.1876, + 0.1861, + 0.1861 + ] + }, + { + "question": "which track is the slowest?", + "category": "superlative", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "first_relevant_rank": 7, + "scores": [ + 0.1902, + 0.1894, + 0.1893, + 0.1884, + 0.1871 + ] + }, + { + "question": "which track is the brightest?", + "category": "superlative", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/reference.wav" + ], + "first_relevant_rank": 6, + "scores": [ + 0.1673, + 0.167, + 0.1668, + 0.1664, + 0.1648 + ] + }, + { + "question": "which track is the darkest?", + "category": "superlative", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "first_relevant_rank": 9, + "scores": [ + 0.1701, + 0.1698, + 0.1691, + 0.1687, + 0.1667 + ] + }, + { + "question": "which song is the loudest?", + "category": "synonym", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1841, + 0.1833, + 0.1826, + 0.181, + 0.1802 + ] + }, + { + "question": "which track has the highest volume?", + "category": "synonym", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1843, + 0.184, + 0.1829, + 0.1827, + 0.1826 + ] + } + ], + "unretrieved": 0, + "ranking_fidelity": { + "checked": 30, + "mismatches": [] + } +} \ No newline at end of file diff --git a/evals/results/retrieval-report-vector.md b/evals/results/retrieval-report-vector.md new file mode 100644 index 0000000..3b488ca --- /dev/null +++ b/evals/results/retrieval-report-vector.md @@ -0,0 +1,81 @@ +# Retrieval evaluation + +Generated 2026-09-17 by `python evals/run.py`. + +- Corpus: **17 assets** across **5 projects**, generated by `scripts/seed_demo.py` +- Queries: **30** labeled questions in `evals/golden_set.py` +- Vector layer: **enabled** +- Ranking fidelity: re-ranked order matched `search()` on **30** queries (0 mismatches) + +## Headline + +| metric | @1 | @3 | @5 | +| --- | --- | --- | --- | +| hit rate | 0.767 | 0.867 | 0.900 | +| recall | 0.574 | 0.785 | 0.879 | +| precision | 0.767 | 0.433 | 0.320 | +| MRR | 0.822 | | | + +## Latency per query + +Measured over 150 calls (p50 **321.887 ms**, p95 **850.876 ms**, max 1256.44 ms, mean 407.779 ms). + +## By question type + +| category | n | hit@1 | hit@3 | hit@5 | MRR | +| --- | --- | --- | --- | --- | --- | +| aggregate | 1 | 1.000 | 1.000 | 1.000 | 1.000 | +| content | 6 | 1.000 | 1.000 | 1.000 | 1.000 | +| format | 2 | 1.000 | 1.000 | 1.000 | 1.000 | +| key | 3 | 1.000 | 1.000 | 1.000 | 1.000 | +| project | 2 | 1.000 | 1.000 | 1.000 | 1.000 | +| status | 3 | 1.000 | 1.000 | 1.000 | 1.000 | +| superlative | 6 | 0.167 | 0.333 | 0.500 | 0.334 | +| synonym | 3 | 0.333 | 1.000 | 1.000 | 0.556 | +| tag | 4 | 1.000 | 1.000 | 1.000 | 1.000 | + +## Computable questions (tool layer) + +Superlative questions ask for a comparison, not for a similar file, so every audio file matches the query words about equally well and ranking cannot pick a winner. `compute_stat` compares the real numbers instead. + +**7/7** returned the correct asset. + +| question | field | op | expected | tool returned | | +| --- | --- | --- | --- | --- | --- | +| which track is the loudest? | rms_db | max | Client Cue 03/reference.wav | Client Cue 03/reference.wav | ok | +| which track is the quietest? | rms_db | min | Neon Horizon/Afterglow/afterglow_reference.wav | Neon Horizon/Afterglow/afterglow_reference.wav | ok | +| which track is the slowest? | bpm | min | Lo-Fi Study/lofi_sketch.wav | Lo-Fi Study/lofi_sketch.wav | ok | +| which tracks are the fastest? | bpm | max | one of 4 tied | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | +| which track is the brightest? | brightness | max | Neon Horizon/Midnight Drive/midnight_drive.wav | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | +| which track is the darkest? | brightness | min | Lo-Fi Study/lofi_sketch.wav | Lo-Fi Study/lofi_sketch.wav | ok | +| which track is the longest? | duration | max | one of 4 tied | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | + +## Blend weight ablation + +Component scores are held fixed and only the blend weights change, so each row isolates what the weighting is worth. + +| configuration | hit@1 | hit@3 | hit@5 | recall@5 | precision@5 | MRR | +| --- | --- | --- | --- | --- | --- | --- | +| production (with vector) | 0.767 | 0.867 | 0.900 | 0.879 | 0.320 | 0.822 | +| lexical only | 0.700 | 0.867 | 0.900 | 0.854 | 0.307 | 0.778 | +| metadata only | 0.367 | 0.600 | 0.700 | 0.688 | 0.227 | 0.525 | +| lexical heavy | 0.733 | 0.867 | 0.900 | 0.854 | 0.307 | 0.806 | +| production (lexical+metadata) | 0.733 | 0.867 | 0.900 | 0.871 | 0.313 | 0.806 | +| balanced | 0.700 | 0.867 | 0.900 | 0.871 | 0.313 | 0.785 | +| metadata heavy | 0.667 | 0.867 | 0.900 | 0.871 | 0.313 | 0.763 | + +## Cases not ranked first + +7 of 30 questions did not put an expected file at rank 1; 0 of those retrieved no expected file at all. + +| question | category | first relevant rank | top result | +| --- | --- | --- | --- | +| which track is the loudest? | superlative | 3 | Neon Horizon/Afterglow/afterglow.wav | +| which track is the quietest? | superlative | 4 | Neon Horizon/Afterglow/afterglow.wav | +| which track is the slowest? | superlative | 7 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track is the brightest? | superlative | 6 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track is the darkest? | superlative | 9 | Neon Horizon/Afterglow/afterglow_master.wav | +| which song is the loudest? | synonym | 3 | Neon Horizon/Afterglow/afterglow.wav | +| which track has the highest volume? | synonym | 3 | Neon Horizon/Afterglow/afterglow_master.wav | + +The superlative questions dominate this list by design — they are the ones the tool layer answers exactly, above. diff --git a/evals/results/retrieval-report.json b/evals/results/retrieval-report.json new file mode 100644 index 0000000..55adfbe --- /dev/null +++ b/evals/results/retrieval-report.json @@ -0,0 +1,593 @@ +{ + "generated": "2026-09-17", + "corpus_size": 17, + "projects": 5, + "vector_enabled": false, + "questions": 30, + "baseline": { + "cases": 30, + "hit_at_k": { + "1": 0.7333, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5403, + "3": 0.7597, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.7333, + "3": 0.4111, + "5": 0.3133 + }, + "mrr": 0.8062 + }, + "by_category": { + "aggregate": { + "cases": 1, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.125, + "3": 0.375, + "5": 0.625 + }, + "precision_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "mrr": 1.0 + }, + "content": { + "cases": 6, + "hit_at_k": { + "1": 0.8333, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.7222, + "3": 0.9444, + "5": 1.0 + }, + "precision_at_k": { + "1": 0.8333, + "3": 0.3889, + "5": 0.2667 + }, + "mrr": 0.9167 + }, + "format": { + "cases": 2, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.625, + "3": 0.875, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.6667, + "5": 0.5 + }, + "mrr": 1.0 + }, + "key": { + "cases": 3, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.5833, + "3": 0.75, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.5556, + "5": 0.4667 + }, + "mrr": 1.0 + }, + "project": { + "cases": 2, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.625, + "3": 0.875, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.6667, + "5": 0.5 + }, + "mrr": 1.0 + }, + "status": { + "cases": 3, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.3333, + "5": 0.2 + }, + "mrr": 1.0 + }, + "superlative": { + "cases": 6, + "hit_at_k": { + "1": 0.1667, + "3": 0.3333, + "5": 0.5 + }, + "recall_at_k": { + "1": 0.0417, + "3": 0.25, + "5": 0.4583 + }, + "precision_at_k": { + "1": 0.1667, + "3": 0.1667, + "5": 0.1667 + }, + "mrr": 0.3363 + }, + "synonym": { + "cases": 3, + "hit_at_k": { + "1": 0.3333, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 0.0833, + "3": 0.8333, + "5": 0.9167 + }, + "precision_at_k": { + "1": 0.3333, + "3": 0.4444, + "5": 0.3333 + }, + "mrr": 0.5556 + }, + "tag": { + "cases": 4, + "hit_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "recall_at_k": { + "1": 1.0, + "3": 1.0, + "5": 1.0 + }, + "precision_at_k": { + "1": 1.0, + "3": 0.3333, + "5": 0.2 + }, + "mrr": 1.0 + } + }, + "ablation": [ + { + "name": "lexical only", + "cases": 30, + "hit_at_k": { + "1": 0.7, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5069, + "3": 0.7708, + "5": 0.8542 + }, + "precision_at_k": { + "1": 0.7, + "3": 0.4222, + "5": 0.3067 + }, + "mrr": 0.7784 + }, + { + "name": "metadata only", + "cases": 30, + "hit_at_k": { + "1": 0.3667, + "3": 0.6, + "5": 0.7 + }, + "recall_at_k": { + "1": 0.2736, + "3": 0.5319, + "5": 0.6875 + }, + "precision_at_k": { + "1": 0.3667, + "3": 0.2667, + "5": 0.2267 + }, + "mrr": 0.5247 + }, + { + "name": "lexical heavy", + "cases": 30, + "hit_at_k": { + "1": 0.7333, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5403, + "3": 0.7597, + "5": 0.8542 + }, + "precision_at_k": { + "1": 0.7333, + "3": 0.4111, + "5": 0.3067 + }, + "mrr": 0.8062 + }, + { + "name": "production (lexical+metadata)", + "cases": 30, + "hit_at_k": { + "1": 0.7333, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5403, + "3": 0.7597, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.7333, + "3": 0.4111, + "5": 0.3133 + }, + "mrr": 0.8062 + }, + { + "name": "balanced", + "cases": 30, + "hit_at_k": { + "1": 0.7, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.5069, + "3": 0.7764, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.7, + "3": 0.4222, + "5": 0.3133 + }, + "mrr": 0.7847 + }, + { + "name": "metadata heavy", + "cases": 30, + "hit_at_k": { + "1": 0.6667, + "3": 0.8667, + "5": 0.9 + }, + "recall_at_k": { + "1": 0.4736, + "3": 0.7764, + "5": 0.8708 + }, + "precision_at_k": { + "1": 0.6667, + "3": 0.4222, + "5": 0.3133 + }, + "mrr": 0.7631 + } + ], + "latency_ms": { + "calls": 150, + "mean": 1.072, + "p50": 1.032, + "p95": 1.902, + "max": 2.69 + }, + "tool_accuracy": { + "passed": 7, + "total": 7, + "cases": [ + { + "question": "which track is the loudest?", + "field": "rms_db", + "op": "max", + "expected": [ + "Client Cue 03/reference.wav" + ], + "actual": "Client Cue 03/reference.wav", + "passed": true + }, + { + "question": "which track is the quietest?", + "field": "rms_db", + "op": "min", + "expected": [ + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Afterglow/afterglow_reference.wav", + "passed": true + }, + { + "question": "which track is the slowest?", + "field": "bpm", + "op": "min", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "actual": "Lo-Fi Study/lofi_sketch.wav", + "passed": true + }, + { + "question": "which tracks are the fastest?", + "field": "bpm", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + }, + { + "question": "which track is the brightest?", + "field": "brightness", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + }, + { + "question": "which track is the darkest?", + "field": "brightness", + "op": "min", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "actual": "Lo-Fi Study/lofi_sketch.wav", + "passed": true + }, + { + "question": "which track is the longest?", + "field": "duration", + "op": "max", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "actual": "Neon Horizon/Midnight Drive/midnight_drive.wav", + "passed": true + } + ] + }, + "weak_cases": [ + { + "question": "which track is the loudest?", + "category": "superlative", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1355, + 0.1348, + 0.1336, + 0.1327, + 0.1322 + ] + }, + { + "question": "which track is the quietest?", + "category": "superlative", + "expected": [ + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 4, + "scores": [ + 0.1355, + 0.1348, + 0.1336, + 0.1327, + 0.1322 + ] + }, + { + "question": "which track is the slowest?", + "category": "superlative", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 6, + "scores": [ + 0.1415, + 0.1408, + 0.1396, + 0.1386, + 0.1381 + ] + }, + { + "question": "which track is the brightest?", + "category": "superlative", + "expected": [ + "Neon Horizon/Midnight Drive/midnight_drive.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "first_relevant_rank": 8, + "scores": [ + 0.1155, + 0.1085, + 0.1081, + 0.1072, + 0.1064 + ] + }, + { + "question": "which track is the darkest?", + "category": "superlative", + "expected": [ + "Lo-Fi Study/lofi_sketch.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav" + ], + "first_relevant_rank": 7, + "scores": [ + 0.1155, + 0.1085, + 0.1081, + 0.1072, + 0.1064 + ] + }, + { + "question": "which song is the loudest?", + "category": "synonym", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1317, + 0.1311, + 0.1299, + 0.129, + 0.1286 + ] + }, + { + "question": "which track has the highest volume?", + "category": "synonym", + "expected": [ + "Client Cue 03/reference.wav" + ], + "retrieved": [ + "Neon Horizon/Afterglow/afterglow_master.wav", + "Neon Horizon/Afterglow/afterglow.wav", + "Client Cue 03/reference.wav", + "Neon Horizon/Afterglow/afterglow_reference.wav", + "Client Cue 03/cue_draft.wav" + ], + "first_relevant_rank": 3, + "scores": [ + 0.1223, + 0.1218, + 0.1207, + 0.1199, + 0.1195 + ] + }, + { + "question": "which notes mention sidechaining the pads?", + "category": "content", + "expected": [ + "Neon Horizon/Midnight Drive/mix_notes.txt" + ], + "retrieved": [ + "Neon Horizon/Afterglow/notes.txt", + "Neon Horizon/Midnight Drive/mix_notes.txt", + "Lo-Fi Study/chords.mid", + "Lo-Fi Study/ideas.txt", + "Client Cue 03/brief.txt" + ], + "first_relevant_rank": 2, + "scores": [ + 0.1746, + 0.173, + 0.0921, + 0.048, + 0.048 + ] + } + ], + "unretrieved": 0, + "ranking_fidelity": { + "checked": 30, + "mismatches": [] + } +} \ No newline at end of file diff --git a/evals/results/retrieval-report.md b/evals/results/retrieval-report.md new file mode 100644 index 0000000..6498b12 --- /dev/null +++ b/evals/results/retrieval-report.md @@ -0,0 +1,81 @@ +# Retrieval evaluation + +Generated 2026-09-17 by `python evals/run.py`. + +- Corpus: **17 assets** across **5 projects**, generated by `scripts/seed_demo.py` +- Queries: **30** labeled questions in `evals/golden_set.py` +- Vector layer: **disabled** +- Ranking fidelity: re-ranked order matched `search()` on **30** queries (0 mismatches) + +## Headline + +| metric | @1 | @3 | @5 | +| --- | --- | --- | --- | +| hit rate | 0.733 | 0.867 | 0.900 | +| recall | 0.540 | 0.760 | 0.871 | +| precision | 0.733 | 0.411 | 0.313 | +| MRR | 0.806 | | | + +## Latency per query + +Measured over 150 calls (p50 **1.032 ms**, p95 **1.902 ms**, max 2.69 ms, mean 1.072 ms). + +## By question type + +| category | n | hit@1 | hit@3 | hit@5 | MRR | +| --- | --- | --- | --- | --- | --- | +| aggregate | 1 | 1.000 | 1.000 | 1.000 | 1.000 | +| content | 6 | 0.833 | 1.000 | 1.000 | 0.917 | +| format | 2 | 1.000 | 1.000 | 1.000 | 1.000 | +| key | 3 | 1.000 | 1.000 | 1.000 | 1.000 | +| project | 2 | 1.000 | 1.000 | 1.000 | 1.000 | +| status | 3 | 1.000 | 1.000 | 1.000 | 1.000 | +| superlative | 6 | 0.167 | 0.333 | 0.500 | 0.336 | +| synonym | 3 | 0.333 | 1.000 | 1.000 | 0.556 | +| tag | 4 | 1.000 | 1.000 | 1.000 | 1.000 | + +## Computable questions (tool layer) + +Superlative questions ask for a comparison, not for a similar file, so every audio file matches the query words about equally well and ranking cannot pick a winner. `compute_stat` compares the real numbers instead. + +**7/7** returned the correct asset. + +| question | field | op | expected | tool returned | | +| --- | --- | --- | --- | --- | --- | +| which track is the loudest? | rms_db | max | Client Cue 03/reference.wav | Client Cue 03/reference.wav | ok | +| which track is the quietest? | rms_db | min | Neon Horizon/Afterglow/afterglow_reference.wav | Neon Horizon/Afterglow/afterglow_reference.wav | ok | +| which track is the slowest? | bpm | min | Lo-Fi Study/lofi_sketch.wav | Lo-Fi Study/lofi_sketch.wav | ok | +| which tracks are the fastest? | bpm | max | one of 4 tied | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | +| which track is the brightest? | brightness | max | Neon Horizon/Midnight Drive/midnight_drive.wav | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | +| which track is the darkest? | brightness | min | Lo-Fi Study/lofi_sketch.wav | Lo-Fi Study/lofi_sketch.wav | ok | +| which track is the longest? | duration | max | one of 4 tied | Neon Horizon/Midnight Drive/midnight_drive.wav | ok | + +## Blend weight ablation + +Component scores are held fixed and only the blend weights change, so each row isolates what the weighting is worth. + +| configuration | hit@1 | hit@3 | hit@5 | recall@5 | precision@5 | MRR | +| --- | --- | --- | --- | --- | --- | --- | +| lexical only | 0.700 | 0.867 | 0.900 | 0.854 | 0.307 | 0.778 | +| metadata only | 0.367 | 0.600 | 0.700 | 0.688 | 0.227 | 0.525 | +| lexical heavy | 0.733 | 0.867 | 0.900 | 0.854 | 0.307 | 0.806 | +| production (lexical+metadata) | 0.733 | 0.867 | 0.900 | 0.871 | 0.313 | 0.806 | +| balanced | 0.700 | 0.867 | 0.900 | 0.871 | 0.313 | 0.785 | +| metadata heavy | 0.667 | 0.867 | 0.900 | 0.871 | 0.313 | 0.763 | + +## Cases not ranked first + +8 of 30 questions did not put an expected file at rank 1; 0 of those retrieved no expected file at all. + +| question | category | first relevant rank | top result | +| --- | --- | --- | --- | +| which track is the loudest? | superlative | 3 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track is the quietest? | superlative | 4 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track is the slowest? | superlative | 6 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track is the brightest? | superlative | 8 | Neon Horizon/Afterglow/notes.txt | +| which track is the darkest? | superlative | 7 | Neon Horizon/Afterglow/notes.txt | +| which song is the loudest? | synonym | 3 | Neon Horizon/Afterglow/afterglow_master.wav | +| which track has the highest volume? | synonym | 3 | Neon Horizon/Afterglow/afterglow_master.wav | +| which notes mention sidechaining the pads? | content | 2 | Neon Horizon/Afterglow/notes.txt | + +The superlative questions dominate this list by design — they are the ones the tool layer answers exactly, above. From e6fddc9a526160f032228ea7a176e84b85e03ddc Mon Sep 17 00:00:00 2001 From: RV <263877755+xfznprojects@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:16:48 +0200 Subject: [PATCH 4/5] Run the evaluation in CI and publish the report The report lands in the job summary so a regression is visible without downloading the artifact. --- .github/workflows/ci.yml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 948ed77..d390c0d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -45,6 +45,25 @@ jobs: - run: pnpm test - run: pnpm build + eval: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-python@v7 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: pyproject.toml + - run: pip install -e ".[dev]" + - name: Run retrieval evaluation + run: python evals/run.py + - name: Publish report in the job summary + run: cat evals/results/retrieval-report.md >> "$GITHUB_STEP_SUMMARY" + - uses: actions/upload-artifact@v7 + with: + name: retrieval-report + path: evals/results/ + docker: runs-on: ubuntu-latest steps: From dc902ca3b8faf6552769ccf052fc3c1ae3de903c Mon Sep 17 00:00:00 2001 From: RV <263877755+xfznprojects@users.noreply.github.com> Date: Thu, 17 Sep 2026 19:20:01 +0200 Subject: [PATCH 5/5] State the cross-platform reproducibility limit exactly The evaluation now also runs on the Linux CI runner, where hit rates match Windows exactly but MRR differs by 0.001 through tie ordering. --- evals/README.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/evals/README.md b/evals/README.md index 6da1714..2e6c1a5 100644 --- a/evals/README.md +++ b/evals/README.md @@ -108,9 +108,11 @@ code it describes and the evaluation exits non-zero. ChromaDB's default local model; repeated runs on one machine varied between roughly 320 ms and 851 ms p95 depending on load. Read it as "hundreds of milliseconds", not as a precise figure. - **The vector configuration is not perfectly repeatable.** Across runs its MRR moved between 0.821 - and 0.822 and its recall@5 between 0.879 and 0.887, while hit@1, hit@3 and hit@5 held steady. The - offline configuration reproduced identically every time. Differences that small are within noise - at 30 questions — treat the vector rows as approximate. + and 0.822 and its recall@5 between 0.879 and 0.887, while hit@1, hit@3 and hit@5 held steady. + Differences that small are within noise at 30 questions — treat the vector rows as approximate. +- **The offline configuration is stable but not bit-identical across platforms.** Hit rates + reproduced exactly on every run and on both Windows and the Linux CI runner; MRR came out 0.806 + on Windows against 0.805 on Linux, which is tie ordering rather than a behavioural difference. - **Labels are hand-written.** They were read off a real ingestion run, and `tests/test_evals.py` checks that every label still names a file the corpus creates, but a mistaken label would quietly depress the score rather than announce itself.