From e0283ae826e00a2b02c174315151a83aa6be33b7 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 04:48:29 -0400 Subject: [PATCH 1/2] get_collection_health: compact healthy collectors to stay under the #4198 budget Measured 41,669 bytes at default arguments with every field on every collector (42-collector fleet) against the 32,768-byte budget. There is no row to cut here: every collector is one row, and a health read must never hide one that is failing, stale, disabled or erroring by leaving it off the page. Cuts fields per-row instead - a collector that is HEALTHY with zero errors, session/extension-missing runs, denials or abandoned runs this window, and either stored rows or is a known event collector resting at zero, compacts to 7 fields (collector/status/compact/total_runs/rows_stored/avg_duration_ms/ last_success). Everything else always keeps every field, including three HEALTHY-banded edge cases the band alone would miss (a partial error rate, an older permission denial behind a fresh success, and a non-event collector's unexplained zero). New full_detail=true opt-in restores every field on every row; collector_detail_note says how many rows this call compacted. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../CollectionHealthPayloadBudgetLiveTests.cs | 204 ++++++++++++++++++ .../Mcp/DarlingMcpDataTools.cs | 57 ++++- Lite/Mcp/McpHealthTools.cs | 58 ++++- 3 files changed, 313 insertions(+), 6 deletions(-) create mode 100644 Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs diff --git a/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs b/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs new file mode 100644 index 000000000..a14685b31 --- /dev/null +++ b/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs @@ -0,0 +1,204 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text; +using System.Text.Json; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #4198: get_collection_health has no row to drop (every collector on the server is one row, and a health +/// read must never hide a failing/stale/disabled/erroring one by leaving it off the page), so its default-size +/// cut is per-field instead. This seeds every SQL Server catalog collector - the realistic per-server shape - +/// mostly HEALTHY-and-boring, plus four deliberately NOT-boring rows that must never compact even though three +/// of the four band HEALTHY, and measures the tool method's own UTF-8 byte count. Its own file/seeding per the +/// #4198 common brief: not shared with any other lane's tonight. +/// +[Collection("live-postgres")] +public sealed class CollectionHealthPayloadBudgetLiveTests +{ + private const string ServerName = "darling-collection-health-budget-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + private readonly ITestOutputHelper _output; + public CollectionHealthPayloadBudgetLiveTests(ITestOutputHelper output) => _output = output; + + [Fact] + public async Task DefaultCall_StaysUnderBudget_AndNeverCompactsANonBoringCollector() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to run the live get_collection_health budget test."); + var ct = TestContext.Current.CancellationToken; + + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + try + { + await RegisterServerAsync(connection, ct); + + var now = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified); + var sqlServerCollectors = CollectorCatalog.All + .Where(d => d.TargetEngine == CollectorTargetEngine.SqlServer) + .Select(d => d.Name) + .ToArray(); + Assert.True(sqlServerCollectors.Length > 30, "expected a realistic SQL Server catalog width"); + + /* four collectors that must NEVER compact, three of them despite banding HEALTHY: */ + var neverCompact = new[] { "wait_stats", "memory_grant_stats", "query_store_health", "database_scoped_config" }; + Assert.All(neverCompact, name => Assert.Contains(name, sqlServerCollectors)); + + foreach (var name in sqlServerCollectors) + { + switch (name) + { + case "wait_stats": + /* FAILING band: recent ERROR runs, never a success, so HealthStatus itself excludes it. */ + for (var i = 0; i < 5; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 6), "ERROR", 120, null, + "Login failed for user 'darling_monitor'.", ct); + break; + + case "memory_grant_stats": + /* WARNING band via a 30% error rate - HealthStatus alone would already exclude this + one, but it also exercises errors > 0 beside a fresh fresh success. */ + for (var i = 0; i < 7; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 20 - 1), "SUCCESS", 80, 40, null, ct); + for (var i = 0; i < 3; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 30 - 2), "ERROR", 90, null, "Timeout expired.", ct); + break; + + case "query_store_health": + /* HEALTHY band (a fresh success, 0 current errors) but PermissionDeniedCount > 0 from + an OLDER denial this window - the case HealthStatus alone would miss and the reason + IsCollectionHealthCompactEligible checks PermissionDeniedCount directly. */ + await InsertLogRowAsync(connection, name, now.AddDays(-6), "PERMISSIONS", 50, null, + "permission denied for function pg_read_file", ct); + await InsertLogRowAsync(connection, name, now.AddDays(-6).AddHours(-1), "PERMISSIONS", 50, null, + "permission denied for function pg_read_file", ct); + for (var i = 0; i < 4; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 12), "SUCCESS", 60, 12, null, ct); + break; + + case "database_scoped_config": + /* HEALTHY band, RowsStored = 0, and NOT an event collector - FormatOutputFinding's + "needs a look" reading, which a compact row must never hide. */ + for (var i = 0; i < 8; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 18), "SUCCESS", 30, 0, null, ct); + break; + + case "deadlocks": + /* Event collector at rest: RowsStored = 0 but IsEventCollector is true, so this one + SHOULD compact - the boring-empty case #1852/#3754 protect deliberately as healthy. */ + for (var i = 0; i < 8; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 18), "SUCCESS", 15, 0, null, ct); + break; + + default: + /* The realistic majority: plainly healthy and productive. */ + for (var i = 0; i < 6; i++) + await InsertLogRowAsync(connection, name, now.AddHours(-i * 24 - 1), "SUCCESS", 100 + i * 15, 50 + i * 5, null, ct); + break; + } + } + + var defaultJson = await DarlingMcpDataTools.GetCollectionHealth(postgres, ServerName); + var defaultBytes = Encoding.UTF8.GetByteCount(defaultJson); + var fullJson = await DarlingMcpDataTools.GetCollectionHealth(postgres, ServerName, full_detail: true); + var fullBytes = Encoding.UTF8.GetByteCount(fullJson); + _output.WriteLine($"get_collection_health: default {defaultBytes:N0} bytes, full_detail=true {fullBytes:N0} bytes, budget {McpResponseBudget.DefaultBytes:N0}."); + + Assert.True(defaultBytes <= McpResponseBudget.DefaultBytes, + $"default get_collection_health is {defaultBytes:N0} bytes, over the {McpResponseBudget.DefaultBytes:N0}-byte budget."); + Assert.True(defaultBytes < fullBytes, "the default call should be smaller than full_detail=true."); + + using var defaultDoc = JsonDocument.Parse(defaultJson); + var defaultRows = defaultDoc.RootElement.GetProperty("collectors").EnumerateArray() + .ToDictionary(r => r.GetProperty("collector").GetString()!, r => r); + Assert.Equal(sqlServerCollectors.Length, defaultRows.Count); + + foreach (var name in neverCompact) + { + Assert.True(defaultRows[name].TryGetProperty("errors", out _), $"{name} must keep full detail by default (it is not boring-healthy)."); + Assert.False(defaultRows[name].TryGetProperty("compact", out _), $"{name} must not be marked compact."); + } + Assert.True(defaultRows["deadlocks"].TryGetProperty("compact", out var deadlocksCompact) && deadlocksCompact.GetBoolean(), + "an event collector resting at zero rows should compact."); + Assert.True(defaultRows.Values.Count(r => r.TryGetProperty("compact", out _)) >= sqlServerCollectors.Length - neverCompact.Length, + "every boring-healthy collector should compact."); + + using var fullDoc = JsonDocument.Parse(fullJson); + Assert.All(fullDoc.RootElement.GetProperty("collectors").EnumerateArray(), + r => Assert.False(r.TryGetProperty("compact", out _), "full_detail=true must serve every field on every row.")); + + var note = defaultDoc.RootElement.GetProperty("collector_detail_note").GetString(); + Assert.Contains($"of {sqlServerCollectors.Length} collector", note, StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static async Task RegisterServerAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct) + { + using var command = new NpgsqlCommand(@" +INSERT INTO servers (server_id, server_name, display_name, is_enabled, sql_major_version, created_date, modified_date) +VALUES ($1, $2, $3, TRUE, 15, $4, $4) +ON CONFLICT (server_id) DO UPDATE SET is_enabled = TRUE, sql_major_version = 15;", connection); + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified)); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task InsertLogRowAsync( + NpgsqlConnection connection, string collectorName, DateTime collectionTime, string status, + int durationMs, int? rowsCollected, string? errorMessage, System.Threading.CancellationToken ct) + { + using var command = new NpgsqlCommand(@" +INSERT INTO collection_log (log_id, collection_time, server_id, server_name, collector_name, status, duration_ms, rows_collected, error_message) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9)", connection); + command.Parameters.AddWithValue(CollectionIdGenerator.Next()); + command.Parameters.AddWithValue(DateTime.SpecifyKind(collectionTime, DateTimeKind.Unspecified)); + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(collectorName); + command.Parameters.AddWithValue(status); + command.Parameters.AddWithValue(durationMs); + command.Parameters.AddWithValue((object?)rowsCollected ?? DBNull.Value); + command.Parameters.AddWithValue((object?)errorMessage ?? DBNull.Value); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct) + { + using var cleanup = new NpgsqlCommand( + $"DELETE FROM collection_log WHERE server_id = {ServerId}; DELETE FROM servers WHERE server_id = {ServerId};", + connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs index 66cb3e7af..122d98e9e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs @@ -990,10 +990,11 @@ a consumer keys on a field instead of parsing the label. */ }, McpHelpers.JsonOptions); } - [McpServerTool(Name = "get_collection_health"), Description("Per-collector 7-day health for a server. STOPPED (gate off) does not count as failing; rows_stored=0 is NOT a fault by itself — output_finding says whether it is a resting event-collector or a denial needing a grant. last_error is a sticky slot, not necessarily current: check last_error_at/denied_since_last_success. regressed_from_productive floors WARNING on an axis SEPARATE from failed_collector_count — never add them. alert_read_health uses a DIFFERENT since-restart window, not this one; zero there means none since restart, not none in 7 days. <> Shows the health status of all data collectors for a server — whether they're running successfully, failing, or stale. A collector reads STOPPED rather than FAILING when it has attempted nothing at all — no success, no error, nothing — for longer than the FAILING cutoff, despite a history of runs: that is a collector whose gate (AppliesTo) flipped off for this target rather than one that keeps running and erroring, and it does not count toward a server's failing-collector total. A collector reads EXTENSION_MISSING when every attempt in the window was skipped because a PostgreSQL extension it declares is not installed on the target - last_error names the extension, CREATE EXTENSION (plus shared_preload_libraries and a restart where the message says so) is the remedy, never a grant, and an optional extension left uninstalled is a legitimate resting state rather than a fault. That last reading holds ONLY for a collector that never produced on this target, and regressed_from_productive is the field that tells you which one you are looking at: it is true when this collector WAS storing rows and has reported a named skip - EXTENSION_MISSING, PERMISSIONS or SESSION_MISSING - on every cycle since. That is a restart or an upgrade changing what the collector can read, which is a regression rather than rest, and it is not something the band can tell you on its own: the skip bands are reached only when the whole window holds no success, so a collector that regressed inside the last day still has a fresh success, reads HEALTHY on the ordinary staleness ladder, and is floored to WARNING by this flag alone. last_productive_at is when this collector last stored anything, rows_in_prior_7d is how much it stored over this seven-day window and is served only when the flag is true (only then is that figure entirely pre-regression, because a named skip stores nothing), and regression_finding is the sentence that puts the three facts in the order they happened. get_fleet_overview counts the same rows per server as regressed_collector_count. That count is a SEPARATE AXIS from failed_collector_count and the two must never be added: one regression reads WARNING on its first day and FAILING once its success clock runs out, and is regressed on both. Check this before investigating data to ensure collectors are working properly. Each row also carries last_note/note_count: what a NON-failing run reported, e.g. an enumeration that came back with 0 items. note_count equal to total_runs means the collector has been collecting nothing all window — not a fault (the target may be legitimately empty), but the reason a HEALTHY collector can still have no data. target_has_user_databases tells those two apart: true means the target DID have user databases in the same window, so an all-window empty enumeration is worth investigating (a login that cannot enter them, an exclusion filter that matched everything, a databases scope on the collector's schedule row that admits nothing - the last a Darling-only schedule knob); false means either no user databases or no inventory to go on. Each row also carries abandoned and abandon_rate_pct: cycles the 120-second whole-server wall-clock budget gave up on, which stored nothing and advanced no watermark. Unlike a yield, which retries, an abandoned cycle is collected data you do not have. A rate above 0.5% bands the collector WARNING, so a WARNING here may have nothing to do with errors - read abandoned beside errors to attribute it. CRITICAL for reading last_error: it is a single slot carrying the newest ERROR, PERMISSIONS, or EXTENSION_MISSING message in the whole window, and a message in it is NOT evidence that the condition is current. Read last_error_at for when it happened, last_denied_at for when the newest DENIAL specifically happened, and denied_since_last_success for the derived answer - true means a denial is the collector's current state, false means every denial in the window predates a later success and the collector is reading fine now. A fault recorded before a code path changed will sit in last_error for the rest of the window while every cycle since succeeds. Do not infer a live condition from last_error alone. Total abandonment still reads FAILING through staleness; the rate exists for the partial case, where a collector abandons some cycles and succeeds often enough to stay fresh, which otherwise read HEALTHY with errors 0 indefinitely. The sweep_pressure block is the server-level roll-up: it compares the collectors' combined execution demand (average duration amortized by cadence) against the minute the fastest cadence holds. SATURATED means the collection body cannot fit inside its cadence, so relaunches are skipped and the server collects at a multiple of its configured interval while every collector still reads healthy — heaviest_collectors names where that budget goes. That verdict is the SUSTAINED answer only. peak_cycle_risk is the separate single-sweep answer: peak_cycle_ms is what the body costs on the cycle where every scheduled cadence comes due together, and BODY_OVERRUN means that one body cannot fit the budget even when the verdict reads OK — the signature of one infrequent heavy collector, which amortization hides and heaviest_collectors therefore ranks out of sight. peak_collector names it, and peak_cycle_note explains it. Read both fields: a server can be OK/BODY_OVERRUN (a schedule-shape problem, fix by moving or splitting that collector) or SATURATED/BODY_OVERRUN (a capacity problem). Every collector row carries avg_duration_ms, p95_duration_ms and max_duration_ms, because a collector's runs are not always one population. Read the three together: avg close to p95 close to max is one population, avg far below p95 is two, and p95 far below max is one pathological run. peak_cycle_ms is built from p95 (floored at the mean, so it can never read lower than a mean-based figure) for exactly that reason, and peak_collector carries peak_run_ms beside avg_duration_ms so the gap is visible. Those three still describe RUNS, and a collector that runs once per DATABASE writes one blended row, so no run-level statistic can say which database cost what. Five fan out from an enumeration on any SQL Server target (query_store, plan_correction, query_store_health, index_object_stats, database_scoped_config); separately, eleven more fan out over a per-database connection loop when the target is Azure SQL DB, and pg_autovacuum_stats always does on PostgreSQL. The per-collector `fanout` block is that answer, null for a collector that does not fan out: `items` is how wide the fan-out was, `slowest`/`slowest_ms` name the dearest database and its cost on the window's worst run, `run_ms` is that whole run, `slowest_share_pct` is slowest_ms / run_ms as a percentage — the slowest item's share of the whole pass — and `dominance` is slowest_ms * items / run_ms, the slowest item against the MEAN item: 1.0 for a perfectly even fan-out, rising with concentration. The remediation decision routes through the SHARE, not through dominance: a low share means the cost is the fan-out's WIDTH — no single database is worth chasing, and bounded parallelism is the lever — while a high share means one database dominates the pass and a per-database schedule override or a stagger is what helps. Dominance is NOT that verdict, because its ceiling is items. A fixed threshold on dominance misreads exactly the widest fan-outs, where the cost lives. Dominance only reads as dominance when it approaches a meaningful fraction of items (share = dominance / items); it stays published as the evenness ratio, for continuity. Do not try to infer any of this from p95 versus avg — on a per-database collector that ratio is usually saturated by empty-versus-productive runs and says nothing about databases. Every field named so far describes what a collector SPENT; rows_stored, runs_with_rows and productive_run_pct are what it BOUGHT, counted over the same window as total_runs and the durations, so cost and output on a row always describe the same runs. Read them together for the readings that need different actions: rows_stored above zero is expensive AND productive; rows_stored zero with denied_since_last_success false, no faulted run and no note is a collector that read and found nothing, which for one that stores a row only when an event occurs (e.g. deadlocks, blocked_process_report, pg_blocking, pg_xmin_horizon) is the correct resting state and needs no action; rows_stored zero with denied_since_last_success true is a collector that could not read and needs a grant. output_finding says which zero reading applies and is null whenever rows_stored is positive. There are five, in the order they are decided: a current denial is the grant case; errors plus session_missing above zero means that many runs recorded a fault rather than a result, so on those runs the collector was UNABLE to read and the resting-state reading is withheld for the whole window (every run faulted is nothing read at all and needs action - the case an Azure SQL DB target produced when two collectors failed on every database every sweep and were recorded SUCCESS with a sentence saying they had read and found nothing); note_count above zero means the runs themselves recorded what they found, so the finding defers to last_note instead of assuming a category - which is what keeps a DELIBERATE zero distinguishable from a collector that quietly stopped storing rows; nothing on the row explaining the zero from an event collector is that collector at rest; and the same from a collector that is NOT event-triggered - a configuration or snapshot read such as database_scoped_config, which returns a row per setting per database - is its source coming back empty on every run, which is not a resting state and the finding says needs a look. The event-collector set is a closed list on the shared classifier, so a collector left off it gets the non-reassuring sentence rather than the reassuring one. query_store on a read-replica target is the deliberate case: it is not an event collector, and every run notes an empty enumeration because Query Store on a readable secondary is excluded by design. This is deliberately NOT a band. A verdict keyed on cost-plus-zero-rows would fire on the healthy quiet install rather than the blind one. These are NOT the hourly per-collector series Darling's get_collector_cost reports as total_rows - a separate series over that caller's own days_back and across every server at once, and Darling-only, so Lite has no twin of it; the top-level output_note names both windows and disclaims that one. rows_stored is also what a run STORED, never what the monitored engine counted, so a zero cannot tell a genuinely quiet source apart from a reader capturing nothing off a busy one - nothing on this surface measures that. One block on this response is deliberately NOT on the seven-day window: alert_read_health, which counts the alerting layer's OWN store reads that failed and were swallowed. A condition check that cannot read the store logs one line and skips - correctly, because firing on absent evidence would fabricate an alert and resolving on it would fabricate a recovery - and that skip is not a collector run, so it writes no collection_log row and reaches no other health surface: only a grep of the service log found the class. It matters out of proportion to the count because the alert pass runs on a much shorter store deadline than the collection sweep, so as store latency rises the alerting layer is the FIRST thing to fail and collection is the last - during one measured episode of store lock contention the service log's error rate rose 41 to 61 per hour, every line an alerting-side read, while collector failures over the same hours FELL from 23 to 2. Read server_read_failures beside server_alert_passes for this server (a pass is one alert evaluation pass containing many reads, so more failures than passes is ordinary and the pair is NOT a ratio; a Darling sweep runs two passes for a SQL Server target, three for a PostgreSQL one, and Lite runs one, so the denominator is comparable within a host and engine but not across them), instance_read_failures for the whole service (which also covers the fleet-scoped conditions that belong to no server and so appear in no per-server count: " + AlertReadFailureCounter.FleetScopedReads + "), fleet_read_failures for how many of that service total belong to no server at all - the figure that makes a nonzero service count readable from a server whose own count is zero, because those two populations take opposite actions: a blind fleet-scoped read means the store's own self-alerts went quiet and two of them are the reads whose alerts would say the store is in trouble, while one on another server is answered by reading this same block there, last_failure_read for which condition went blind most recently, last_failure_elapsed_ms for how long that read ran before it faulted - which is the term that says whose deadline ended it ONLY WHERE THE ALERT PASS SETS ONE. The Darling service does, on every store read; Lite's alerting reads go into its local store with no command deadline at all, so on Lite this is a plain duration that says a read became slow and nothing about who ended it. Where there is a deadline: an elapsed at or about it means this process stopped waiting while the statement was still running on the store, one well below that bound means the store returned a fault, and the exception text cannot make that distinction because a client-side deadline renders as a torn stream with no SQLSTATE exactly like a dropped connection. A figure well ABOVE the bound is a third reading: the failure was not a single bounded read, which each site's clock restart between consecutive awaits makes rare and which is expected only on the shared engine sweep entry, whose awaited operation is a whole alert pass - and last_failure_at to tell a healed episode from a live one: this count never ages out of a window, so a nonzero value with a stamp from days ago is history. Every count on this block carries its own newest-failure stamp, read name and elapsed, so none of them sits undated: last_failure_* are THIS server's, fleet_last_failure_* are the fleet-scoped conditions' and are attributable by construction, and instance_last_failure_* are the newest failure anywhere in this service whatever its scope - that last trio may therefore name a read on a server you did not ask about and makes no claim about which, which is why the fleet trio is reported separately rather than inferred from it. The third population is a subtraction: instance_read_failures minus server_read_failures minus fleet_read_failures is how many failed on OTHER servers, and that one has no stamp here by design - nothing holds a newest failure for it, and a count with nothing to date or name it is the gap the three trios close, so read this block on those servers to attribute it. retried_reads is the SECOND population and on the Darling service is where most of what this block used to count now lands: reads that crossed the 10 s alert-pass deadline once and succeeded on the single retry two seconds later - the store's write bands' cost, counted rather than blinding an alert. Every one of those would once have been a swallowed failure, so read the two together: retries rising with failures at zero is a store under write pressure whose alerting is intact, and both rising is a store where a second attempt twelve seconds later still found the band on, which wants the store's write schedule looked at rather than the reader's. A read that failed twice counts in both. It carries no stamp and no read name, deliberately - a retried read did not go blind, so there is no episode to date or attribute, and a stamp would invite reading a retry as a soft failure. instance_retried_reads is the same figure across the whole service, with no fleet part because nothing records a fleet-scoped retry today. On Lite both read zero as a property of the SKU and not of a quiet store: Lite's alerting reads carry no command deadline, so there is no deadline to cross and nothing to retry. counting_since is when this process began counting, early in its own startup - these are in-memory counts and a restart takes them to zero, so a zero means \"none since counting_since\" and NOT \"none in seven days\"; check the stamp before reading the zero as reassurance. Deliberately not persisted, because what it counts is a failure to READ the store. It does NOT count alerts that failed to DELIVER and makes no claim about them - that is get_alert_history's question. And deliberately not a band, for the same reason the output figures are not: any threshold over it would have to guess how many blind reads make alerting unhealthy, and a wrong guess on this particular surface fails by saying nothing is wrong. The service block beside alert_read_health is the build-attribution read: nothing else on this surface says WHAT build is answering, so the question an operator watching an install actually asks - \"is the running service the build that carries fix X\" - is answered by version directly instead of by restart inference plus a merge-list lookup the watcher has no way to perform. version is the running build's informational version - the same read the Darling service's --version verb prints, with any SemVer build-metadata suffix stripped and the prerelease suffix KEPT, because the nightly stamp lives in the prerelease and one nightly differs from the next in nothing else; Lite has no version verb and reads the same attribute off its own app assembly, stamped from the same single declaration. counting_since remains the restart detector, and a restart is NOT a build: a crash-restart moves counting_since exactly the way an install does, which is why inferring a build from it misattributes. started_at is the SAME instant counting_since carries, deliberately - both mean \"when this process came up\", and a second clock for one fact would put two near-identical stamps on one payload whose skew a reader would have to explain away. compiled_schema_version is the schema rung this BUILD expects - the compiled constant, not a read of the store's migrated rung - so beside version it says what this build requires of the store it serves, whether or not that store has caught up. When a recent analysis pass could not read one of its fact families for this server, the payload also carries analysis_caveats (analysis_time, families_failed, families_total, entries[{family, read, outcome, message, failed_in_last_passes}]) — the analysis pass's own reads of the collectors' tables, a different layer from the collector rows; absent when the last 24 remembered passes were clean. Process memory: a service restart forgets.")] + [McpServerTool(Name = "get_collection_health"), Description("Per-collector 7-day health for a server. STOPPED (gate off) does not count as failing; rows_stored=0 is NOT a fault by itself — output_finding says whether it is a resting event-collector or a denial needing a grant. last_error is a sticky slot, not necessarily current: check last_error_at/denied_since_last_success. regressed_from_productive floors WARNING on an axis SEPARATE from failed_collector_count — never add them. alert_read_health uses a DIFFERENT since-restart window, not this one; zero there means none since restart, not none in 7 days. <> Shows the health status of all data collectors for a server — whether they're running successfully, failing, or stale. A collector reads STOPPED rather than FAILING when it has attempted nothing at all — no success, no error, nothing — for longer than the FAILING cutoff, despite a history of runs: that is a collector whose gate (AppliesTo) flipped off for this target rather than one that keeps running and erroring, and it does not count toward a server's failing-collector total. A collector reads EXTENSION_MISSING when every attempt in the window was skipped because a PostgreSQL extension it declares is not installed on the target - last_error names the extension, CREATE EXTENSION (plus shared_preload_libraries and a restart where the message says so) is the remedy, never a grant, and an optional extension left uninstalled is a legitimate resting state rather than a fault. That last reading holds ONLY for a collector that never produced on this target, and regressed_from_productive is the field that tells you which one you are looking at: it is true when this collector WAS storing rows and has reported a named skip - EXTENSION_MISSING, PERMISSIONS or SESSION_MISSING - on every cycle since. That is a restart or an upgrade changing what the collector can read, which is a regression rather than rest, and it is not something the band can tell you on its own: the skip bands are reached only when the whole window holds no success, so a collector that regressed inside the last day still has a fresh success, reads HEALTHY on the ordinary staleness ladder, and is floored to WARNING by this flag alone. last_productive_at is when this collector last stored anything, rows_in_prior_7d is how much it stored over this seven-day window and is served only when the flag is true (only then is that figure entirely pre-regression, because a named skip stores nothing), and regression_finding is the sentence that puts the three facts in the order they happened. get_fleet_overview counts the same rows per server as regressed_collector_count. That count is a SEPARATE AXIS from failed_collector_count and the two must never be added: one regression reads WARNING on its first day and FAILING once its success clock runs out, and is regressed on both. Check this before investigating data to ensure collectors are working properly. Each row also carries last_note/note_count: what a NON-failing run reported, e.g. an enumeration that came back with 0 items. note_count equal to total_runs means the collector has been collecting nothing all window — not a fault (the target may be legitimately empty), but the reason a HEALTHY collector can still have no data. target_has_user_databases tells those two apart: true means the target DID have user databases in the same window, so an all-window empty enumeration is worth investigating (a login that cannot enter them, an exclusion filter that matched everything, a databases scope on the collector's schedule row that admits nothing - the last a Darling-only schedule knob); false means either no user databases or no inventory to go on. Each row also carries abandoned and abandon_rate_pct: cycles the 120-second whole-server wall-clock budget gave up on, which stored nothing and advanced no watermark. Unlike a yield, which retries, an abandoned cycle is collected data you do not have. A rate above 0.5% bands the collector WARNING, so a WARNING here may have nothing to do with errors - read abandoned beside errors to attribute it. CRITICAL for reading last_error: it is a single slot carrying the newest ERROR, PERMISSIONS, or EXTENSION_MISSING message in the whole window, and a message in it is NOT evidence that the condition is current. Read last_error_at for when it happened, last_denied_at for when the newest DENIAL specifically happened, and denied_since_last_success for the derived answer - true means a denial is the collector's current state, false means every denial in the window predates a later success and the collector is reading fine now. A fault recorded before a code path changed will sit in last_error for the rest of the window while every cycle since succeeds. Do not infer a live condition from last_error alone. Total abandonment still reads FAILING through staleness; the rate exists for the partial case, where a collector abandons some cycles and succeeds often enough to stay fresh, which otherwise read HEALTHY with errors 0 indefinitely. The sweep_pressure block is the server-level roll-up: it compares the collectors' combined execution demand (average duration amortized by cadence) against the minute the fastest cadence holds. SATURATED means the collection body cannot fit inside its cadence, so relaunches are skipped and the server collects at a multiple of its configured interval while every collector still reads healthy — heaviest_collectors names where that budget goes. That verdict is the SUSTAINED answer only. peak_cycle_risk is the separate single-sweep answer: peak_cycle_ms is what the body costs on the cycle where every scheduled cadence comes due together, and BODY_OVERRUN means that one body cannot fit the budget even when the verdict reads OK — the signature of one infrequent heavy collector, which amortization hides and heaviest_collectors therefore ranks out of sight. peak_collector names it, and peak_cycle_note explains it. Read both fields: a server can be OK/BODY_OVERRUN (a schedule-shape problem, fix by moving or splitting that collector) or SATURATED/BODY_OVERRUN (a capacity problem). Every collector row carries avg_duration_ms, p95_duration_ms and max_duration_ms, because a collector's runs are not always one population. Read the three together: avg close to p95 close to max is one population, avg far below p95 is two, and p95 far below max is one pathological run. peak_cycle_ms is built from p95 (floored at the mean, so it can never read lower than a mean-based figure) for exactly that reason, and peak_collector carries peak_run_ms beside avg_duration_ms so the gap is visible. Those three still describe RUNS, and a collector that runs once per DATABASE writes one blended row, so no run-level statistic can say which database cost what. Five fan out from an enumeration on any SQL Server target (query_store, plan_correction, query_store_health, index_object_stats, database_scoped_config); separately, eleven more fan out over a per-database connection loop when the target is Azure SQL DB, and pg_autovacuum_stats always does on PostgreSQL. The per-collector `fanout` block is that answer, null for a collector that does not fan out: `items` is how wide the fan-out was, `slowest`/`slowest_ms` name the dearest database and its cost on the window's worst run, `run_ms` is that whole run, `slowest_share_pct` is slowest_ms / run_ms as a percentage — the slowest item's share of the whole pass — and `dominance` is slowest_ms * items / run_ms, the slowest item against the MEAN item: 1.0 for a perfectly even fan-out, rising with concentration. The remediation decision routes through the SHARE, not through dominance: a low share means the cost is the fan-out's WIDTH — no single database is worth chasing, and bounded parallelism is the lever — while a high share means one database dominates the pass and a per-database schedule override or a stagger is what helps. Dominance is NOT that verdict, because its ceiling is items. A fixed threshold on dominance misreads exactly the widest fan-outs, where the cost lives. Dominance only reads as dominance when it approaches a meaningful fraction of items (share = dominance / items); it stays published as the evenness ratio, for continuity. Do not try to infer any of this from p95 versus avg — on a per-database collector that ratio is usually saturated by empty-versus-productive runs and says nothing about databases. Every field named so far describes what a collector SPENT; rows_stored, runs_with_rows and productive_run_pct are what it BOUGHT, counted over the same window as total_runs and the durations, so cost and output on a row always describe the same runs. Read them together for the readings that need different actions: rows_stored above zero is expensive AND productive; rows_stored zero with denied_since_last_success false, no faulted run and no note is a collector that read and found nothing, which for one that stores a row only when an event occurs (e.g. deadlocks, blocked_process_report, pg_blocking, pg_xmin_horizon) is the correct resting state and needs no action; rows_stored zero with denied_since_last_success true is a collector that could not read and needs a grant. output_finding says which zero reading applies and is null whenever rows_stored is positive. There are five, in the order they are decided: a current denial is the grant case; errors plus session_missing above zero means that many runs recorded a fault rather than a result, so on those runs the collector was UNABLE to read and the resting-state reading is withheld for the whole window (every run faulted is nothing read at all and needs action - the case an Azure SQL DB target produced when two collectors failed on every database every sweep and were recorded SUCCESS with a sentence saying they had read and found nothing); note_count above zero means the runs themselves recorded what they found, so the finding defers to last_note instead of assuming a category - which is what keeps a DELIBERATE zero distinguishable from a collector that quietly stopped storing rows; nothing on the row explaining the zero from an event collector is that collector at rest; and the same from a collector that is NOT event-triggered - a configuration or snapshot read such as database_scoped_config, which returns a row per setting per database - is its source coming back empty on every run, which is not a resting state and the finding says needs a look. The event-collector set is a closed list on the shared classifier, so a collector left off it gets the non-reassuring sentence rather than the reassuring one. query_store on a read-replica target is the deliberate case: it is not an event collector, and every run notes an empty enumeration because Query Store on a readable secondary is excluded by design. This is deliberately NOT a band. A verdict keyed on cost-plus-zero-rows would fire on the healthy quiet install rather than the blind one. These are NOT the hourly per-collector series Darling's get_collector_cost reports as total_rows - a separate series over that caller's own days_back and across every server at once, and Darling-only, so Lite has no twin of it; the top-level output_note names both windows and disclaims that one. rows_stored is also what a run STORED, never what the monitored engine counted, so a zero cannot tell a genuinely quiet source apart from a reader capturing nothing off a busy one - nothing on this surface measures that. One block on this response is deliberately NOT on the seven-day window: alert_read_health, which counts the alerting layer's OWN store reads that failed and were swallowed. A condition check that cannot read the store logs one line and skips - correctly, because firing on absent evidence would fabricate an alert and resolving on it would fabricate a recovery - and that skip is not a collector run, so it writes no collection_log row and reaches no other health surface: only a grep of the service log found the class. It matters out of proportion to the count because the alert pass runs on a much shorter store deadline than the collection sweep, so as store latency rises the alerting layer is the FIRST thing to fail and collection is the last - during one measured episode of store lock contention the service log's error rate rose 41 to 61 per hour, every line an alerting-side read, while collector failures over the same hours FELL from 23 to 2. Read server_read_failures beside server_alert_passes for this server (a pass is one alert evaluation pass containing many reads, so more failures than passes is ordinary and the pair is NOT a ratio; a Darling sweep runs two passes for a SQL Server target, three for a PostgreSQL one, and Lite runs one, so the denominator is comparable within a host and engine but not across them), instance_read_failures for the whole service (which also covers the fleet-scoped conditions that belong to no server and so appear in no per-server count: " + AlertReadFailureCounter.FleetScopedReads + "), fleet_read_failures for how many of that service total belong to no server at all - the figure that makes a nonzero service count readable from a server whose own count is zero, because those two populations take opposite actions: a blind fleet-scoped read means the store's own self-alerts went quiet and two of them are the reads whose alerts would say the store is in trouble, while one on another server is answered by reading this same block there, last_failure_read for which condition went blind most recently, last_failure_elapsed_ms for how long that read ran before it faulted - which is the term that says whose deadline ended it ONLY WHERE THE ALERT PASS SETS ONE. The Darling service does, on every store read; Lite's alerting reads go into its local store with no command deadline at all, so on Lite this is a plain duration that says a read became slow and nothing about who ended it. Where there is a deadline: an elapsed at or about it means this process stopped waiting while the statement was still running on the store, one well below that bound means the store returned a fault, and the exception text cannot make that distinction because a client-side deadline renders as a torn stream with no SQLSTATE exactly like a dropped connection. A figure well ABOVE the bound is a third reading: the failure was not a single bounded read, which each site's clock restart between consecutive awaits makes rare and which is expected only on the shared engine sweep entry, whose awaited operation is a whole alert pass - and last_failure_at to tell a healed episode from a live one: this count never ages out of a window, so a nonzero value with a stamp from days ago is history. Every count on this block carries its own newest-failure stamp, read name and elapsed, so none of them sits undated: last_failure_* are THIS server's, fleet_last_failure_* are the fleet-scoped conditions' and are attributable by construction, and instance_last_failure_* are the newest failure anywhere in this service whatever its scope - that last trio may therefore name a read on a server you did not ask about and makes no claim about which, which is why the fleet trio is reported separately rather than inferred from it. The third population is a subtraction: instance_read_failures minus server_read_failures minus fleet_read_failures is how many failed on OTHER servers, and that one has no stamp here by design - nothing holds a newest failure for it, and a count with nothing to date or name it is the gap the three trios close, so read this block on those servers to attribute it. retried_reads is the SECOND population and on the Darling service is where most of what this block used to count now lands: reads that crossed the 10 s alert-pass deadline once and succeeded on the single retry two seconds later - the store's write bands' cost, counted rather than blinding an alert. Every one of those would once have been a swallowed failure, so read the two together: retries rising with failures at zero is a store under write pressure whose alerting is intact, and both rising is a store where a second attempt twelve seconds later still found the band on, which wants the store's write schedule looked at rather than the reader's. A read that failed twice counts in both. It carries no stamp and no read name, deliberately - a retried read did not go blind, so there is no episode to date or attribute, and a stamp would invite reading a retry as a soft failure. instance_retried_reads is the same figure across the whole service, with no fleet part because nothing records a fleet-scoped retry today. On Lite both read zero as a property of the SKU and not of a quiet store: Lite's alerting reads carry no command deadline, so there is no deadline to cross and nothing to retry. counting_since is when this process began counting, early in its own startup - these are in-memory counts and a restart takes them to zero, so a zero means \"none since counting_since\" and NOT \"none in seven days\"; check the stamp before reading the zero as reassurance. Deliberately not persisted, because what it counts is a failure to READ the store. It does NOT count alerts that failed to DELIVER and makes no claim about them - that is get_alert_history's question. And deliberately not a band, for the same reason the output figures are not: any threshold over it would have to guess how many blind reads make alerting unhealthy, and a wrong guess on this particular surface fails by saying nothing is wrong. The service block beside alert_read_health is the build-attribution read: nothing else on this surface says WHAT build is answering, so the question an operator watching an install actually asks - \"is the running service the build that carries fix X\" - is answered by version directly instead of by restart inference plus a merge-list lookup the watcher has no way to perform. version is the running build's informational version - the same read the Darling service's --version verb prints, with any SemVer build-metadata suffix stripped and the prerelease suffix KEPT, because the nightly stamp lives in the prerelease and one nightly differs from the next in nothing else; Lite has no version verb and reads the same attribute off its own app assembly, stamped from the same single declaration. counting_since remains the restart detector, and a restart is NOT a build: a crash-restart moves counting_since exactly the way an install does, which is why inferring a build from it misattributes. started_at is the SAME instant counting_since carries, deliberately - both mean \"when this process came up\", and a second clock for one fact would put two near-identical stamps on one payload whose skew a reader would have to explain away. compiled_schema_version is the schema rung this BUILD expects - the compiled constant, not a read of the store's migrated rung - so beside version it says what this build requires of the store it serves, whether or not that store has caught up. When a recent analysis pass could not read one of its fact families for this server, the payload also carries analysis_caveats (analysis_time, families_failed, families_total, entries[{family, read, outcome, message, failed_in_last_passes}]) — the analysis pass's own reads of the collectors' tables, a different layer from the collector rows; absent when the last 24 remembered passes were clean. Process memory: a service restart forgets. full_detail (#4198): every collector row above defaults to a compact shape - collector, status, compact, total_runs, rows_stored, avg_duration_ms, last_success - for a collector that is HEALTHY with zero errors, session_missing, extension_missing, permission_denied and abandoned runs this window, not regressed, and either stored rows or is a known event collector reading zero at rest. Any other row - failing, stale, stopped, erroring, denied, regressed, or a non-event collector's unexplained zero - always carries every field named earlier in this guide, full_detail or not: a cut never hides a collector that needs a look. Pass full_detail=true for every field on every row regardless of health. collector_detail_note on the envelope names how many rows this call compacted and how to get the rest.")] public static async Task GetCollectionHealth( NpgsqlDataSource postgres, - [Description("Server name or display name.")] string? server_name = null) + [Description("Server name or display name.")] string? server_name = null, + [Description("Return every field for every collector. Default compacts a HEALTHY collector with nothing to report; failing, stale, stopped, erroring, denied or regressed collectors always keep every field.")] bool full_detail = false) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); if (error != null) return error; @@ -1022,7 +1023,10 @@ that one. */ if (rows.Count == 0) return McpHelpers.Status("unavailable", "No collection health data available."); - var result = rows.Select(r => new + var compactCount = full_detail ? 0 : rows.Count(IsCollectionHealthCompactEligible); + var result = rows.Select(r => !full_detail && IsCollectionHealthCompactEligible(r) + ? (object)CompactCollectionHealthRow(r) + : new { collector = r.CollectorName, status = r.HealthStatus, @@ -1403,6 +1407,12 @@ sentence claiming both windows were read here would be the same defect it was wr to avoid. It also says outright that rows are what a run STORED and never what the monitored engine counted, because nothing on this surface measures the second. */ output_note = CollectorHealthClassifier.OutputWindowNote, + /* #4198: what the default cut left out and how to get it back, read fresh every call rather + than a static sentence — full_detail=true changes compactCount to 0, so the note always + describes what THIS response actually did instead of what the argument nominally requested. */ + collector_detail_note = full_detail + ? "full_detail=true: every field is served for every collector below." + : $"{compactCount} of {rows.Count} collector(s) below are HEALTHY with nothing to report (no errors, denials, session/extension-missing runs or abandoned cycles this window) and are compacted to collector/status/compact/total_runs/rows_stored/avg_duration_ms/last_success. Any collector that is failing, stale, stopped, erroring, denied, regressed, or whose zero output needs a look always keeps every field. Pass full_detail=true for every field on every collector.", collectors = result, /* #3856: how old the collector half of this payload is, in whole seconds — the same field, the same name and the same semantics get_fleet_overview has carried since #3735, because it is @@ -1424,6 +1434,47 @@ column of the statement behind it follows. */ } } + /* #4198: the default-argument size cut for get_collection_health, which unlike a row-limited tool has no + row to drop — every collector on the server is one row, and a health read must never hide one that is + failing, stale, disabled or erroring by leaving it off the page. So the cut is per-FIELD instead: a + collector this predicate calls boring gets the seven-field CompactCollectionHealthRow shape instead of + the ~30-field full one. Every check here is a fact this window's aggregate already computed, not a new + read, and each one guards against exactly the "erroring collector went quiet" failure #4198 warns about: + HealthStatus alone is not enough, because Classify() bands WARNING only above a 20% error rate or a 0.5% + abandon rate, so a collector could carry a handful of errors, session-missing runs or abandoned cycles + and still read HEALTHY. RowsStored > 0 (or a known event collector reading zero at rest) rules out the + "non-event collector came back empty and needs a look" reading FormatOutputFinding would otherwise carry + — dropped here specifically because it is the one non-obvious way a HEALTHY-banded row can still be + worth a second look. HealthStatus == Healthy already implies AnyRegression is false (HealthStatus floors + regressed rows to WARNING, #3819) and PermissionDeniedCount == 0 already implies + DeniedSinceLastSuccess is false (it requires a PERMISSIONS run to exist), so neither is re-checked here + — this predicate tests only what HealthStatus does NOT already cover. YieldCount is deliberately absent: + a yield retries and is benign by design (#1805), unlike the counts checked here. */ + private static bool IsCollectionHealthCompactEligible(CollectorHealth r) => + r.HealthStatus == CollectorHealthClassifier.Healthy + && r.ErrorCount == 0 + && r.SessionMissingCount == 0 + && r.ExtensionMissingCount == 0 + && r.PermissionDeniedCount == 0 + && r.AbandonedCount == 0 + && (r.RowsStored > 0 || CollectorHealthClassifier.IsEventCollector(r.CollectorName)); + + /// The compact shape rows get by default: enough + /// to confirm the collector is fine and cheap (it ran, stored what it should, and did not cost much) + /// without the ~30 fields a row with nothing to report does not need. compact: true is the caller's + /// signal that this row was shortened — a full row never carries the property, so its mere presence is + /// unambiguous without a second lookup against full_detail. + private static object CompactCollectionHealthRow(CollectorHealth r) => new + { + collector = r.CollectorName, + status = r.HealthStatus, + compact = true, + total_runs = r.TotalRuns, + rows_stored = r.RowsStored, + avg_duration_ms = Math.Round(r.AvgDurationMs, 0), + last_success = r.LastSuccessTime?.ToString("o"), + }; + [McpServerTool(Name = "get_server_properties"), Description("Gets SQL Server instance properties: edition, version, CPU count, memory, socket/core topology, HADR, clustering, and the clock (utc_offset_minutes, time_zone_id). LATEST IS A TIME: the newest snapshot, not a window; captured_at is when it was collected, and on a stalled collector it is the only sign of staleness. time_zone_id is CURRENT_TIMEZONE_ID() (SQL Server 2022+/Azure SQL only); null means a pre-2022 engine, so only the offset in force at captured_at is known, and an instant across a DST transition from it can read an hour off. <> Gets SQL Server instance properties: edition, version, CPU count, physical memory, socket/core topology, HADR status, clustering, and the server's clock: utc_offset_minutes is the UTC offset in force when the snapshot was collected, and time_zone_id is the engine's own time-zone name (CURRENT_TIMEZONE_ID(), SQL Server 2022+ and Azure SQL only) - a null time_zone_id means a pre-2022 engine, where only the offset is known and any instant on the far side of a DST transition from the snapshot is placed an hour off by that offset. Use for capacity planning and edition-aware recommendations. LATEST IS A TIME: this reads the newest properties snapshot, not a window, and captured_at is the instant it was collected - a core count or memory figure here is what the server reported AT that stamp, and on a server whose collector has stalled the stamp is the only thing that says how stale it is.")] public static async Task GetServerProperties( NpgsqlDataSource postgres, diff --git a/Lite/Mcp/McpHealthTools.cs b/Lite/Mcp/McpHealthTools.cs index c96a82843..507836d2a 100644 --- a/Lite/Mcp/McpHealthTools.cs +++ b/Lite/Mcp/McpHealthTools.cs @@ -277,11 +277,12 @@ first row's is the range's. */ } } - [McpServerTool(Name = "get_collection_health"), Description("Per-collector 7-day health for a server. STOPPED (gate off) does not count as failing; rows_stored=0 is NOT a fault by itself — output_finding says whether it is a resting event-collector or a denial needing a grant. last_error is a sticky slot, not necessarily current: check last_error_at/denied_since_last_success. regressed_from_productive floors WARNING on an axis SEPARATE from failed_collector_count — never add them. alert_read_health uses a DIFFERENT since-restart window, not this one; zero there means none since restart, not none in 7 days. <> Shows the health status of all data collectors for a server — whether they're running successfully, failing, or stale. A collector reads STOPPED rather than FAILING when it has attempted nothing at all — no success, no error, nothing — for longer than the FAILING cutoff, despite a history of runs: that is a collector whose gate (AppliesTo) flipped off for this target rather than one that keeps running and erroring, and it does not count toward a server's failing-collector total. A collector reads EXTENSION_MISSING when every attempt in the window was skipped because a PostgreSQL extension it declares is not installed on the target - last_error names the extension, CREATE EXTENSION (plus shared_preload_libraries and a restart where the message says so) is the remedy, never a grant, and an optional extension left uninstalled is a legitimate resting state rather than a fault. That last reading holds ONLY for a collector that never produced on this target, and regressed_from_productive is the field that tells you which one you are looking at: it is true when this collector WAS storing rows and has reported a named skip - EXTENSION_MISSING, PERMISSIONS or SESSION_MISSING - on every cycle since. That is a restart or an upgrade changing what the collector can read, which is a regression rather than rest, and it is not something the band can tell you on its own: the skip bands are reached only when the whole window holds no success, so a collector that regressed inside the last day still has a fresh success, reads HEALTHY on the ordinary staleness ladder, and is floored to WARNING by this flag alone. last_productive_at is when this collector last stored anything, rows_in_prior_7d is how much it stored over this seven-day window and is served only when the flag is true (only then is that figure entirely pre-regression, because a named skip stores nothing), and regression_finding is the sentence that puts the three facts in the order they happened. get_fleet_overview counts the same rows per server as regressed_collector_count. That count is a SEPARATE AXIS from failed_collector_count and the two must never be added: one regression reads WARNING on its first day and FAILING once its success clock runs out, and is regressed on both. Check this before investigating data to ensure collectors are working properly. Each row also carries last_note/note_count: what a NON-failing run reported, e.g. an enumeration that came back with 0 items. note_count equal to total_runs means the collector has been collecting nothing all window — not a fault (the target may be legitimately empty), but the reason a HEALTHY collector can still have no data. target_has_user_databases tells those two apart: true means the target DID have user databases in the same window, so an all-window empty enumeration is worth investigating (a login that cannot enter them, an exclusion filter that matched everything, a databases scope on the collector's schedule row that admits nothing - the last a Darling-only schedule knob); false means either no user databases or no inventory to go on. Each row also carries abandoned and abandon_rate_pct: cycles the 120-second whole-server wall-clock budget gave up on, which stored nothing and advanced no watermark. Unlike a yield, which retries, an abandoned cycle is collected data you do not have. A rate above 0.5% bands the collector WARNING, so a WARNING here may have nothing to do with errors - read abandoned beside errors to attribute it. CRITICAL for reading last_error: it is a single slot carrying the newest ERROR, PERMISSIONS, or EXTENSION_MISSING message in the whole window, and a message in it is NOT evidence that the condition is current. Read last_error_at for when it happened, last_denied_at for when the newest DENIAL specifically happened, and denied_since_last_success for the derived answer - true means a denial is the collector's current state, false means every denial in the window predates a later success and the collector is reading fine now. A fault recorded before a code path changed will sit in last_error for the rest of the window while every cycle since succeeds. Do not infer a live condition from last_error alone. Total abandonment still reads FAILING through staleness; the rate exists for the partial case, where a collector abandons some cycles and succeeds often enough to stay fresh, which otherwise read HEALTHY with errors 0 indefinitely. The sweep_pressure block is the server-level roll-up: it compares the collectors' combined execution demand (average duration amortized by cadence) against the minute the fastest cadence holds. SATURATED means the collection body cannot fit inside its cadence, so relaunches are skipped and the server collects at a multiple of its configured interval while every collector still reads healthy — heaviest_collectors names where that budget goes. That verdict is the SUSTAINED answer only. peak_cycle_risk is the separate single-sweep answer: peak_cycle_ms is what the body costs on the cycle where every scheduled cadence comes due together, and BODY_OVERRUN means that one body cannot fit the budget even when the verdict reads OK — the signature of one infrequent heavy collector, which amortization hides and heaviest_collectors therefore ranks out of sight. peak_collector names it, and peak_cycle_note explains it. Read both fields: a server can be OK/BODY_OVERRUN (a schedule-shape problem, fix by moving or splitting that collector) or SATURATED/BODY_OVERRUN (a capacity problem). Every collector row carries avg_duration_ms, p95_duration_ms and max_duration_ms, because a collector's runs are not always one population. Read the three together: avg close to p95 close to max is one population, avg far below p95 is two, and p95 far below max is one pathological run. peak_cycle_ms is built from p95 (floored at the mean, so it can never read lower than a mean-based figure) for exactly that reason, and peak_collector carries peak_run_ms beside avg_duration_ms so the gap is visible. Those three still describe RUNS, and a collector that runs once per DATABASE writes one blended row, so no run-level statistic can say which database cost what. Five fan out from an enumeration on any SQL Server target (query_store, plan_correction, query_store_health, index_object_stats, database_scoped_config); separately, eleven more fan out over a per-database connection loop when the target is Azure SQL DB, and pg_autovacuum_stats always does on PostgreSQL. The per-collector `fanout` block is that answer, null for a collector that does not fan out: `items` is how wide the fan-out was, `slowest`/`slowest_ms` name the dearest database and its cost on the window's worst run, `run_ms` is that whole run, `slowest_share_pct` is slowest_ms / run_ms as a percentage — the slowest item's share of the whole pass — and `dominance` is slowest_ms * items / run_ms, the slowest item against the MEAN item: 1.0 for a perfectly even fan-out, rising with concentration. The remediation decision routes through the SHARE, not through dominance: a low share means the cost is the fan-out's WIDTH — no single database is worth chasing, and bounded parallelism is the lever — while a high share means one database dominates the pass and a per-database schedule override or a stagger is what helps. Dominance is NOT that verdict, because its ceiling is items. A fixed threshold on dominance misreads exactly the widest fan-outs, where the cost lives. Dominance only reads as dominance when it approaches a meaningful fraction of items (share = dominance / items); it stays published as the evenness ratio, for continuity. Do not try to infer any of this from p95 versus avg — on a per-database collector that ratio is usually saturated by empty-versus-productive runs and says nothing about databases. Every field named so far describes what a collector SPENT; rows_stored, runs_with_rows and productive_run_pct are what it BOUGHT, counted over the same window as total_runs and the durations, so cost and output on a row always describe the same runs. Read them together for the readings that need different actions: rows_stored above zero is expensive AND productive; rows_stored zero with denied_since_last_success false, no faulted run and no note is a collector that read and found nothing, which for one that stores a row only when an event occurs (e.g. deadlocks, blocked_process_report, pg_blocking, pg_xmin_horizon) is the correct resting state and needs no action; rows_stored zero with denied_since_last_success true is a collector that could not read and needs a grant. output_finding says which zero reading applies and is null whenever rows_stored is positive. There are five, in the order they are decided: a current denial is the grant case; errors plus session_missing above zero means that many runs recorded a fault rather than a result, so on those runs the collector was UNABLE to read and the resting-state reading is withheld for the whole window (every run faulted is nothing read at all and needs action - the case an Azure SQL DB target produced when two collectors failed on every database every sweep and were recorded SUCCESS with a sentence saying they had read and found nothing); note_count above zero means the runs themselves recorded what they found, so the finding defers to last_note instead of assuming a category - which is what keeps a DELIBERATE zero distinguishable from a collector that quietly stopped storing rows; nothing on the row explaining the zero from an event collector is that collector at rest; and the same from a collector that is NOT event-triggered - a configuration or snapshot read such as database_scoped_config, which returns a row per setting per database - is its source coming back empty on every run, which is not a resting state and the finding says needs a look. The event-collector set is a closed list on the shared classifier, so a collector left off it gets the non-reassuring sentence rather than the reassuring one. query_store on a read-replica target is the deliberate case: it is not an event collector, and every run notes an empty enumeration because Query Store on a readable secondary is excluded by design. This is deliberately NOT a band. A verdict keyed on cost-plus-zero-rows would fire on the healthy quiet install rather than the blind one. These are NOT the hourly per-collector series Darling's get_collector_cost reports as total_rows - a separate series over that caller's own days_back and across every server at once, and Darling-only, so Lite has no twin of it; the top-level output_note names both windows and disclaims that one. rows_stored is also what a run STORED, never what the monitored engine counted, so a zero cannot tell a genuinely quiet source apart from a reader capturing nothing off a busy one - nothing on this surface measures that. One block on this response is deliberately NOT on the seven-day window: alert_read_health, which counts the alerting layer's OWN store reads that failed and were swallowed. A condition check that cannot read the store logs one line and skips - correctly, because firing on absent evidence would fabricate an alert and resolving on it would fabricate a recovery - and that skip is not a collector run, so it writes no collection_log row and reaches no other health surface: only a grep of the service log found the class. It matters out of proportion to the count because the alert pass runs on a much shorter store deadline than the collection sweep, so as store latency rises the alerting layer is the FIRST thing to fail and collection is the last - during one measured episode of store lock contention the service log's error rate rose 41 to 61 per hour, every line an alerting-side read, while collector failures over the same hours FELL from 23 to 2. Read server_read_failures beside server_alert_passes for this server (a pass is one alert evaluation pass containing many reads, so more failures than passes is ordinary and the pair is NOT a ratio; a Darling sweep runs two passes for a SQL Server target, three for a PostgreSQL one, and Lite runs one, so the denominator is comparable within a host and engine but not across them), instance_read_failures for the whole service (which also covers the fleet-scoped conditions that belong to no server and so appear in no per-server count: " + AlertReadFailureCounter.FleetScopedReads + "), fleet_read_failures for how many of that service total belong to no server at all - the figure that makes a nonzero service count readable from a server whose own count is zero, because those two populations take opposite actions: a blind fleet-scoped read means the store's own self-alerts went quiet and two of them are the reads whose alerts would say the store is in trouble, while one on another server is answered by reading this same block there, last_failure_read for which condition went blind most recently, last_failure_elapsed_ms for how long that read ran before it faulted - which is the term that says whose deadline ended it ONLY WHERE THE ALERT PASS SETS ONE. The Darling service does, on every store read; Lite's alerting reads go into its local store with no command deadline at all, so on Lite this is a plain duration that says a read became slow and nothing about who ended it. Where there is a deadline: an elapsed at or about it means this process stopped waiting while the statement was still running on the store, one well below that bound means the store returned a fault, and the exception text cannot make that distinction because a client-side deadline renders as a torn stream with no SQLSTATE exactly like a dropped connection. A figure well ABOVE the bound is a third reading: the failure was not a single bounded read, which each site's clock restart between consecutive awaits makes rare and which is expected only on the shared engine sweep entry, whose awaited operation is a whole alert pass - and last_failure_at to tell a healed episode from a live one: this count never ages out of a window, so a nonzero value with a stamp from days ago is history. Every count on this block carries its own newest-failure stamp, read name and elapsed, so none of them sits undated: last_failure_* are THIS server's, fleet_last_failure_* are the fleet-scoped conditions' and are attributable by construction, and instance_last_failure_* are the newest failure anywhere in this service whatever its scope - that last trio may therefore name a read on a server you did not ask about and makes no claim about which, which is why the fleet trio is reported separately rather than inferred from it. The third population is a subtraction: instance_read_failures minus server_read_failures minus fleet_read_failures is how many failed on OTHER servers, and that one has no stamp here by design - nothing holds a newest failure for it, and a count with nothing to date or name it is the gap the three trios close, so read this block on those servers to attribute it. retried_reads is the SECOND population and on the Darling service is where most of what this block used to count now lands: reads that crossed the 10 s alert-pass deadline once and succeeded on the single retry two seconds later - the store's write bands' cost, counted rather than blinding an alert. Every one of those would once have been a swallowed failure, so read the two together: retries rising with failures at zero is a store under write pressure whose alerting is intact, and both rising is a store where a second attempt twelve seconds later still found the band on, which wants the store's write schedule looked at rather than the reader's. A read that failed twice counts in both. It carries no stamp and no read name, deliberately - a retried read did not go blind, so there is no episode to date or attribute, and a stamp would invite reading a retry as a soft failure. instance_retried_reads is the same figure across the whole service, with no fleet part because nothing records a fleet-scoped retry today. On Lite both read zero as a property of the SKU and not of a quiet store: Lite's alerting reads carry no command deadline, so there is no deadline to cross and nothing to retry. counting_since is when this process began counting, early in its own startup - these are in-memory counts and a restart takes them to zero, so a zero means \"none since counting_since\" and NOT \"none in seven days\"; check the stamp before reading the zero as reassurance. Deliberately not persisted, because what it counts is a failure to READ the store. It does NOT count alerts that failed to DELIVER and makes no claim about them - that is get_alert_history's question. And deliberately not a band, for the same reason the output figures are not: any threshold over it would have to guess how many blind reads make alerting unhealthy, and a wrong guess on this particular surface fails by saying nothing is wrong. The service block beside alert_read_health is the build-attribution read: nothing else on this surface says WHAT build is answering, so the question an operator watching an install actually asks - \"is the running service the build that carries fix X\" - is answered by version directly instead of by restart inference plus a merge-list lookup the watcher has no way to perform. version is the running build's informational version - the same read the Darling service's --version verb prints, with any SemVer build-metadata suffix stripped and the prerelease suffix KEPT, because the nightly stamp lives in the prerelease and one nightly differs from the next in nothing else; Lite has no version verb and reads the same attribute off its own app assembly, stamped from the same single declaration. counting_since remains the restart detector, and a restart is NOT a build: a crash-restart moves counting_since exactly the way an install does, which is why inferring a build from it misattributes. started_at is the SAME instant counting_since carries, deliberately - both mean \"when this process came up\", and a second clock for one fact would put two near-identical stamps on one payload whose skew a reader would have to explain away. compiled_schema_version is the schema rung this BUILD expects - the compiled constant, not a read of the store's migrated rung - so beside version it says what this build requires of the store it serves, whether or not that store has caught up. When a recent analysis pass could not read one of its fact families for this server, the payload also carries analysis_caveats (analysis_time, families_failed, families_total, entries[{family, read, outcome, message, failed_in_last_passes}]) — the analysis pass's own reads of the collectors' tables, a different layer from the collector rows; absent when the last 24 remembered passes were clean. Process memory: a service restart forgets.")] + [McpServerTool(Name = "get_collection_health"), Description("Per-collector 7-day health for a server. STOPPED (gate off) does not count as failing; rows_stored=0 is NOT a fault by itself — output_finding says whether it is a resting event-collector or a denial needing a grant. last_error is a sticky slot, not necessarily current: check last_error_at/denied_since_last_success. regressed_from_productive floors WARNING on an axis SEPARATE from failed_collector_count — never add them. alert_read_health uses a DIFFERENT since-restart window, not this one; zero there means none since restart, not none in 7 days. <> Shows the health status of all data collectors for a server — whether they're running successfully, failing, or stale. A collector reads STOPPED rather than FAILING when it has attempted nothing at all — no success, no error, nothing — for longer than the FAILING cutoff, despite a history of runs: that is a collector whose gate (AppliesTo) flipped off for this target rather than one that keeps running and erroring, and it does not count toward a server's failing-collector total. A collector reads EXTENSION_MISSING when every attempt in the window was skipped because a PostgreSQL extension it declares is not installed on the target - last_error names the extension, CREATE EXTENSION (plus shared_preload_libraries and a restart where the message says so) is the remedy, never a grant, and an optional extension left uninstalled is a legitimate resting state rather than a fault. That last reading holds ONLY for a collector that never produced on this target, and regressed_from_productive is the field that tells you which one you are looking at: it is true when this collector WAS storing rows and has reported a named skip - EXTENSION_MISSING, PERMISSIONS or SESSION_MISSING - on every cycle since. That is a restart or an upgrade changing what the collector can read, which is a regression rather than rest, and it is not something the band can tell you on its own: the skip bands are reached only when the whole window holds no success, so a collector that regressed inside the last day still has a fresh success, reads HEALTHY on the ordinary staleness ladder, and is floored to WARNING by this flag alone. last_productive_at is when this collector last stored anything, rows_in_prior_7d is how much it stored over this seven-day window and is served only when the flag is true (only then is that figure entirely pre-regression, because a named skip stores nothing), and regression_finding is the sentence that puts the three facts in the order they happened. get_fleet_overview counts the same rows per server as regressed_collector_count. That count is a SEPARATE AXIS from failed_collector_count and the two must never be added: one regression reads WARNING on its first day and FAILING once its success clock runs out, and is regressed on both. Check this before investigating data to ensure collectors are working properly. Each row also carries last_note/note_count: what a NON-failing run reported, e.g. an enumeration that came back with 0 items. note_count equal to total_runs means the collector has been collecting nothing all window — not a fault (the target may be legitimately empty), but the reason a HEALTHY collector can still have no data. target_has_user_databases tells those two apart: true means the target DID have user databases in the same window, so an all-window empty enumeration is worth investigating (a login that cannot enter them, an exclusion filter that matched everything, a databases scope on the collector's schedule row that admits nothing - the last a Darling-only schedule knob); false means either no user databases or no inventory to go on. Each row also carries abandoned and abandon_rate_pct: cycles the 120-second whole-server wall-clock budget gave up on, which stored nothing and advanced no watermark. Unlike a yield, which retries, an abandoned cycle is collected data you do not have. A rate above 0.5% bands the collector WARNING, so a WARNING here may have nothing to do with errors - read abandoned beside errors to attribute it. CRITICAL for reading last_error: it is a single slot carrying the newest ERROR, PERMISSIONS, or EXTENSION_MISSING message in the whole window, and a message in it is NOT evidence that the condition is current. Read last_error_at for when it happened, last_denied_at for when the newest DENIAL specifically happened, and denied_since_last_success for the derived answer - true means a denial is the collector's current state, false means every denial in the window predates a later success and the collector is reading fine now. A fault recorded before a code path changed will sit in last_error for the rest of the window while every cycle since succeeds. Do not infer a live condition from last_error alone. Total abandonment still reads FAILING through staleness; the rate exists for the partial case, where a collector abandons some cycles and succeeds often enough to stay fresh, which otherwise read HEALTHY with errors 0 indefinitely. The sweep_pressure block is the server-level roll-up: it compares the collectors' combined execution demand (average duration amortized by cadence) against the minute the fastest cadence holds. SATURATED means the collection body cannot fit inside its cadence, so relaunches are skipped and the server collects at a multiple of its configured interval while every collector still reads healthy — heaviest_collectors names where that budget goes. That verdict is the SUSTAINED answer only. peak_cycle_risk is the separate single-sweep answer: peak_cycle_ms is what the body costs on the cycle where every scheduled cadence comes due together, and BODY_OVERRUN means that one body cannot fit the budget even when the verdict reads OK — the signature of one infrequent heavy collector, which amortization hides and heaviest_collectors therefore ranks out of sight. peak_collector names it, and peak_cycle_note explains it. Read both fields: a server can be OK/BODY_OVERRUN (a schedule-shape problem, fix by moving or splitting that collector) or SATURATED/BODY_OVERRUN (a capacity problem). Every collector row carries avg_duration_ms, p95_duration_ms and max_duration_ms, because a collector's runs are not always one population. Read the three together: avg close to p95 close to max is one population, avg far below p95 is two, and p95 far below max is one pathological run. peak_cycle_ms is built from p95 (floored at the mean, so it can never read lower than a mean-based figure) for exactly that reason, and peak_collector carries peak_run_ms beside avg_duration_ms so the gap is visible. Those three still describe RUNS, and a collector that runs once per DATABASE writes one blended row, so no run-level statistic can say which database cost what. Five fan out from an enumeration on any SQL Server target (query_store, plan_correction, query_store_health, index_object_stats, database_scoped_config); separately, eleven more fan out over a per-database connection loop when the target is Azure SQL DB, and pg_autovacuum_stats always does on PostgreSQL. The per-collector `fanout` block is that answer, null for a collector that does not fan out: `items` is how wide the fan-out was, `slowest`/`slowest_ms` name the dearest database and its cost on the window's worst run, `run_ms` is that whole run, `slowest_share_pct` is slowest_ms / run_ms as a percentage — the slowest item's share of the whole pass — and `dominance` is slowest_ms * items / run_ms, the slowest item against the MEAN item: 1.0 for a perfectly even fan-out, rising with concentration. The remediation decision routes through the SHARE, not through dominance: a low share means the cost is the fan-out's WIDTH — no single database is worth chasing, and bounded parallelism is the lever — while a high share means one database dominates the pass and a per-database schedule override or a stagger is what helps. Dominance is NOT that verdict, because its ceiling is items. A fixed threshold on dominance misreads exactly the widest fan-outs, where the cost lives. Dominance only reads as dominance when it approaches a meaningful fraction of items (share = dominance / items); it stays published as the evenness ratio, for continuity. Do not try to infer any of this from p95 versus avg — on a per-database collector that ratio is usually saturated by empty-versus-productive runs and says nothing about databases. Every field named so far describes what a collector SPENT; rows_stored, runs_with_rows and productive_run_pct are what it BOUGHT, counted over the same window as total_runs and the durations, so cost and output on a row always describe the same runs. Read them together for the readings that need different actions: rows_stored above zero is expensive AND productive; rows_stored zero with denied_since_last_success false, no faulted run and no note is a collector that read and found nothing, which for one that stores a row only when an event occurs (e.g. deadlocks, blocked_process_report, pg_blocking, pg_xmin_horizon) is the correct resting state and needs no action; rows_stored zero with denied_since_last_success true is a collector that could not read and needs a grant. output_finding says which zero reading applies and is null whenever rows_stored is positive. There are five, in the order they are decided: a current denial is the grant case; errors plus session_missing above zero means that many runs recorded a fault rather than a result, so on those runs the collector was UNABLE to read and the resting-state reading is withheld for the whole window (every run faulted is nothing read at all and needs action - the case an Azure SQL DB target produced when two collectors failed on every database every sweep and were recorded SUCCESS with a sentence saying they had read and found nothing); note_count above zero means the runs themselves recorded what they found, so the finding defers to last_note instead of assuming a category - which is what keeps a DELIBERATE zero distinguishable from a collector that quietly stopped storing rows; nothing on the row explaining the zero from an event collector is that collector at rest; and the same from a collector that is NOT event-triggered - a configuration or snapshot read such as database_scoped_config, which returns a row per setting per database - is its source coming back empty on every run, which is not a resting state and the finding says needs a look. The event-collector set is a closed list on the shared classifier, so a collector left off it gets the non-reassuring sentence rather than the reassuring one. query_store on a read-replica target is the deliberate case: it is not an event collector, and every run notes an empty enumeration because Query Store on a readable secondary is excluded by design. This is deliberately NOT a band. A verdict keyed on cost-plus-zero-rows would fire on the healthy quiet install rather than the blind one. These are NOT the hourly per-collector series Darling's get_collector_cost reports as total_rows - a separate series over that caller's own days_back and across every server at once, and Darling-only, so Lite has no twin of it; the top-level output_note names both windows and disclaims that one. rows_stored is also what a run STORED, never what the monitored engine counted, so a zero cannot tell a genuinely quiet source apart from a reader capturing nothing off a busy one - nothing on this surface measures that. One block on this response is deliberately NOT on the seven-day window: alert_read_health, which counts the alerting layer's OWN store reads that failed and were swallowed. A condition check that cannot read the store logs one line and skips - correctly, because firing on absent evidence would fabricate an alert and resolving on it would fabricate a recovery - and that skip is not a collector run, so it writes no collection_log row and reaches no other health surface: only a grep of the service log found the class. It matters out of proportion to the count because the alert pass runs on a much shorter store deadline than the collection sweep, so as store latency rises the alerting layer is the FIRST thing to fail and collection is the last - during one measured episode of store lock contention the service log's error rate rose 41 to 61 per hour, every line an alerting-side read, while collector failures over the same hours FELL from 23 to 2. Read server_read_failures beside server_alert_passes for this server (a pass is one alert evaluation pass containing many reads, so more failures than passes is ordinary and the pair is NOT a ratio; a Darling sweep runs two passes for a SQL Server target, three for a PostgreSQL one, and Lite runs one, so the denominator is comparable within a host and engine but not across them), instance_read_failures for the whole service (which also covers the fleet-scoped conditions that belong to no server and so appear in no per-server count: " + AlertReadFailureCounter.FleetScopedReads + "), fleet_read_failures for how many of that service total belong to no server at all - the figure that makes a nonzero service count readable from a server whose own count is zero, because those two populations take opposite actions: a blind fleet-scoped read means the store's own self-alerts went quiet and two of them are the reads whose alerts would say the store is in trouble, while one on another server is answered by reading this same block there, last_failure_read for which condition went blind most recently, last_failure_elapsed_ms for how long that read ran before it faulted - which is the term that says whose deadline ended it ONLY WHERE THE ALERT PASS SETS ONE. The Darling service does, on every store read; Lite's alerting reads go into its local store with no command deadline at all, so on Lite this is a plain duration that says a read became slow and nothing about who ended it. Where there is a deadline: an elapsed at or about it means this process stopped waiting while the statement was still running on the store, one well below that bound means the store returned a fault, and the exception text cannot make that distinction because a client-side deadline renders as a torn stream with no SQLSTATE exactly like a dropped connection. A figure well ABOVE the bound is a third reading: the failure was not a single bounded read, which each site's clock restart between consecutive awaits makes rare and which is expected only on the shared engine sweep entry, whose awaited operation is a whole alert pass - and last_failure_at to tell a healed episode from a live one: this count never ages out of a window, so a nonzero value with a stamp from days ago is history. Every count on this block carries its own newest-failure stamp, read name and elapsed, so none of them sits undated: last_failure_* are THIS server's, fleet_last_failure_* are the fleet-scoped conditions' and are attributable by construction, and instance_last_failure_* are the newest failure anywhere in this service whatever its scope - that last trio may therefore name a read on a server you did not ask about and makes no claim about which, which is why the fleet trio is reported separately rather than inferred from it. The third population is a subtraction: instance_read_failures minus server_read_failures minus fleet_read_failures is how many failed on OTHER servers, and that one has no stamp here by design - nothing holds a newest failure for it, and a count with nothing to date or name it is the gap the three trios close, so read this block on those servers to attribute it. retried_reads is the SECOND population and on the Darling service is where most of what this block used to count now lands: reads that crossed the 10 s alert-pass deadline once and succeeded on the single retry two seconds later - the store's write bands' cost, counted rather than blinding an alert. Every one of those would once have been a swallowed failure, so read the two together: retries rising with failures at zero is a store under write pressure whose alerting is intact, and both rising is a store where a second attempt twelve seconds later still found the band on, which wants the store's write schedule looked at rather than the reader's. A read that failed twice counts in both. It carries no stamp and no read name, deliberately - a retried read did not go blind, so there is no episode to date or attribute, and a stamp would invite reading a retry as a soft failure. instance_retried_reads is the same figure across the whole service, with no fleet part because nothing records a fleet-scoped retry today. On Lite both read zero as a property of the SKU and not of a quiet store: Lite's alerting reads carry no command deadline, so there is no deadline to cross and nothing to retry. counting_since is when this process began counting, early in its own startup - these are in-memory counts and a restart takes them to zero, so a zero means \"none since counting_since\" and NOT \"none in seven days\"; check the stamp before reading the zero as reassurance. Deliberately not persisted, because what it counts is a failure to READ the store. It does NOT count alerts that failed to DELIVER and makes no claim about them - that is get_alert_history's question. And deliberately not a band, for the same reason the output figures are not: any threshold over it would have to guess how many blind reads make alerting unhealthy, and a wrong guess on this particular surface fails by saying nothing is wrong. The service block beside alert_read_health is the build-attribution read: nothing else on this surface says WHAT build is answering, so the question an operator watching an install actually asks - \"is the running service the build that carries fix X\" - is answered by version directly instead of by restart inference plus a merge-list lookup the watcher has no way to perform. version is the running build's informational version - the same read the Darling service's --version verb prints, with any SemVer build-metadata suffix stripped and the prerelease suffix KEPT, because the nightly stamp lives in the prerelease and one nightly differs from the next in nothing else; Lite has no version verb and reads the same attribute off its own app assembly, stamped from the same single declaration. counting_since remains the restart detector, and a restart is NOT a build: a crash-restart moves counting_since exactly the way an install does, which is why inferring a build from it misattributes. started_at is the SAME instant counting_since carries, deliberately - both mean \"when this process came up\", and a second clock for one fact would put two near-identical stamps on one payload whose skew a reader would have to explain away. compiled_schema_version is the schema rung this BUILD expects - the compiled constant, not a read of the store's migrated rung - so beside version it says what this build requires of the store it serves, whether or not that store has caught up. When a recent analysis pass could not read one of its fact families for this server, the payload also carries analysis_caveats (analysis_time, families_failed, families_total, entries[{family, read, outcome, message, failed_in_last_passes}]) — the analysis pass's own reads of the collectors' tables, a different layer from the collector rows; absent when the last 24 remembered passes were clean. Process memory: a service restart forgets. full_detail (#4198): every collector row above defaults to a compact shape - collector, status, compact, total_runs, rows_stored, avg_duration_ms, last_success - for a collector that is HEALTHY with zero errors, session_missing, extension_missing, permission_denied and abandoned runs this window, not regressed, and either stored rows or is a known event collector reading zero at rest. Any other row - failing, stale, stopped, erroring, denied, regressed, or a non-event collector's unexplained zero - always carries every field named earlier in this guide, full_detail or not: a cut never hides a collector that needs a look. Pass full_detail=true for every field on every row regardless of health. collector_detail_note on the envelope names how many rows this call compacted and how to get the rest.")] public static async Task GetCollectionHealth( LocalDataService dataService, ServerManager serverManager, - [Description("Server name or display name.")] string? server_name = null) + [Description("Server name or display name.")] string? server_name = null, + [Description("Return every field for every collector. Default compacts a HEALTHY collector with nothing to report; failing, stale, stopped, erroring, denied or regressed collectors always keep every field.")] bool full_detail = false) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); if (error != null) return error; @@ -294,7 +295,10 @@ public static async Task GetCollectionHealth( return McpHelpers.Status("unavailable", "No collection health data available."); } - var result = rows.Select(r => new + var compactCount = full_detail ? 0 : rows.Count(IsCollectionHealthCompactEligible); + var result = rows.Select(r => !full_detail && IsCollectionHealthCompactEligible(r) + ? (object)CompactCollectionHealthRow(r) + : new { collector = r.CollectorName, status = r.HealthStatus, @@ -657,6 +661,12 @@ sentence claiming both windows were read here would be the same defect it was wr to avoid. It also says outright that rows are what a run STORED and never what the monitored engine counted, because nothing on this surface measures the second. */ output_note = CollectorHealthClassifier.OutputWindowNote, + /* #4198: what the default cut left out and how to get it back, read fresh every call rather + than a static sentence — full_detail=true changes compactCount to 0, so the note always + describes what THIS response actually did instead of what the argument nominally requested. */ + collector_detail_note = full_detail + ? "full_detail=true: every field is served for every collector below." + : $"{compactCount} of {rows.Count} collector(s) below are HEALTHY with nothing to report (no errors, denials, session/extension-missing runs or abandoned cycles this window) and are compacted to collector/status/compact/total_runs/rows_stored/avg_duration_ms/last_success. Any collector that is failing, stale, stopped, erroring, denied, regressed, or whose zero output needs a look always keeps every field. Pass full_detail=true for every field on every collector.", collectors = result }, resolved.ServerId, McpHelpers.JsonOptions), McpHelpers.JsonOptions); } @@ -666,6 +676,48 @@ to avoid. It also says outright that rows are what a run STORED and never what t } } + /* #4198: the default-argument size cut for get_collection_health, which unlike a row-limited tool has no + row to drop — every collector on the server is one row, and a health read must never hide one that is + failing, stale, disabled or erroring by leaving it off the page. So the cut is per-FIELD instead: a + collector this predicate calls boring gets the seven-field CompactCollectionHealthRow shape instead of + the ~30-field full one. Every check here is a fact this window's aggregate already computed, not a new + read, and each one guards against exactly the "erroring collector went quiet" failure #4198 warns about: + HealthStatus alone is not enough, because Classify() bands WARNING only above a 20% error rate or a 0.5% + abandon rate, so a collector could carry a handful of errors, session-missing runs or abandoned cycles + and still read HEALTHY. RowsStored > 0 (or a known event collector reading zero at rest) rules out the + "non-event collector came back empty and needs a look" reading FormatOutputFinding would otherwise carry + — dropped here specifically because it is the one non-obvious way a HEALTHY-banded row can still be + worth a second look. HealthStatus == Healthy already implies AnyRegression is false (HealthStatus floors + regressed rows to WARNING, #3819) and PermissionDeniedCount == 0 already implies + DeniedSinceLastSuccess is false (it requires a PERMISSIONS run to exist), so neither is re-checked here + — this predicate tests only what HealthStatus does NOT already cover. YieldCount is deliberately absent: + a yield retries and is benign by design (#1805), unlike the counts checked here. Field-for-field + Darling's twin. */ + private static bool IsCollectionHealthCompactEligible(CollectorHealthRow r) => + r.HealthStatus == CollectorHealthClassifier.Healthy + && r.ErrorCount == 0 + && r.SessionMissingCount == 0 + && r.ExtensionMissingCount == 0 + && r.PermissionDeniedCount == 0 + && r.AbandonedCount == 0 + && (r.RowsStored > 0 || CollectorHealthClassifier.IsEventCollector(r.CollectorName)); + + /// The compact shape rows get by default: enough + /// to confirm the collector is fine and cheap (it ran, stored what it should, and did not cost much) + /// without the ~30 fields a row with nothing to report does not need. compact: true is the caller's + /// signal that this row was shortened — a full row never carries the property, so its mere presence is + /// unambiguous without a second lookup against full_detail. + private static object CompactCollectionHealthRow(CollectorHealthRow r) => new + { + collector = r.CollectorName, + status = r.HealthStatus, + compact = true, + total_runs = r.TotalRuns, + rows_stored = r.RowsStored, + avg_duration_ms = Math.Round(r.AvgDurationMs, 0), + last_success = r.LastSuccessTime?.ToString("o"), + }; + [McpServerTool(Name = "get_collection_log"), Description("Raw per-run collector log: duration split into monitored-server and store-write time, rows, status, error. NEWEST FIRST by default; min_duration_ms flips it to SLOWEST FIRST, ranked by cost. hours_back is the ask; oldest/newest_returned_collection_time bound what you actually got — under a min_duration_ms floor that is the cost-ranked sample's age, not reach. Filters apply before the cap; truncated/run_count reflect matches. status is the failure filter: an unknown value is refused, never silently empty. get_collection_health is the rollup; this is the underlying runs. <> Gets the RAW per-run collection log for a server, NEWEST FIRST by default and SLOWEST FIRST whenever min_duration_ms is supplied: one row per collector run with its total duration, the part spent querying the monitored server, the part spent writing to the local store, rows collected, status and any error. get_collection_health rolls these into a per-collector verdict; this is the underlying runs, which is what you need when the rollup says healthy and collection still looks wrong, or when you want to see what a collector was doing during a specific incident window. READ THE PAGE-SPAN FIELDS BEFORE CONCLUDING ANYTHING FROM THE ROWS. hours_back is the span you ASKED for; oldest_returned_collection_time and newest_returned_collection_time bound the page you GOT, and the row cap can make those wildly different — enough collectors writing often enough will satisfy a 24-hour request out of the last few seconds of activity. truncated says the cap bit; the two timestamps say what the page holds. THE TWO FIELDS MEAN DIFFERENT THINGS UNDER THE TWO ORDERINGS and the difference matters: under the default newest-first ordering the page is a contiguous slice of the window's tail, so oldest_returned_collection_time IS how far back this read reached; under a min_duration_ms floor the page is a cost-RANKED sample drawn from the whole window, so it tells you how old the slowest matching runs are and NOTHING about reach. Read order to know which you have. Neither field is a window floor: nothing here probes for the oldest row the window could have held. A read whose newest and oldest are seconds apart has told you nothing about the window you named, and raising limit does NOT fix it under the default ordering because the slow runs are not the recent ones — min_duration_ms is the knob for that, because supplying it ranks by duration instead of by time. All THREE filters are applied in SQL, BEFORE the cap, so truncated and run_count describe the MATCHING rows rather than the unfiltered window. order names which ordering you got, so a caller never has to infer it from the filters it sent. status IS THE FAILURE-HUNTING FILTER and the reason to reach for this tool during an incident: 'show me the failures' is the most common question asked of this log, and without it a caller pages the newest-first tail eyeballing status — which the page-span contract above explains cannot work, because the cap covers a fraction of the window and raising limit does not reach a failure that is not recent. Pass one of SUCCESS, SKIPPED, YIELDED, ABANDONED, ERROR, PERMISSIONS, EXTENSION_MISSING, SESSION_MISSING, WARNING (case-insensitive); an unknown value is REFUSED and the refusal names the whole set, rather than being applied as an equality filter that returns an empty page a caller would read as 'no failures'. get_collection_health is not this question's answer either: it carries one last_error per collector over a rollup, not the runs, their timestamps or their sequence — which is what says whether every collector failed at once or one collector failed all night. A status filter changes the page from a contiguous tail to a filtered one, so read the two page-span timestamps the same way you would under a duration floor. The filter you sent is echoed back as status_filter (not status, which on an empty result is the miss word instead), in the stored UPPERCASE spelling whatever case you sent. Two of those statuses are Darling-only in practice — EXTENSION_MISSING and WARNING are written by PostgreSQL collectors and the fleet-maintenance passes, neither of which exists on this SKU — and they are accepted rather than refused here so the vocabulary is ONE set on both products: a filter that matches nothing on this SKU is an honest empty answer, where a refusal would teach a caller the value does not exist. One precision on sql_duration_ms, which Darling's twin of this tool states at length: on the collectors that enumerate databases it is the driver's per-item stopwatch, which also wraps the per-database watermark refresh - a read against the LOCAL store, so a small part of it is not the monitored server. What does NOT apply here is the large part: this SKU never enables the deferred plan-XML or statement-text fetches, so none of the store probe or write-back that dominates Darling's figure for query_store is in this one, and there is nothing here to attribute. Five parameters carry more guidance than their 200-character cap allows; the rest of each below. collector_name: A name this server has never run returns the no-matches status rather than a quiet-window one. min_duration_ms: Applied in SQL before the cap. 0 is a real value: it admits every run and is how you ask for the whole window ranked by cost. A negative is refused. Omit for no floor and newest-first order. status: THE FAILURE FILTER — 'show me the failures' is what this log exists to answer, and paging the newest-first tail cannot reach a failure that is not recent. An unknown value is REFUSED, naming the accepted set, rather than applied as a filter that matches nothing. Omit for every status. server_name: Omitted, blank, or \"*\" reads the WHOLE FLEET (#4199) — every enabled server's runs, merged and ranked together, each row carrying server_name — matching Darling's twin exactly (Lite has no (fleet) maintenance sentinel to protect). limit: Default 200 for one server. The fleet-wide form (server_name omitted or \"*\") defaults instead to McpResponseBudget.CollectionLogFleetDefaultLimit, sized from measured bytes/row so a default fleet call stays under the shared response-size target; pass limit explicitly for more rows either way.")] public static async Task GetCollectionLog( LocalDataService dataService, From 7abb4ad6edba9db3061c7175a2288056885fb0dd Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 04:54:13 -0400 Subject: [PATCH 2/2] get_collection_health: web viewer keeps full detail, tools/list budget updated, Lite parity test DarlingWebEndpoints.cs's /api/read row now passes full_detail: true explicitly so the web viewer and Custom Views keep today's full per-collector payload; a no-rig source-scan pin holds it. Custom Views' Collection Health template also asks for full_detail so its errors/avg_duration_ms columns stay populated. McpToolsListBudget/DarlingMcpDataTools.txt and TotalCeilingBytes updated for the new full_detail parameter (+240 bytes measured). Adds the Lite.Tests twin of the Darling live test: same seeding shape, same never-compact assertions, against local DuckDB (no rig needed). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../CollectionHealthPayloadBudgetLiveTests.cs | 25 +++ .../DarlingMcpDataTools.txt | 1 + .../Darling.Tests/McpToolsListBudgetTests.cs | 6 +- .../DarlingWebEndpoints.cs | 5 +- .../wwwroot/js/view-templates.js | 4 +- .../CollectionHealthPayloadBudgetToolTests.cs | 165 ++++++++++++++++++ 6 files changed, 203 insertions(+), 3 deletions(-) create mode 100644 Lite.Tests/CollectionHealthPayloadBudgetToolTests.cs diff --git a/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs b/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs index a14685b31..7712a054a 100644 --- a/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs +++ b/Darling/Darling.Tests/CollectionHealthPayloadBudgetLiveTests.cs @@ -39,6 +39,31 @@ public sealed class CollectionHealthPayloadBudgetLiveTests private readonly ITestOutputHelper _output; public CollectionHealthPayloadBudgetLiveTests(ITestOutputHelper output) => _output = output; + /// No-rig pin: the web viewer's /api/read row must keep today's full per-collector payload, not + /// the new compact default. Source-scanned rather than rig-driven so it runs everywhere, including CI legs + /// with no Postgres rig. + [Fact] + public void WebViewerRow_PassesFullDetailTrue() + { + var path = FindRepoFile("Darling", "PerformanceMonitor.Darling.Service", "DarlingWebEndpoints.cs"); + var source = System.IO.File.ReadAllText(path); + Assert.Contains( + "[\"get_collection_health\"] = (c, pg, an) => DarlingMcpDataTools.GetCollectionHealth(pg, Server(c), full_detail: true)", + source, StringComparison.Ordinal); + } + + private static string FindRepoFile(params string[] relativeParts) + { + var dir = AppContext.BaseDirectory; + for (var i = 0; i < 8; i++) + { + var candidate = System.IO.Path.Combine(new[] { dir }.Concat(relativeParts).ToArray()); + if (System.IO.File.Exists(candidate)) return candidate; + dir = System.IO.Path.GetDirectoryName(dir) ?? dir; + } + throw new System.IO.FileNotFoundException(System.IO.Path.Combine(relativeParts)); + } + [Fact] public async Task DefaultCall_StaysUnderBudget_AndNeverCompactsANonBoringCollector() { diff --git a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpDataTools.txt b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpDataTools.txt index 98611e363..4c6e18fd8 100644 --- a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpDataTools.txt +++ b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpDataTools.txt @@ -5,6 +5,7 @@ param get_blocking_stats.hours_back 191 param get_blocking_stats.server_name 28 tool get_collection_health 583 +param get_collection_health.full_detail 191 param get_collection_health.server_name 28 tool get_collection_log 606 diff --git a/Darling/Darling.Tests/McpToolsListBudgetTests.cs b/Darling/Darling.Tests/McpToolsListBudgetTests.cs index 47a154984..47f91439e 100644 --- a/Darling/Darling.Tests/McpToolsListBudgetTests.cs +++ b/Darling/Darling.Tests/McpToolsListBudgetTests.cs @@ -99,7 +99,11 @@ so neither counts here. */ /* #4198 (lane TB): +364 bytes for get_deadlock_detail's default-preview note in its served description and its new full_graph opt-in parameter (deadlock_graph_xml, the wide field, is now a 2000-char preview by default). */ - private const int TotalCeilingBytes = 172_220; + /* #4198 (lane TK): +240 bytes for get_collection_health's new full_detail opt-in parameter (the head is + unchanged; the compaction rule lives in the tail get_tool_guide serves, not the served head). Default + calls now compact HEALTHY collectors with nothing to report, which took the default response from + 41,669 bytes (measured, every field on every collector) to 20,707 bytes, under the shared 32 KB budget. */ + private const int TotalCeilingBytes = 172_460; private const int ConvertedHeadCap = 1_000; private const int ConvertedParameterCap = 200; diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs index 5a9116b55..63a16fa60 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs @@ -2662,7 +2662,10 @@ budget cut does not silently shrink what the viewer renders. */ ["get_trace_flag_changes"] = (c, pg, an) => DarlingMcpConfigHistoryTools.GetTraceFlagChanges(pg, Server(c), Hours(c, 168), as_of: AsOf(c)), /* ── core data reads ── */ - ["get_collection_health"] = (c, pg, an) => DarlingMcpDataTools.GetCollectionHealth(pg, Server(c)), + /* #4198: full_detail=true keeps the web viewer's payload exactly what it was before the default + cut — every field on every collector row, never the compact shape a boring-healthy row gets + by default. */ + ["get_collection_health"] = (c, pg, an) => DarlingMcpDataTools.GetCollectionHealth(pg, Server(c), full_detail: true), ["get_collection_log"] = (c, pg, an) => OptionalDouble(c, "min_duration_ms", out var minDurationMs) ? DarlingMcpDataTools.GetCollectionLog(pg, Server(c), Hours(c, 24), Rows(c, "limit", 200), AsOf(c), Str(c, "collector_name"), minDurationMs) : UnparseableParam("min_duration_ms"), diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/view-templates.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/view-templates.js index 2337e1c87..8b887e9d4 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/view-templates.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/view-templates.js @@ -110,7 +110,9 @@ export const DASHBOARD_TEMPLATES = [ { title: "Collection Health", read: "get_collection_health", - params: { server }, + /* #4198: full_detail keeps every column below populated - errors, avg_duration_ms etc. are + omitted by default on a boring-healthy collector row, and this template reads them. */ + params: { server, full_detail: true }, viz: "table", span: 2, rowsKey: "collectors", diff --git a/Lite.Tests/CollectionHealthPayloadBudgetToolTests.cs b/Lite.Tests/CollectionHealthPayloadBudgetToolTests.cs new file mode 100644 index 000000000..6ca7286ca --- /dev/null +++ b/Lite.Tests/CollectionHealthPayloadBudgetToolTests.cs @@ -0,0 +1,165 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Linq; +using System.Text; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #4198, Lite twin of Darling's CollectionHealthPayloadBudgetLiveTests: get_collection_health has no row to +/// drop (every collector is one row, and a health read must never hide a failing/stale/disabled/erroring one), +/// so the default-size cut is per-field. Seeds every SQL Server catalog collector, mostly boring-healthy plus +/// four rows that must never compact, and measures McpHealthTools.GetCollectionHealth's own UTF-8 bytes. +/// +public sealed class CollectionHealthPayloadBudgetToolTests : IClassFixture, IDisposable +{ + private const string ServerName = "CollHealthBudgetSrv"; + private readonly int _serverId; + private readonly DuckDbInitializer _duckDb; + private readonly string _configDir; + private readonly ServerManager _serverManager; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public CollectionHealthPayloadBudgetToolTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + _configDir = Path.Combine(Path.GetTempPath(), "pmlite-collhealthbudget-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_configDir); + _serverManager = new ServerManager(_configDir); + var server = new ServerConnection { Id = Guid.NewGuid().ToString(), ServerName = ServerName, IsEnabled = true }; + _serverManager.AddServer(server); + _serverId = RemoteCollectorService.GetDeterministicHashCode(RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { Directory.Delete(_configDir, recursive: true); } catch (IOException) { /* temp dir */ } + } + + [Fact] + public async Task DefaultCall_StaysUnderBudget_AndNeverCompactsANonBoringCollector() + { + var service = new LocalDataService(_duckDb); + var now = DateTime.UtcNow; + var sqlServerCollectors = CollectorCatalog.All + .Where(d => d.TargetEngine == CollectorTargetEngine.SqlServer) + .Select(d => d.Name) + .ToArray(); + var neverCompact = new[] { "wait_stats", "memory_grant_stats", "query_store_health", "database_scoped_config" }; + + foreach (var name in sqlServerCollectors) + { + switch (name) + { + case "wait_stats": + for (var i = 0; i < 5; i++) + await SeedLogAsync(name, now.AddHours(-i * 6), "ERROR", 120, null, "Login failed for user 'darling_monitor'."); + break; + case "memory_grant_stats": + for (var i = 0; i < 7; i++) + await SeedLogAsync(name, now.AddHours(-i * 20 - 1), "SUCCESS", 80, 40, null); + for (var i = 0; i < 3; i++) + await SeedLogAsync(name, now.AddHours(-i * 30 - 2), "ERROR", 90, null, "Timeout expired."); + break; + case "query_store_health": + await SeedLogAsync(name, now.AddDays(-6), "PERMISSIONS", 50, null, "permission denied for function pg_read_file"); + await SeedLogAsync(name, now.AddDays(-6).AddHours(-1), "PERMISSIONS", 50, null, "permission denied for function pg_read_file"); + for (var i = 0; i < 4; i++) + await SeedLogAsync(name, now.AddHours(-i * 12), "SUCCESS", 60, 12, null); + break; + case "database_scoped_config": + for (var i = 0; i < 8; i++) + await SeedLogAsync(name, now.AddHours(-i * 18), "SUCCESS", 30, 0, null); + break; + case "deadlocks": + for (var i = 0; i < 8; i++) + await SeedLogAsync(name, now.AddHours(-i * 18), "SUCCESS", 15, 0, null); + break; + default: + for (var i = 0; i < 6; i++) + await SeedLogAsync(name, now.AddHours(-i * 24 - 1), "SUCCESS", 100 + i * 15, 50 + i * 5, null); + break; + } + } + + var defaultJson = await McpHealthTools.GetCollectionHealth(service, _serverManager, ServerName); + var defaultBytes = Encoding.UTF8.GetByteCount(defaultJson); + var fullJson = await McpHealthTools.GetCollectionHealth(service, _serverManager, ServerName, full_detail: true); + var fullBytes = Encoding.UTF8.GetByteCount(fullJson); + + Assert.True(defaultBytes <= McpResponseBudget.DefaultBytes, + $"default get_collection_health is {defaultBytes:N0} bytes, over the {McpResponseBudget.DefaultBytes:N0}-byte budget."); + Assert.True(defaultBytes < fullBytes); + + using var defaultDoc = JsonDocument.Parse(defaultJson); + var defaultRows = defaultDoc.RootElement.GetProperty("collectors").EnumerateArray() + .ToDictionary(r => r.GetProperty("collector").GetString()!, r => r); + Assert.Equal(sqlServerCollectors.Length, defaultRows.Count); + + foreach (var name in neverCompact) + { + Assert.True(defaultRows[name].TryGetProperty("errors", out _), $"{name} must keep full detail by default."); + Assert.False(defaultRows[name].TryGetProperty("compact", out _)); + } + Assert.True(defaultRows["deadlocks"].TryGetProperty("compact", out var deadlocksCompact) && deadlocksCompact.GetBoolean()); + + using var fullDoc = JsonDocument.Parse(fullJson); + Assert.All(fullDoc.RootElement.GetProperty("collectors").EnumerateArray(), + r => Assert.False(r.TryGetProperty("compact", out _))); + } + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task SeedLogAsync(string collector, DateTime collectionTimeUtc, string status, double durationMs, int? rowsCollected, string? errorMessage) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO collection_log + (log_id, server_id, server_name, collector_name, collection_time, + duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = _serverId }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = collector }); + cmd.Parameters.Add(new DuckDBParameter { Value = DateTime.SpecifyKind(collectionTimeUtc, DateTimeKind.Unspecified) }); + cmd.Parameters.Add(new DuckDBParameter { Value = durationMs }); + cmd.Parameters.Add(new DuckDBParameter { Value = status }); + cmd.Parameters.Add(new DuckDBParameter { Value = (object?)errorMessage ?? DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = (object?)rowsCollected ?? DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = durationMs * 0.8 }); + cmd.Parameters.Add(new DuckDBParameter { Value = durationMs * 0.2 }); + await cmd.ExecuteNonQueryAsync(); + } +}